"""Fetching of the source document and of its assets. Both paths go through UrlGuard. Asset failures never abort the render, they are collected and reported back to the caller so nobody silently receives a PDF with missing images. """ from __future__ import annotations import io import logging from dataclasses import dataclass, field import httpx from ..config import Settings from ..errors import LimitExceededError, SourceUnavailableError from ..models import MissingAsset from .security import UrlGuard logger = logging.getLogger(__name__) USER_AGENT = "html-to-pdf/1.0 (AppFactory)" @dataclass class FetchedDocument: html: str base_url: str @dataclass class AssetReport: """Collects assets that could not be loaded during a render.""" missing: list[MissingAsset] = field(default_factory=list) _seen: set[str] = field(default_factory=set) def add(self, url: str, reason: str) -> None: if url in self._seen: return self._seen.add(url) self.missing.append(MissingAsset(url=url, reason=reason)) logger.warning("Asset could not be loaded", extra={"asset_url": url, "reason": reason}) def fetch_document(url: str, guard: UrlGuard, settings: Settings) -> FetchedDocument: """Download the source HTML, validating every redirect hop.""" current = url timeout = httpx.Timeout(settings.fetch_timeout_seconds) with httpx.Client(follow_redirects=False, timeout=timeout, headers={"User-Agent": USER_AGENT}) as client: for hop in range(settings.max_redirects + 1): guard.check(current) try: response = client.get(current) except httpx.HTTPError as exc: raise SourceUnavailableError( "Zdrojovy dokument se nepodarilo stahnout.", {"url": current, "reason": str(exc)}, ) from exc if response.is_redirect: location = response.headers.get("location") if not location: raise SourceUnavailableError( "Zdroj vratil presmerovani bez hlavicky Location.", {"url": current} ) current = str(response.url.join(location)) logger.info("Following redirect", extra={"hop": hop + 1, "target": current}) continue if response.status_code >= 400: raise SourceUnavailableError( f"Zdrojovy dokument vratil HTTP {response.status_code}.", {"url": current, "status_code": response.status_code}, ) _check_size(len(response.content), settings) return FetchedDocument(html=response.text, base_url=str(response.url)) raise SourceUnavailableError( "Prekrocen maximalni pocet presmerovani.", {"url": url, "max_redirects": settings.max_redirects}, ) def _check_size(size: int, settings: Settings) -> None: if settings.max_html_bytes and size > settings.max_html_bytes: raise LimitExceededError( "Zdrojovy dokument je vetsi nez nakonfigurovany limit MAX_HTML_BYTES.", {"size_bytes": size, "limit_bytes": settings.max_html_bytes}, ) def build_url_fetcher(guard: UrlGuard, report: AssetReport, allow_remote: bool, timeout_seconds: int): """Return a WeasyPrint url_fetcher with SSRF checks and failure reporting.""" from weasyprint import default_url_fetcher def fetcher(url: str, timeout: int = 10, ssl_context=None): # noqa: ARG001 if url.startswith("data:"): return default_url_fetcher(url) if not allow_remote: report.add(url, "stahovani externich assetu je vypnute") return _empty_asset() try: guard.check(url) except Exception as exc: # noqa: BLE001 - reported, never silent report.add(url, f"zablokovano: {exc}") return _empty_asset() try: response = httpx.get( url, timeout=timeout_seconds, follow_redirects=False, headers={"User-Agent": USER_AGENT}, ) while response.is_redirect: location = response.headers.get("location") if not location: raise httpx.HTTPError("presmerovani bez hlavicky Location") target = str(response.url.join(location)) guard.check(target) response = httpx.get( target, timeout=timeout_seconds, follow_redirects=False, headers={"User-Agent": USER_AGENT}, ) if response.status_code >= 400: report.add(url, f"HTTP {response.status_code}") return _empty_asset() return { "string": response.content, "mime_type": response.headers.get("content-type", "").split(";")[0] or None, "redirected_url": str(response.url), } except Exception as exc: # noqa: BLE001 - reported, never silent report.add(url, str(exc)) return _empty_asset() return fetcher def _empty_asset() -> dict: """Placeholder returned instead of a failed asset so the render continues.""" return {"file_obj": io.BytesIO(b""), "mime_type": "application/octet-stream"} @dataclass class AssetGate: """Per job gateway for everything the render pulls from the network.""" guard: UrlGuard report: AssetReport allow_remote: bool = True timeout_seconds: int = 10 def weasy_fetcher(self): return build_url_fetcher(self.guard, self.report, self.allow_remote, self.timeout_seconds) def allowed(self, url: str) -> tuple[bool, str]: """Decide whether a browser initiated request may proceed.""" if url.startswith("data:") or url.startswith("blob:") or url.startswith("about:"): return True, "" if not url.startswith(("http://", "https://")): return False, "nepovolene schema" if not self.allow_remote: return False, "stahovani externich assetu je vypnute" try: self.guard.check(url) except Exception as exc: # noqa: BLE001 - reported, never silent return False, str(exc) return True, ""