Sluzba prijme adresu HTML dokumentu nebo HTML v tele requestu a vrati PDF. Navrzena pro dokumenty o stovkach az tisicich stranek. Rendering: - WeasyPrint jako vychozi engine, spravne CSS Paged Media, nizka pametova narocnost, bez JavaScriptu - Chromium pres Playwright pro dokumenty dokreslovane skripty - rezim auto s detekci skriptu a fallbackem pri selhani WeasyPrintu Velke dokumenty: - deleni na casti na strukturalnich hranicich, rez nikdy uvnitr tabulky nebo odstavce - dvoupruchodovy render obsahu se skutecnymi cisly stranek, pozice nadpisu se ctou z kotev hlasenych u kazde stranky - cislovani stranek bud pres CSS countery, nebo pres cislovaci vrstvu nastampovanou na hotove PDF, rozmer stranky se cte z vysledneho souboru - Chromium se restartuje po N jobech, nikdy vsak behem beziciho renderu API: - POST /convert synchronne, POST /jobs asynchronne se sledovanim stavu, stahovanim vysledku, rusenim a volitelnym callbackem - GET /health s overenim dostupnosti obou enginu a stavem fronty - OpenAPI respektuje prefix reverse proxy pres root_path Bezpecnost a provoz: - SSRF kontrola po DNS resolvu, na kazdem presmerovani a u vsech pozadavku prohlizece - nedostupne assety render nezastavi, ale hlasi se v odpovedi i v logu - fronta s omezenym poctem workeru, rozpracovane joby se pri ukonceni oznaci jako failed, nezmizi potichu - strukturovane JSON logovani s job_id - vsechny limity vypnute ve vychozim stavu Dockerfile je dvoufazovy, obsahuje zavislosti WeasyPrintu, Chromium a fonty s ceskou diakritikou. Autentizace zamerne neni implementovana, zpusob predavani neni domluveny. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
182 lines
6.3 KiB
Python
182 lines
6.3 KiB
Python
"""Fetching of the source document and of its assets.
|
|
|
|
Both paths go through UrlGuard. Asset failures never abort the render, they are
|
|
collected and reported back to the caller so nobody silently receives a PDF with
|
|
missing images.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import io
|
|
import logging
|
|
from dataclasses import dataclass, field
|
|
|
|
import httpx
|
|
|
|
from ..config import Settings
|
|
from ..errors import LimitExceededError, SourceUnavailableError
|
|
from ..models import MissingAsset
|
|
from .security import UrlGuard
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
USER_AGENT = "html-to-pdf/1.0 (AppFactory)"
|
|
|
|
|
|
@dataclass
|
|
class FetchedDocument:
|
|
html: str
|
|
base_url: str
|
|
|
|
|
|
@dataclass
|
|
class AssetReport:
|
|
"""Collects assets that could not be loaded during a render."""
|
|
|
|
missing: list[MissingAsset] = field(default_factory=list)
|
|
_seen: set[str] = field(default_factory=set)
|
|
|
|
def add(self, url: str, reason: str) -> None:
|
|
if url in self._seen:
|
|
return
|
|
self._seen.add(url)
|
|
self.missing.append(MissingAsset(url=url, reason=reason))
|
|
logger.warning("Asset could not be loaded", extra={"asset_url": url, "reason": reason})
|
|
|
|
|
|
def fetch_document(url: str, guard: UrlGuard, settings: Settings) -> FetchedDocument:
|
|
"""Download the source HTML, validating every redirect hop."""
|
|
|
|
current = url
|
|
timeout = httpx.Timeout(settings.fetch_timeout_seconds)
|
|
|
|
with httpx.Client(follow_redirects=False, timeout=timeout, headers={"User-Agent": USER_AGENT}) as client:
|
|
for hop in range(settings.max_redirects + 1):
|
|
guard.check(current)
|
|
try:
|
|
response = client.get(current)
|
|
except httpx.HTTPError as exc:
|
|
raise SourceUnavailableError(
|
|
"Zdrojovy dokument se nepodarilo stahnout.",
|
|
{"url": current, "reason": str(exc)},
|
|
) from exc
|
|
|
|
if response.is_redirect:
|
|
location = response.headers.get("location")
|
|
if not location:
|
|
raise SourceUnavailableError(
|
|
"Zdroj vratil presmerovani bez hlavicky Location.", {"url": current}
|
|
)
|
|
current = str(response.url.join(location))
|
|
logger.info("Following redirect", extra={"hop": hop + 1, "target": current})
|
|
continue
|
|
|
|
if response.status_code >= 400:
|
|
raise SourceUnavailableError(
|
|
f"Zdrojovy dokument vratil HTTP {response.status_code}.",
|
|
{"url": current, "status_code": response.status_code},
|
|
)
|
|
|
|
_check_size(len(response.content), settings)
|
|
return FetchedDocument(html=response.text, base_url=str(response.url))
|
|
|
|
raise SourceUnavailableError(
|
|
"Prekrocen maximalni pocet presmerovani.",
|
|
{"url": url, "max_redirects": settings.max_redirects},
|
|
)
|
|
|
|
|
|
def _check_size(size: int, settings: Settings) -> None:
|
|
if settings.max_html_bytes and size > settings.max_html_bytes:
|
|
raise LimitExceededError(
|
|
"Zdrojovy dokument je vetsi nez nakonfigurovany limit MAX_HTML_BYTES.",
|
|
{"size_bytes": size, "limit_bytes": settings.max_html_bytes},
|
|
)
|
|
|
|
|
|
def build_url_fetcher(guard: UrlGuard, report: AssetReport, allow_remote: bool, timeout_seconds: int):
|
|
"""Return a WeasyPrint url_fetcher with SSRF checks and failure reporting."""
|
|
|
|
from weasyprint import default_url_fetcher
|
|
|
|
def fetcher(url: str, timeout: int = 10, ssl_context=None): # noqa: ARG001
|
|
if url.startswith("data:"):
|
|
return default_url_fetcher(url)
|
|
|
|
if not allow_remote:
|
|
report.add(url, "stahovani externich assetu je vypnute")
|
|
return _empty_asset()
|
|
|
|
try:
|
|
guard.check(url)
|
|
except Exception as exc: # noqa: BLE001 - reported, never silent
|
|
report.add(url, f"zablokovano: {exc}")
|
|
return _empty_asset()
|
|
|
|
try:
|
|
response = httpx.get(
|
|
url,
|
|
timeout=timeout_seconds,
|
|
follow_redirects=False,
|
|
headers={"User-Agent": USER_AGENT},
|
|
)
|
|
while response.is_redirect:
|
|
location = response.headers.get("location")
|
|
if not location:
|
|
raise httpx.HTTPError("presmerovani bez hlavicky Location")
|
|
target = str(response.url.join(location))
|
|
guard.check(target)
|
|
response = httpx.get(
|
|
target,
|
|
timeout=timeout_seconds,
|
|
follow_redirects=False,
|
|
headers={"User-Agent": USER_AGENT},
|
|
)
|
|
|
|
if response.status_code >= 400:
|
|
report.add(url, f"HTTP {response.status_code}")
|
|
return _empty_asset()
|
|
|
|
return {
|
|
"string": response.content,
|
|
"mime_type": response.headers.get("content-type", "").split(";")[0] or None,
|
|
"redirected_url": str(response.url),
|
|
}
|
|
except Exception as exc: # noqa: BLE001 - reported, never silent
|
|
report.add(url, str(exc))
|
|
return _empty_asset()
|
|
|
|
return fetcher
|
|
|
|
|
|
def _empty_asset() -> dict:
|
|
"""Placeholder returned instead of a failed asset so the render continues."""
|
|
return {"file_obj": io.BytesIO(b""), "mime_type": "application/octet-stream"}
|
|
|
|
|
|
@dataclass
|
|
class AssetGate:
|
|
"""Per job gateway for everything the render pulls from the network."""
|
|
|
|
guard: UrlGuard
|
|
report: AssetReport
|
|
allow_remote: bool = True
|
|
timeout_seconds: int = 10
|
|
|
|
def weasy_fetcher(self):
|
|
return build_url_fetcher(self.guard, self.report, self.allow_remote, self.timeout_seconds)
|
|
|
|
def allowed(self, url: str) -> tuple[bool, str]:
|
|
"""Decide whether a browser initiated request may proceed."""
|
|
if url.startswith("data:") or url.startswith("blob:") or url.startswith("about:"):
|
|
return True, ""
|
|
if not url.startswith(("http://", "https://")):
|
|
return False, "nepovolene schema"
|
|
if not self.allow_remote:
|
|
return False, "stahovani externich assetu je vypnute"
|
|
try:
|
|
self.guard.check(url)
|
|
except Exception as exc: # noqa: BLE001 - reported, never silent
|
|
return False, str(exc)
|
|
return True, ""
|