Implementace prevodu HTML na PDF

Sluzba prijme adresu HTML dokumentu nebo HTML v tele requestu a vrati PDF.
Navrzena pro dokumenty o stovkach az tisicich stranek.

Rendering:
- WeasyPrint jako vychozi engine, spravne CSS Paged Media, nizka pametova
  narocnost, bez JavaScriptu
- Chromium pres Playwright pro dokumenty dokreslovane skripty
- rezim auto s detekci skriptu a fallbackem pri selhani WeasyPrintu

Velke dokumenty:
- deleni na casti na strukturalnich hranicich, rez nikdy uvnitr tabulky
  nebo odstavce
- dvoupruchodovy render obsahu se skutecnymi cisly stranek, pozice nadpisu
  se ctou z kotev hlasenych u kazde stranky
- cislovani stranek bud pres CSS countery, nebo pres cislovaci vrstvu
  nastampovanou na hotove PDF, rozmer stranky se cte z vysledneho souboru
- Chromium se restartuje po N jobech, nikdy vsak behem beziciho renderu

API:
- POST /convert synchronne, POST /jobs asynchronne se sledovanim stavu,
  stahovanim vysledku, rusenim a volitelnym callbackem
- GET /health s overenim dostupnosti obou enginu a stavem fronty
- OpenAPI respektuje prefix reverse proxy pres root_path

Bezpecnost a provoz:
- SSRF kontrola po DNS resolvu, na kazdem presmerovani a u vsech pozadavku
  prohlizece
- nedostupne assety render nezastavi, ale hlasi se v odpovedi i v logu
- fronta s omezenym poctem workeru, rozpracovane joby se pri ukonceni
  oznaci jako failed, nezmizi potichu
- strukturovane JSON logovani s job_id
- vsechny limity vypnute ve vychozim stavu

Dockerfile je dvoufazovy, obsahuje zavislosti WeasyPrintu, Chromium
a fonty s ceskou diakritikou.

Autentizace zamerne neni implementovana, zpusob predavani neni domluveny.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
JiriUhlir
2026-08-27 14:50:10 +02:00
co-authored by Claude Opus 5
parent e3cc8f418b
commit 156289fe2d
48 changed files with 4043 additions and 24 deletions
+181
View File
@@ -0,0 +1,181 @@
"""Fetching of the source document and of its assets.
Both paths go through UrlGuard. Asset failures never abort the render, they are
collected and reported back to the caller so nobody silently receives a PDF with
missing images.
"""
from __future__ import annotations
import io
import logging
from dataclasses import dataclass, field
import httpx
from ..config import Settings
from ..errors import LimitExceededError, SourceUnavailableError
from ..models import MissingAsset
from .security import UrlGuard
logger = logging.getLogger(__name__)
USER_AGENT = "html-to-pdf/1.0 (AppFactory)"
@dataclass
class FetchedDocument:
html: str
base_url: str
@dataclass
class AssetReport:
"""Collects assets that could not be loaded during a render."""
missing: list[MissingAsset] = field(default_factory=list)
_seen: set[str] = field(default_factory=set)
def add(self, url: str, reason: str) -> None:
if url in self._seen:
return
self._seen.add(url)
self.missing.append(MissingAsset(url=url, reason=reason))
logger.warning("Asset could not be loaded", extra={"asset_url": url, "reason": reason})
def fetch_document(url: str, guard: UrlGuard, settings: Settings) -> FetchedDocument:
"""Download the source HTML, validating every redirect hop."""
current = url
timeout = httpx.Timeout(settings.fetch_timeout_seconds)
with httpx.Client(follow_redirects=False, timeout=timeout, headers={"User-Agent": USER_AGENT}) as client:
for hop in range(settings.max_redirects + 1):
guard.check(current)
try:
response = client.get(current)
except httpx.HTTPError as exc:
raise SourceUnavailableError(
"Zdrojovy dokument se nepodarilo stahnout.",
{"url": current, "reason": str(exc)},
) from exc
if response.is_redirect:
location = response.headers.get("location")
if not location:
raise SourceUnavailableError(
"Zdroj vratil presmerovani bez hlavicky Location.", {"url": current}
)
current = str(response.url.join(location))
logger.info("Following redirect", extra={"hop": hop + 1, "target": current})
continue
if response.status_code >= 400:
raise SourceUnavailableError(
f"Zdrojovy dokument vratil HTTP {response.status_code}.",
{"url": current, "status_code": response.status_code},
)
_check_size(len(response.content), settings)
return FetchedDocument(html=response.text, base_url=str(response.url))
raise SourceUnavailableError(
"Prekrocen maximalni pocet presmerovani.",
{"url": url, "max_redirects": settings.max_redirects},
)
def _check_size(size: int, settings: Settings) -> None:
if settings.max_html_bytes and size > settings.max_html_bytes:
raise LimitExceededError(
"Zdrojovy dokument je vetsi nez nakonfigurovany limit MAX_HTML_BYTES.",
{"size_bytes": size, "limit_bytes": settings.max_html_bytes},
)
def build_url_fetcher(guard: UrlGuard, report: AssetReport, allow_remote: bool, timeout_seconds: int):
"""Return a WeasyPrint url_fetcher with SSRF checks and failure reporting."""
from weasyprint import default_url_fetcher
def fetcher(url: str, timeout: int = 10, ssl_context=None): # noqa: ARG001
if url.startswith("data:"):
return default_url_fetcher(url)
if not allow_remote:
report.add(url, "stahovani externich assetu je vypnute")
return _empty_asset()
try:
guard.check(url)
except Exception as exc: # noqa: BLE001 - reported, never silent
report.add(url, f"zablokovano: {exc}")
return _empty_asset()
try:
response = httpx.get(
url,
timeout=timeout_seconds,
follow_redirects=False,
headers={"User-Agent": USER_AGENT},
)
while response.is_redirect:
location = response.headers.get("location")
if not location:
raise httpx.HTTPError("presmerovani bez hlavicky Location")
target = str(response.url.join(location))
guard.check(target)
response = httpx.get(
target,
timeout=timeout_seconds,
follow_redirects=False,
headers={"User-Agent": USER_AGENT},
)
if response.status_code >= 400:
report.add(url, f"HTTP {response.status_code}")
return _empty_asset()
return {
"string": response.content,
"mime_type": response.headers.get("content-type", "").split(";")[0] or None,
"redirected_url": str(response.url),
}
except Exception as exc: # noqa: BLE001 - reported, never silent
report.add(url, str(exc))
return _empty_asset()
return fetcher
def _empty_asset() -> dict:
"""Placeholder returned instead of a failed asset so the render continues."""
return {"file_obj": io.BytesIO(b""), "mime_type": "application/octet-stream"}
@dataclass
class AssetGate:
"""Per job gateway for everything the render pulls from the network."""
guard: UrlGuard
report: AssetReport
allow_remote: bool = True
timeout_seconds: int = 10
def weasy_fetcher(self):
return build_url_fetcher(self.guard, self.report, self.allow_remote, self.timeout_seconds)
def allowed(self, url: str) -> tuple[bool, str]:
"""Decide whether a browser initiated request may proceed."""
if url.startswith("data:") or url.startswith("blob:") or url.startswith("about:"):
return True, ""
if not url.startswith(("http://", "https://")):
return False, "nepovolene schema"
if not self.allow_remote:
return False, "stahovani externich assetu je vypnute"
try:
self.guard.check(url)
except Exception as exc: # noqa: BLE001 - reported, never silent
return False, str(exc)
return True, ""