Implementace prevodu HTML na PDF
Sluzba prijme adresu HTML dokumentu nebo HTML v tele requestu a vrati PDF. Navrzena pro dokumenty o stovkach az tisicich stranek. Rendering: - WeasyPrint jako vychozi engine, spravne CSS Paged Media, nizka pametova narocnost, bez JavaScriptu - Chromium pres Playwright pro dokumenty dokreslovane skripty - rezim auto s detekci skriptu a fallbackem pri selhani WeasyPrintu Velke dokumenty: - deleni na casti na strukturalnich hranicich, rez nikdy uvnitr tabulky nebo odstavce - dvoupruchodovy render obsahu se skutecnymi cisly stranek, pozice nadpisu se ctou z kotev hlasenych u kazde stranky - cislovani stranek bud pres CSS countery, nebo pres cislovaci vrstvu nastampovanou na hotove PDF, rozmer stranky se cte z vysledneho souboru - Chromium se restartuje po N jobech, nikdy vsak behem beziciho renderu API: - POST /convert synchronne, POST /jobs asynchronne se sledovanim stavu, stahovanim vysledku, rusenim a volitelnym callbackem - GET /health s overenim dostupnosti obou enginu a stavem fronty - OpenAPI respektuje prefix reverse proxy pres root_path Bezpecnost a provoz: - SSRF kontrola po DNS resolvu, na kazdem presmerovani a u vsech pozadavku prohlizece - nedostupne assety render nezastavi, ale hlasi se v odpovedi i v logu - fronta s omezenym poctem workeru, rozpracovane joby se pri ukonceni oznaci jako failed, nezmizi potichu - strukturovane JSON logovani s job_id - vsechny limity vypnute ve vychozim stavu Dockerfile je dvoufazovy, obsahuje zavislosti WeasyPrintu, Chromium a fonty s ceskou diakritikou. Autentizace zamerne neni implementovana, zpusob predavani neni domluveny. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
e3cc8f418b
commit
156289fe2d
@@ -0,0 +1,181 @@
|
||||
"""Fetching of the source document and of its assets.
|
||||
|
||||
Both paths go through UrlGuard. Asset failures never abort the render, they are
|
||||
collected and reported back to the caller so nobody silently receives a PDF with
|
||||
missing images.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import logging
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
import httpx
|
||||
|
||||
from ..config import Settings
|
||||
from ..errors import LimitExceededError, SourceUnavailableError
|
||||
from ..models import MissingAsset
|
||||
from .security import UrlGuard
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
USER_AGENT = "html-to-pdf/1.0 (AppFactory)"
|
||||
|
||||
|
||||
@dataclass
|
||||
class FetchedDocument:
|
||||
html: str
|
||||
base_url: str
|
||||
|
||||
|
||||
@dataclass
|
||||
class AssetReport:
|
||||
"""Collects assets that could not be loaded during a render."""
|
||||
|
||||
missing: list[MissingAsset] = field(default_factory=list)
|
||||
_seen: set[str] = field(default_factory=set)
|
||||
|
||||
def add(self, url: str, reason: str) -> None:
|
||||
if url in self._seen:
|
||||
return
|
||||
self._seen.add(url)
|
||||
self.missing.append(MissingAsset(url=url, reason=reason))
|
||||
logger.warning("Asset could not be loaded", extra={"asset_url": url, "reason": reason})
|
||||
|
||||
|
||||
def fetch_document(url: str, guard: UrlGuard, settings: Settings) -> FetchedDocument:
|
||||
"""Download the source HTML, validating every redirect hop."""
|
||||
|
||||
current = url
|
||||
timeout = httpx.Timeout(settings.fetch_timeout_seconds)
|
||||
|
||||
with httpx.Client(follow_redirects=False, timeout=timeout, headers={"User-Agent": USER_AGENT}) as client:
|
||||
for hop in range(settings.max_redirects + 1):
|
||||
guard.check(current)
|
||||
try:
|
||||
response = client.get(current)
|
||||
except httpx.HTTPError as exc:
|
||||
raise SourceUnavailableError(
|
||||
"Zdrojovy dokument se nepodarilo stahnout.",
|
||||
{"url": current, "reason": str(exc)},
|
||||
) from exc
|
||||
|
||||
if response.is_redirect:
|
||||
location = response.headers.get("location")
|
||||
if not location:
|
||||
raise SourceUnavailableError(
|
||||
"Zdroj vratil presmerovani bez hlavicky Location.", {"url": current}
|
||||
)
|
||||
current = str(response.url.join(location))
|
||||
logger.info("Following redirect", extra={"hop": hop + 1, "target": current})
|
||||
continue
|
||||
|
||||
if response.status_code >= 400:
|
||||
raise SourceUnavailableError(
|
||||
f"Zdrojovy dokument vratil HTTP {response.status_code}.",
|
||||
{"url": current, "status_code": response.status_code},
|
||||
)
|
||||
|
||||
_check_size(len(response.content), settings)
|
||||
return FetchedDocument(html=response.text, base_url=str(response.url))
|
||||
|
||||
raise SourceUnavailableError(
|
||||
"Prekrocen maximalni pocet presmerovani.",
|
||||
{"url": url, "max_redirects": settings.max_redirects},
|
||||
)
|
||||
|
||||
|
||||
def _check_size(size: int, settings: Settings) -> None:
|
||||
if settings.max_html_bytes and size > settings.max_html_bytes:
|
||||
raise LimitExceededError(
|
||||
"Zdrojovy dokument je vetsi nez nakonfigurovany limit MAX_HTML_BYTES.",
|
||||
{"size_bytes": size, "limit_bytes": settings.max_html_bytes},
|
||||
)
|
||||
|
||||
|
||||
def build_url_fetcher(guard: UrlGuard, report: AssetReport, allow_remote: bool, timeout_seconds: int):
|
||||
"""Return a WeasyPrint url_fetcher with SSRF checks and failure reporting."""
|
||||
|
||||
from weasyprint import default_url_fetcher
|
||||
|
||||
def fetcher(url: str, timeout: int = 10, ssl_context=None): # noqa: ARG001
|
||||
if url.startswith("data:"):
|
||||
return default_url_fetcher(url)
|
||||
|
||||
if not allow_remote:
|
||||
report.add(url, "stahovani externich assetu je vypnute")
|
||||
return _empty_asset()
|
||||
|
||||
try:
|
||||
guard.check(url)
|
||||
except Exception as exc: # noqa: BLE001 - reported, never silent
|
||||
report.add(url, f"zablokovano: {exc}")
|
||||
return _empty_asset()
|
||||
|
||||
try:
|
||||
response = httpx.get(
|
||||
url,
|
||||
timeout=timeout_seconds,
|
||||
follow_redirects=False,
|
||||
headers={"User-Agent": USER_AGENT},
|
||||
)
|
||||
while response.is_redirect:
|
||||
location = response.headers.get("location")
|
||||
if not location:
|
||||
raise httpx.HTTPError("presmerovani bez hlavicky Location")
|
||||
target = str(response.url.join(location))
|
||||
guard.check(target)
|
||||
response = httpx.get(
|
||||
target,
|
||||
timeout=timeout_seconds,
|
||||
follow_redirects=False,
|
||||
headers={"User-Agent": USER_AGENT},
|
||||
)
|
||||
|
||||
if response.status_code >= 400:
|
||||
report.add(url, f"HTTP {response.status_code}")
|
||||
return _empty_asset()
|
||||
|
||||
return {
|
||||
"string": response.content,
|
||||
"mime_type": response.headers.get("content-type", "").split(";")[0] or None,
|
||||
"redirected_url": str(response.url),
|
||||
}
|
||||
except Exception as exc: # noqa: BLE001 - reported, never silent
|
||||
report.add(url, str(exc))
|
||||
return _empty_asset()
|
||||
|
||||
return fetcher
|
||||
|
||||
|
||||
def _empty_asset() -> dict:
|
||||
"""Placeholder returned instead of a failed asset so the render continues."""
|
||||
return {"file_obj": io.BytesIO(b""), "mime_type": "application/octet-stream"}
|
||||
|
||||
|
||||
@dataclass
|
||||
class AssetGate:
|
||||
"""Per job gateway for everything the render pulls from the network."""
|
||||
|
||||
guard: UrlGuard
|
||||
report: AssetReport
|
||||
allow_remote: bool = True
|
||||
timeout_seconds: int = 10
|
||||
|
||||
def weasy_fetcher(self):
|
||||
return build_url_fetcher(self.guard, self.report, self.allow_remote, self.timeout_seconds)
|
||||
|
||||
def allowed(self, url: str) -> tuple[bool, str]:
|
||||
"""Decide whether a browser initiated request may proceed."""
|
||||
if url.startswith("data:") or url.startswith("blob:") or url.startswith("about:"):
|
||||
return True, ""
|
||||
if not url.startswith(("http://", "https://")):
|
||||
return False, "nepovolene schema"
|
||||
if not self.allow_remote:
|
||||
return False, "stahovani externich assetu je vypnute"
|
||||
try:
|
||||
self.guard.check(url)
|
||||
except Exception as exc: # noqa: BLE001 - reported, never silent
|
||||
return False, str(exc)
|
||||
return True, ""
|
||||
Reference in New Issue
Block a user