Implementace prevodu HTML na PDF
Sluzba prijme adresu HTML dokumentu nebo HTML v tele requestu a vrati PDF. Navrzena pro dokumenty o stovkach az tisicich stranek. Rendering: - WeasyPrint jako vychozi engine, spravne CSS Paged Media, nizka pametova narocnost, bez JavaScriptu - Chromium pres Playwright pro dokumenty dokreslovane skripty - rezim auto s detekci skriptu a fallbackem pri selhani WeasyPrintu Velke dokumenty: - deleni na casti na strukturalnich hranicich, rez nikdy uvnitr tabulky nebo odstavce - dvoupruchodovy render obsahu se skutecnymi cisly stranek, pozice nadpisu se ctou z kotev hlasenych u kazde stranky - cislovani stranek bud pres CSS countery, nebo pres cislovaci vrstvu nastampovanou na hotove PDF, rozmer stranky se cte z vysledneho souboru - Chromium se restartuje po N jobech, nikdy vsak behem beziciho renderu API: - POST /convert synchronne, POST /jobs asynchronne se sledovanim stavu, stahovanim vysledku, rusenim a volitelnym callbackem - GET /health s overenim dostupnosti obou enginu a stavem fronty - OpenAPI respektuje prefix reverse proxy pres root_path Bezpecnost a provoz: - SSRF kontrola po DNS resolvu, na kazdem presmerovani a u vsech pozadavku prohlizece - nedostupne assety render nezastavi, ale hlasi se v odpovedi i v logu - fronta s omezenym poctem workeru, rozpracovane joby se pri ukonceni oznaci jako failed, nezmizi potichu - strukturovane JSON logovani s job_id - vsechny limity vypnute ve vychozim stavu Dockerfile je dvoufazovy, obsahuje zavislosti WeasyPrintu, Chromium a fonty s ceskou diakritikou. Autentizace zamerne neni implementovana, zpusob predavani neni domluveny. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
e3cc8f418b
commit
156289fe2d
@@ -0,0 +1,350 @@
|
||||
"""The conversion pipeline.
|
||||
|
||||
Order of operations:
|
||||
|
||||
1. load the source HTML, validating the target address
|
||||
2. pick the engine
|
||||
3. decide how page numbers will be produced
|
||||
4. build the document, inject the page stylesheet, optionally insert the table
|
||||
of contents placeholder
|
||||
5. split into chunks
|
||||
6. first render pass, which yields the real page count and anchor positions
|
||||
7. second render pass when a table of contents needs real page numbers
|
||||
8. merge the chunks
|
||||
9. stamp the numbering overlay when CSS counters cannot be used
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Callable
|
||||
|
||||
from ..config import Settings, get_settings
|
||||
from ..errors import (
|
||||
ConversionError,
|
||||
LimitExceededError,
|
||||
RenderTimeoutError,
|
||||
UnsupportedCombinationError,
|
||||
)
|
||||
from ..models import ConvertRequest, MissingAsset
|
||||
from ..pdf import merger, paginator
|
||||
from ..pdf.chunker import split_document
|
||||
from ..pdf.document import SourceDocument
|
||||
from ..pdf.styles import build_page_css
|
||||
from .fetcher import AssetGate, AssetReport, fetch_document
|
||||
from .security import UrlGuard
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
SCRIPT_TAG = re.compile(r"<script\b[^>]*>(.*?)</script>", re.IGNORECASE | re.DOTALL)
|
||||
SCRIPT_SRC = re.compile(r"<script\b[^>]*\bsrc\s*=", re.IGNORECASE)
|
||||
|
||||
ProgressCallback = Callable[[int, int, int, int], None]
|
||||
|
||||
|
||||
@dataclass
|
||||
class ConversionResult:
|
||||
path: Path
|
||||
page_count: int
|
||||
engine_used: str
|
||||
missing_assets: list[MissingAsset] = field(default_factory=list)
|
||||
warnings: list[str] = field(default_factory=list)
|
||||
|
||||
|
||||
class ConversionPipeline:
|
||||
def __init__(self, engines: dict, settings: Settings | None = None) -> None:
|
||||
self._engines = engines
|
||||
self._settings = settings or get_settings()
|
||||
self._guard = UrlGuard(self._settings)
|
||||
|
||||
async def run(
|
||||
self,
|
||||
request: ConvertRequest,
|
||||
workdir: Path,
|
||||
progress: ProgressCallback | None = None,
|
||||
) -> ConversionResult:
|
||||
workdir.mkdir(parents=True, exist_ok=True)
|
||||
timeout = self._settings.max_render_seconds
|
||||
|
||||
coroutine = self._run_with_fallback(request, workdir, progress)
|
||||
if timeout:
|
||||
try:
|
||||
return await asyncio.wait_for(coroutine, timeout=timeout)
|
||||
except asyncio.TimeoutError as exc:
|
||||
raise RenderTimeoutError(
|
||||
"Render prekrocil nakonfigurovany limit MAX_RENDER_SECONDS.",
|
||||
{"limit_seconds": timeout},
|
||||
) from exc
|
||||
return await coroutine
|
||||
|
||||
async def _run_with_fallback(
|
||||
self, request: ConvertRequest, workdir: Path, progress: ProgressCallback | None
|
||||
) -> ConversionResult:
|
||||
raw_html, base_url = await self._load_source(request)
|
||||
requested = request.engine
|
||||
engine_name = self._select_engine(requested, raw_html)
|
||||
|
||||
try:
|
||||
return await self._execute(request, raw_html, base_url, engine_name, workdir, progress)
|
||||
except ConversionError:
|
||||
raise
|
||||
except Exception as exc:
|
||||
if requested != "auto" or engine_name != "weasyprint" or "chromium" not in self._engines:
|
||||
raise
|
||||
logger.warning(
|
||||
"WeasyPrint render failed, falling back to Chromium",
|
||||
exc_info=exc,
|
||||
extra={"engine": engine_name},
|
||||
)
|
||||
result = await self._execute(request, raw_html, base_url, "chromium", workdir, progress)
|
||||
result.warnings.append(
|
||||
"Render pres WeasyPrint selhal, dokument byl vygenerovan pres Chromium."
|
||||
)
|
||||
return result
|
||||
|
||||
def _select_engine(self, requested: str, raw_html: str) -> str:
|
||||
if requested != "auto":
|
||||
if requested not in self._engines:
|
||||
raise UnsupportedCombinationError(
|
||||
f"Engine {requested} neni v teto instanci k dispozici.",
|
||||
{"available": sorted(self._engines)},
|
||||
)
|
||||
return requested
|
||||
|
||||
if self._has_active_scripts(raw_html) and "chromium" in self._engines:
|
||||
logger.info("Auto engine selected chromium because the document contains scripts")
|
||||
return "chromium"
|
||||
|
||||
if "weasyprint" in self._engines:
|
||||
return "weasyprint"
|
||||
|
||||
return next(iter(self._engines))
|
||||
|
||||
@staticmethod
|
||||
def _has_active_scripts(raw_html: str) -> bool:
|
||||
if SCRIPT_SRC.search(raw_html):
|
||||
return True
|
||||
return any(body.strip() for body in SCRIPT_TAG.findall(raw_html))
|
||||
|
||||
async def _load_source(self, request: ConvertRequest) -> tuple:
|
||||
if request.source.html is not None:
|
||||
html = request.source.html
|
||||
limit = self._settings.max_html_bytes
|
||||
if limit and len(html.encode("utf-8")) > limit:
|
||||
raise LimitExceededError(
|
||||
"Zdrojove HTML je vetsi nez nakonfigurovany limit MAX_HTML_BYTES.",
|
||||
{"limit_bytes": limit},
|
||||
)
|
||||
return html, request.source.base_url
|
||||
|
||||
fetched = await asyncio.to_thread(
|
||||
fetch_document, request.source.url, self._guard, self._settings
|
||||
)
|
||||
return fetched.html, request.source.base_url or fetched.base_url
|
||||
|
||||
async def _execute(
|
||||
self,
|
||||
request: ConvertRequest,
|
||||
raw_html: str,
|
||||
base_url: str | None,
|
||||
engine_name: str,
|
||||
workdir: Path,
|
||||
progress: ProgressCallback | None,
|
||||
) -> ConversionResult:
|
||||
engine = self._engines[engine_name]
|
||||
report = AssetReport()
|
||||
gate = AssetGate(
|
||||
guard=self._guard,
|
||||
report=report,
|
||||
allow_remote=request.assets.allow_remote,
|
||||
timeout_seconds=request.assets.timeout_seconds,
|
||||
)
|
||||
warnings: list[str] = []
|
||||
|
||||
numbering_mode = self._numbering_mode(request, engine_name)
|
||||
|
||||
if request.toc.enabled and not getattr(engine, "supports_anchor_pages", False):
|
||||
raise UnsupportedCombinationError(
|
||||
"Generovani obsahu s cisly stranek podporuje pouze engine weasyprint.",
|
||||
{"engine": engine_name},
|
||||
)
|
||||
|
||||
document = SourceDocument(raw_html)
|
||||
document.set_base_url(base_url)
|
||||
document.append_stylesheet(
|
||||
build_page_css(
|
||||
request.page,
|
||||
request.page_numbers if numbering_mode == "css" else None,
|
||||
total_pages=None,
|
||||
outline=request.outline,
|
||||
)
|
||||
)
|
||||
|
||||
headings = document.collect_headings(request.toc.depth) if request.toc.enabled else []
|
||||
if request.toc.enabled:
|
||||
if not headings:
|
||||
warnings.append("Dokument neobsahuje zadne nadpisy, obsah nebyl vygenerovan.")
|
||||
request.toc.enabled = False
|
||||
else:
|
||||
document.insert_toc(headings, request.toc.title, pages=None)
|
||||
|
||||
chunks = self._split(document, request)
|
||||
direct_navigation = self._can_navigate_directly(request, engine_name, chunks)
|
||||
if direct_navigation:
|
||||
logger.info("Chromium will navigate to the source URL directly so its scripts run in context")
|
||||
|
||||
renders = await self._render_pass(
|
||||
engine, chunks, base_url, request, workdir, gate, progress, 1, direct_navigation
|
||||
)
|
||||
|
||||
if request.toc.enabled:
|
||||
anchor_pages = self._absolute_anchor_pages(renders)
|
||||
missing = [item.anchor for item in headings if item.anchor not in anchor_pages]
|
||||
if missing:
|
||||
logger.warning(
|
||||
"Some headings have no anchor position, their page numbers stay empty",
|
||||
extra={"missing_anchors": len(missing)},
|
||||
)
|
||||
document.remove_toc()
|
||||
document.insert_toc(headings, request.toc.title, pages=anchor_pages)
|
||||
chunks = self._split(document, request)
|
||||
renders = await self._render_pass(
|
||||
engine, chunks, base_url, request, workdir, gate, progress, 2
|
||||
)
|
||||
|
||||
if len(chunks) > 1:
|
||||
warnings.append(
|
||||
"Dokument byl rozdelen na casti, odkazy v obsahu proto nejsou klikatelne. "
|
||||
"Cisla stranek jsou spravna."
|
||||
)
|
||||
|
||||
merged_path = workdir / "merged.pdf"
|
||||
total_pages = merger.merge([item.path for item in renders], merged_path)
|
||||
self._check_page_limit(total_pages)
|
||||
|
||||
final_path = merged_path
|
||||
if request.page_numbers.enabled and numbering_mode == "overlay":
|
||||
final_path = await asyncio.to_thread(
|
||||
self._stamp_numbers, merged_path, workdir, request, total_pages
|
||||
)
|
||||
|
||||
on_job_finished = getattr(engine, "on_job_finished", None)
|
||||
if callable(on_job_finished):
|
||||
on_job_finished()
|
||||
|
||||
return ConversionResult(
|
||||
path=final_path,
|
||||
page_count=total_pages,
|
||||
engine_used=engine_name,
|
||||
missing_assets=report.missing,
|
||||
warnings=warnings,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _can_navigate_directly(request: ConvertRequest, engine_name: str, chunks: list) -> bool:
|
||||
"""Chromium renders a foreign page best when it loads the URL itself.
|
||||
|
||||
Only possible when the document is not split and needs no injected
|
||||
markup, otherwise the modified HTML has to be pushed into the page.
|
||||
"""
|
||||
return (
|
||||
engine_name == "chromium"
|
||||
and request.source.url is not None
|
||||
and not request.toc.enabled
|
||||
and len(chunks) == 1
|
||||
)
|
||||
|
||||
def _numbering_mode(self, request: ConvertRequest, engine_name: str) -> str:
|
||||
mode = request.page_numbers.mode
|
||||
if mode == "auto":
|
||||
if engine_name == "weasyprint" and not request.chunking.enabled:
|
||||
return "css"
|
||||
return "overlay"
|
||||
|
||||
if mode == "css" and engine_name != "weasyprint":
|
||||
raise UnsupportedCombinationError(
|
||||
"Rezim cislovani css funguje pouze s enginem weasyprint. Pouzijte overlay nebo auto.",
|
||||
{"engine": engine_name},
|
||||
)
|
||||
if mode == "css" and request.chunking.enabled:
|
||||
raise UnsupportedCombinationError(
|
||||
"Rezim cislovani css nelze kombinovat s chunkovanim, protoze citac stranek se v kazde "
|
||||
"casti restartuje. Vypnete chunking nebo pouzijte overlay.",
|
||||
)
|
||||
return mode
|
||||
|
||||
def _split(self, document: SourceDocument, request: ConvertRequest) -> list:
|
||||
if not request.chunking.enabled:
|
||||
return [document.to_html()]
|
||||
return split_document(document, request.chunking.pages_per_chunk)
|
||||
|
||||
async def _render_pass(
|
||||
self,
|
||||
engine,
|
||||
chunks: list,
|
||||
base_url: str | None,
|
||||
request: ConvertRequest,
|
||||
workdir: Path,
|
||||
gate: AssetGate,
|
||||
progress: ProgressCallback | None,
|
||||
pass_number: int,
|
||||
direct_navigation: bool = False,
|
||||
) -> list:
|
||||
renders = []
|
||||
pages_rendered = 0
|
||||
|
||||
for index, chunk_html in enumerate(chunks):
|
||||
output = workdir / f"pass{pass_number}-chunk{index:04d}.pdf"
|
||||
render = await engine.render_chunk(
|
||||
None if direct_navigation else chunk_html,
|
||||
base_url,
|
||||
request,
|
||||
output,
|
||||
total_pages=None,
|
||||
asset_gate=gate,
|
||||
)
|
||||
renders.append(render)
|
||||
pages_rendered += render.page_count
|
||||
|
||||
if progress is not None:
|
||||
progress(pages_rendered, index + 1, len(chunks), pass_number)
|
||||
|
||||
self._check_page_limit(pages_rendered)
|
||||
|
||||
logger.info(
|
||||
"Render pass finished",
|
||||
extra={"pass_number": pass_number, "chunks": len(chunks), "pages": pages_rendered},
|
||||
)
|
||||
return renders
|
||||
|
||||
@staticmethod
|
||||
def _absolute_anchor_pages(renders: list) -> dict:
|
||||
pages: dict = {}
|
||||
offset = 0
|
||||
for render in renders:
|
||||
for anchor, local_page in render.anchor_pages.items():
|
||||
pages.setdefault(anchor, offset + local_page + 1)
|
||||
offset += render.page_count
|
||||
return pages
|
||||
|
||||
def _check_page_limit(self, pages: int) -> None:
|
||||
if self._settings.max_pages and pages > self._settings.max_pages:
|
||||
raise LimitExceededError(
|
||||
"Dokument ma vice stranek nez nakonfigurovany limit MAX_PAGES.",
|
||||
{"pages": pages, "limit": self._settings.max_pages},
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _stamp_numbers(merged_path: Path, workdir: Path, request: ConvertRequest, total_pages: int) -> Path:
|
||||
width, height = merger.first_page_size(merged_path)
|
||||
overlay_path = workdir / "overlay.pdf"
|
||||
paginator.build_overlay(
|
||||
total_pages, width, height, request.page, request.page_numbers, overlay_path
|
||||
)
|
||||
numbered_path = workdir / "numbered.pdf"
|
||||
paginator.apply_overlay(merged_path, overlay_path, numbered_path)
|
||||
return numbered_path
|
||||
Reference in New Issue
Block a user