Implementace prevodu HTML na PDF

Sluzba prijme adresu HTML dokumentu nebo HTML v tele requestu a vrati PDF.
Navrzena pro dokumenty o stovkach az tisicich stranek.

Rendering:
- WeasyPrint jako vychozi engine, spravne CSS Paged Media, nizka pametova
  narocnost, bez JavaScriptu
- Chromium pres Playwright pro dokumenty dokreslovane skripty
- rezim auto s detekci skriptu a fallbackem pri selhani WeasyPrintu

Velke dokumenty:
- deleni na casti na strukturalnich hranicich, rez nikdy uvnitr tabulky
  nebo odstavce
- dvoupruchodovy render obsahu se skutecnymi cisly stranek, pozice nadpisu
  se ctou z kotev hlasenych u kazde stranky
- cislovani stranek bud pres CSS countery, nebo pres cislovaci vrstvu
  nastampovanou na hotove PDF, rozmer stranky se cte z vysledneho souboru
- Chromium se restartuje po N jobech, nikdy vsak behem beziciho renderu

API:
- POST /convert synchronne, POST /jobs asynchronne se sledovanim stavu,
  stahovanim vysledku, rusenim a volitelnym callbackem
- GET /health s overenim dostupnosti obou enginu a stavem fronty
- OpenAPI respektuje prefix reverse proxy pres root_path

Bezpecnost a provoz:
- SSRF kontrola po DNS resolvu, na kazdem presmerovani a u vsech pozadavku
  prohlizece
- nedostupne assety render nezastavi, ale hlasi se v odpovedi i v logu
- fronta s omezenym poctem workeru, rozpracovane joby se pri ukonceni
  oznaci jako failed, nezmizi potichu
- strukturovane JSON logovani s job_id
- vsechny limity vypnute ve vychozim stavu

Dockerfile je dvoufazovy, obsahuje zavislosti WeasyPrintu, Chromium
a fonty s ceskou diakritikou.

Autentizace zamerne neni implementovana, zpusob predavani neni domluveny.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
JiriUhlir
2026-08-27 14:50:10 +02:00
co-authored by Claude Opus 5
parent e3cc8f418b
commit 156289fe2d
48 changed files with 4043 additions and 24 deletions
+186
View File
@@ -0,0 +1,186 @@
"""Parsing of the source document, heading extraction and table of contents."""
from __future__ import annotations
import logging
import re
from dataclasses import dataclass
from lxml import html as lxml_html
logger = logging.getLogger(__name__)
HEADING_TAGS = ("h1", "h2", "h3", "h4", "h5", "h6")
ID_SAFE = re.compile(r"[^a-zA-Z0-9_-]+")
TOC_PLACEHOLDER = "\u2007\u2007\u2007" # figure spaces, keeps the pass 1 layout stable
@dataclass
class Heading:
level: int
text: str
anchor: str
page: int | None = None
class SourceDocument:
"""Thin wrapper over the parsed document with the operations we need."""
def __init__(self, html: str) -> None:
self.tree = lxml_html.document_fromstring(html)
self.head = self.tree.find("head")
if self.head is None:
self.head = lxml_html.Element("head")
self.tree.insert(0, self.head)
self.body = self.tree.find("body")
if self.body is None:
raise ValueError("Dokument neobsahuje element body.")
self.container = self._find_container()
self._toc_css_added = False
def _find_container(self):
"""Element whose children are the natural split points.
Many documents wrap everything in a single <main> or <div>. Splitting the
body would then produce a single chunk, so we descend through up to two
single child wrappers.
"""
container = self.body
for _ in range(2):
children = [child for child in container if isinstance(child.tag, str)]
if len(children) == 1 and len(list(children[0])) > 1:
container = children[0]
continue
break
return container
# -- styles ---------------------------------------------------------
def append_stylesheet(self, css: str) -> None:
"""Append a stylesheet as the last element of head so it wins on ties."""
style = lxml_html.Element("style")
style.set("type", "text/css")
style.text = css
self.head.append(style)
def set_base_url(self, base_url: str | None) -> None:
if not base_url or self.head.find("base") is not None:
return
base = lxml_html.Element("base")
base.set("href", base_url)
self.head.insert(0, base)
# -- headings -------------------------------------------------------
def collect_headings(self, max_depth: int) -> list[Heading]:
"""Assign ids to headings that lack one and return them in document order."""
headings: list[Heading] = []
used: set[str] = {el.get("id") for el in self.tree.iter() if el.get("id")}
for element in self.body.iter(*HEADING_TAGS):
level = int(element.tag[1])
if level > max_depth:
continue
text = " ".join(element.text_content().split())
if not text:
continue
anchor = element.get("id")
if not anchor:
anchor = self._unique_anchor(text, used)
element.set("id", anchor)
used.add(anchor)
headings.append(Heading(level=level, text=text, anchor=anchor))
return headings
@staticmethod
def _unique_anchor(text: str, used: set[str]) -> str:
base = ID_SAFE.sub("-", text.strip().lower()).strip("-") or "nadpis"
base = f"htp-{base[:60]}"
candidate = base
counter = 2
while candidate in used:
candidate = f"{base}-{counter}"
counter += 1
return candidate
# -- table of contents ----------------------------------------------
def insert_toc(self, headings: list[Heading], title: str, pages: dict[str, int] | None) -> None:
"""Insert the table of contents as the first block of the body.
Pass 1 uses a fixed width placeholder instead of the page number so the
table keeps exactly the same layout in pass 2.
"""
container = lxml_html.Element("nav")
container.set("id", "htp-toc")
container.set("class", "htp-toc")
heading = lxml_html.Element("h1")
heading.set("class", "htp-toc-title")
heading.text = title
container.append(heading)
table = lxml_html.Element("table")
table.set("class", "htp-toc-table")
tbody = lxml_html.Element("tbody")
for item in headings:
if item.anchor == "htp-toc-title":
continue
row = lxml_html.Element("tr")
row.set("class", f"htp-toc-level-{item.level}")
label_cell = lxml_html.Element("td")
label_cell.set("class", "htp-toc-label")
link = lxml_html.Element("a")
link.set("href", f"#{item.anchor}")
link.text = item.text
label_cell.append(link)
page_cell = lxml_html.Element("td")
page_cell.set("class", "htp-toc-page")
if pages is None:
page_cell.text = TOC_PLACEHOLDER
else:
page_cell.text = str(pages.get(item.anchor, "")) or TOC_PLACEHOLDER
row.append(label_cell)
row.append(page_cell)
tbody.append(row)
table.append(tbody)
container.append(table)
spacer = lxml_html.Element("div")
spacer.set("class", "htp-toc-break")
container.append(spacer)
self.container.insert(0, container)
if not self._toc_css_added:
self.append_stylesheet(TOC_CSS)
self._toc_css_added = True
def remove_toc(self) -> None:
existing = self.tree.find(".//nav[@id='htp-toc']")
if existing is not None:
existing.getparent().remove(existing)
# -- serialization ---------------------------------------------------
def to_html(self) -> str:
return "<!DOCTYPE html>\n" + lxml_html.tostring(self.tree, encoding="unicode")
TOC_CSS = """
.htp-toc { break-after: page; }
.htp-toc-table { width: 100%; border-collapse: collapse; }
.htp-toc-table td { padding: 2pt 0; vertical-align: bottom; }
.htp-toc-label a { text-decoration: none; color: inherit; }
.htp-toc-page { text-align: right; width: 4em; font-variant-numeric: tabular-nums; white-space: nowrap; }
.htp-toc-level-2 .htp-toc-label { padding-left: 1.2em; }
.htp-toc-level-3 .htp-toc-label { padding-left: 2.4em; }
.htp-toc-level-4 .htp-toc-label { padding-left: 3.6em; }
.htp-toc-level-5 .htp-toc-label { padding-left: 4.8em; }
.htp-toc-level-6 .htp-toc-label { padding-left: 6em; }
"""