Implementace prevodu HTML na PDF

Sluzba prijme adresu HTML dokumentu nebo HTML v tele requestu a vrati PDF.
Navrzena pro dokumenty o stovkach az tisicich stranek.

Rendering:
- WeasyPrint jako vychozi engine, spravne CSS Paged Media, nizka pametova
  narocnost, bez JavaScriptu
- Chromium pres Playwright pro dokumenty dokreslovane skripty
- rezim auto s detekci skriptu a fallbackem pri selhani WeasyPrintu

Velke dokumenty:
- deleni na casti na strukturalnich hranicich, rez nikdy uvnitr tabulky
  nebo odstavce
- dvoupruchodovy render obsahu se skutecnymi cisly stranek, pozice nadpisu
  se ctou z kotev hlasenych u kazde stranky
- cislovani stranek bud pres CSS countery, nebo pres cislovaci vrstvu
  nastampovanou na hotove PDF, rozmer stranky se cte z vysledneho souboru
- Chromium se restartuje po N jobech, nikdy vsak behem beziciho renderu

API:
- POST /convert synchronne, POST /jobs asynchronne se sledovanim stavu,
  stahovanim vysledku, rusenim a volitelnym callbackem
- GET /health s overenim dostupnosti obou enginu a stavem fronty
- OpenAPI respektuje prefix reverse proxy pres root_path

Bezpecnost a provoz:
- SSRF kontrola po DNS resolvu, na kazdem presmerovani a u vsech pozadavku
  prohlizece
- nedostupne assety render nezastavi, ale hlasi se v odpovedi i v logu
- fronta s omezenym poctem workeru, rozpracovane joby se pri ukonceni
  oznaci jako failed, nezmizi potichu
- strukturovane JSON logovani s job_id
- vsechny limity vypnute ve vychozim stavu

Dockerfile je dvoufazovy, obsahuje zavislosti WeasyPrintu, Chromium
a fonty s ceskou diakritikou.

Autentizace zamerne neni implementovana, zpusob predavani neni domluveny.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
JiriUhlir
2026-08-27 14:50:10 +02:00
co-authored by Claude Opus 5
parent e3cc8f418b
commit 156289fe2d
48 changed files with 4043 additions and 24 deletions
View File
+142
View File
@@ -0,0 +1,142 @@
"""Splitting of a large document into renderable chunks.
A thousand page document rendered in one pass keeps the whole page tree in
memory. Splitting it on structural boundaries keeps memory flat at the cost of
having to reassemble the page numbering afterwards.
Cuts are only ever made between direct children of the container element, so a
table or a paragraph is never torn in half.
"""
from __future__ import annotations
import logging
from lxml import html as lxml_html
from .document import SourceDocument
logger = logging.getLogger(__name__)
SPLIT_TAGS = {"section", "article", "h1"}
BREAK_KEYWORDS = ("page-break-before", "break-before")
# Rough page size heuristic. The real page count is only known after rendering,
# so this only has to be good enough to keep chunks roughly even.
CHARS_PER_PAGE = 2200
IMAGE_CHAR_WEIGHT = 900
TABLE_ROW_CHAR_WEIGHT = 120
def is_split_point(element) -> bool:
if not isinstance(element.tag, str):
return False
if element.tag.lower() in SPLIT_TAGS:
return True
if element.get("data-chunk") is not None:
return True
style = (element.get("style") or "").lower()
return any(keyword in style for keyword in BREAK_KEYWORDS)
def estimate_pages(element) -> float:
text_length = len(element.text_content())
text_length += IMAGE_CHAR_WEIGHT * len(element.findall(".//img"))
text_length += TABLE_ROW_CHAR_WEIGHT * len(element.findall(".//tr"))
return max(text_length / CHARS_PER_PAGE, 0.01)
def split_document(document: SourceDocument, pages_per_chunk: int) -> list[str]:
"""Return one complete HTML document per chunk.
Falls back to a single chunk when the document has no usable split points.
"""
container = document.container
children = [child for child in container if isinstance(child.tag, str)]
split_indexes = [index for index, child in enumerate(children) if is_split_point(child)]
if len(split_indexes) < 2:
logger.info(
"Document has no usable split points, rendering in one pass",
extra={"split_points": len(split_indexes), "children": len(children)},
)
return [document.to_html()]
groups = _group_children(children, set(split_indexes), pages_per_chunk)
if len(groups) < 2:
logger.info("Document fits into a single chunk", extra={"children": len(children)})
return [document.to_html()]
prefix, suffix = _skeleton(document, container)
child_html = [lxml_html.tostring(child, encoding="unicode") for child in children]
chunks = ["".join((prefix, *(child_html[index] for index in group), suffix)) for group in groups]
logger.info(
"Document split into chunks",
extra={"chunks": len(chunks), "children": len(children), "pages_per_chunk": pages_per_chunk},
)
return chunks
def _group_children(children, split_indexes: set[int], pages_per_chunk: int) -> list[list[int]]:
groups: list[list[int]] = []
current: list[int] = []
current_pages = 0.0
for index, child in enumerate(children):
starts_chunk = index in split_indexes and current and current_pages >= pages_per_chunk
if starts_chunk:
groups.append(current)
current = []
current_pages = 0.0
current.append(index)
current_pages += estimate_pages(child)
if current:
groups.append(current)
return groups
def _skeleton(document: SourceDocument, container) -> tuple[str, str]:
"""Opening and closing markup shared by every chunk.
The whole ancestor chain is recreated with its attributes so CSS selectors
that depend on it keep matching inside a chunk.
"""
head_html = lxml_html.tostring(document.head, encoding="unicode")
chain = []
node = container
while node is not None and node is not document.tree:
chain.append(node)
node = node.getparent()
chain.reverse()
opens = [_open_tag(document.tree)]
opens.append(head_html)
closes = ["</html>"]
for node in chain:
opens.append(_open_tag(node))
closes.append(f"</{node.tag}>")
closes.reverse()
return "<!DOCTYPE html>\n" + "".join(opens), "".join(closes)
def _open_tag(element) -> str:
attributes = "".join(
f' {name}="{_escape(value)}"' for name, value in element.attrib.items()
)
return f"<{element.tag}{attributes}>"
def _escape(value: str) -> str:
return (
value.replace("&", "&amp;")
.replace('"', "&quot;")
.replace("<", "&lt;")
.replace(">", "&gt;")
)
+186
View File
@@ -0,0 +1,186 @@
"""Parsing of the source document, heading extraction and table of contents."""
from __future__ import annotations
import logging
import re
from dataclasses import dataclass
from lxml import html as lxml_html
logger = logging.getLogger(__name__)
HEADING_TAGS = ("h1", "h2", "h3", "h4", "h5", "h6")
ID_SAFE = re.compile(r"[^a-zA-Z0-9_-]+")
TOC_PLACEHOLDER = "\u2007\u2007\u2007" # figure spaces, keeps the pass 1 layout stable
@dataclass
class Heading:
level: int
text: str
anchor: str
page: int | None = None
class SourceDocument:
"""Thin wrapper over the parsed document with the operations we need."""
def __init__(self, html: str) -> None:
self.tree = lxml_html.document_fromstring(html)
self.head = self.tree.find("head")
if self.head is None:
self.head = lxml_html.Element("head")
self.tree.insert(0, self.head)
self.body = self.tree.find("body")
if self.body is None:
raise ValueError("Dokument neobsahuje element body.")
self.container = self._find_container()
self._toc_css_added = False
def _find_container(self):
"""Element whose children are the natural split points.
Many documents wrap everything in a single <main> or <div>. Splitting the
body would then produce a single chunk, so we descend through up to two
single child wrappers.
"""
container = self.body
for _ in range(2):
children = [child for child in container if isinstance(child.tag, str)]
if len(children) == 1 and len(list(children[0])) > 1:
container = children[0]
continue
break
return container
# -- styles ---------------------------------------------------------
def append_stylesheet(self, css: str) -> None:
"""Append a stylesheet as the last element of head so it wins on ties."""
style = lxml_html.Element("style")
style.set("type", "text/css")
style.text = css
self.head.append(style)
def set_base_url(self, base_url: str | None) -> None:
if not base_url or self.head.find("base") is not None:
return
base = lxml_html.Element("base")
base.set("href", base_url)
self.head.insert(0, base)
# -- headings -------------------------------------------------------
def collect_headings(self, max_depth: int) -> list[Heading]:
"""Assign ids to headings that lack one and return them in document order."""
headings: list[Heading] = []
used: set[str] = {el.get("id") for el in self.tree.iter() if el.get("id")}
for element in self.body.iter(*HEADING_TAGS):
level = int(element.tag[1])
if level > max_depth:
continue
text = " ".join(element.text_content().split())
if not text:
continue
anchor = element.get("id")
if not anchor:
anchor = self._unique_anchor(text, used)
element.set("id", anchor)
used.add(anchor)
headings.append(Heading(level=level, text=text, anchor=anchor))
return headings
@staticmethod
def _unique_anchor(text: str, used: set[str]) -> str:
base = ID_SAFE.sub("-", text.strip().lower()).strip("-") or "nadpis"
base = f"htp-{base[:60]}"
candidate = base
counter = 2
while candidate in used:
candidate = f"{base}-{counter}"
counter += 1
return candidate
# -- table of contents ----------------------------------------------
def insert_toc(self, headings: list[Heading], title: str, pages: dict[str, int] | None) -> None:
"""Insert the table of contents as the first block of the body.
Pass 1 uses a fixed width placeholder instead of the page number so the
table keeps exactly the same layout in pass 2.
"""
container = lxml_html.Element("nav")
container.set("id", "htp-toc")
container.set("class", "htp-toc")
heading = lxml_html.Element("h1")
heading.set("class", "htp-toc-title")
heading.text = title
container.append(heading)
table = lxml_html.Element("table")
table.set("class", "htp-toc-table")
tbody = lxml_html.Element("tbody")
for item in headings:
if item.anchor == "htp-toc-title":
continue
row = lxml_html.Element("tr")
row.set("class", f"htp-toc-level-{item.level}")
label_cell = lxml_html.Element("td")
label_cell.set("class", "htp-toc-label")
link = lxml_html.Element("a")
link.set("href", f"#{item.anchor}")
link.text = item.text
label_cell.append(link)
page_cell = lxml_html.Element("td")
page_cell.set("class", "htp-toc-page")
if pages is None:
page_cell.text = TOC_PLACEHOLDER
else:
page_cell.text = str(pages.get(item.anchor, "")) or TOC_PLACEHOLDER
row.append(label_cell)
row.append(page_cell)
tbody.append(row)
table.append(tbody)
container.append(table)
spacer = lxml_html.Element("div")
spacer.set("class", "htp-toc-break")
container.append(spacer)
self.container.insert(0, container)
if not self._toc_css_added:
self.append_stylesheet(TOC_CSS)
self._toc_css_added = True
def remove_toc(self) -> None:
existing = self.tree.find(".//nav[@id='htp-toc']")
if existing is not None:
existing.getparent().remove(existing)
# -- serialization ---------------------------------------------------
def to_html(self) -> str:
return "<!DOCTYPE html>\n" + lxml_html.tostring(self.tree, encoding="unicode")
TOC_CSS = """
.htp-toc { break-after: page; }
.htp-toc-table { width: 100%; border-collapse: collapse; }
.htp-toc-table td { padding: 2pt 0; vertical-align: bottom; }
.htp-toc-label a { text-decoration: none; color: inherit; }
.htp-toc-page { text-align: right; width: 4em; font-variant-numeric: tabular-nums; white-space: nowrap; }
.htp-toc-level-2 .htp-toc-label { padding-left: 1.2em; }
.htp-toc-level-3 .htp-toc-label { padding-left: 2.4em; }
.htp-toc-level-4 .htp-toc-label { padding-left: 3.6em; }
.htp-toc-level-5 .htp-toc-label { padding-left: 4.8em; }
.htp-toc-level-6 .htp-toc-label { padding-left: 6em; }
"""
+53
View File
@@ -0,0 +1,53 @@
"""Merging of chunk PDFs and reading of basic page geometry."""
from __future__ import annotations
import logging
from pathlib import Path
logger = logging.getLogger(__name__)
def merge(paths: list[Path], output_path: Path) -> int:
"""Concatenate chunk PDFs into one file.
PdfWriter.append is used on purpose, it carries over bookmarks and internal
links and shifts their page references by the running offset.
"""
from pypdf import PdfWriter
if not paths:
raise ValueError("Neni co slucovat, seznam PDF je prazdny.")
if len(paths) == 1:
paths[0].replace(output_path)
return page_count(output_path)
writer = PdfWriter()
try:
for path in paths:
writer.append(str(path))
with output_path.open("wb") as handle:
writer.write(handle)
finally:
writer.close()
total = page_count(output_path)
logger.info("Chunks merged", extra={"chunks": len(paths), "pages": total})
return total
def page_count(path: Path) -> int:
from pypdf import PdfReader
with path.open("rb") as handle:
return len(PdfReader(handle).pages)
def first_page_size(path: Path) -> tuple[float, float]:
"""Width and height of the first page in points."""
from pypdf import PdfReader
with path.open("rb") as handle:
box = PdfReader(handle).pages[0].mediabox
return float(box.width), float(box.height)
+80
View File
@@ -0,0 +1,80 @@
"""Page numbering of the merged document.
Two ways to number pages:
css
CSS counters do the work during the render. Correct and cheap, but it only
works when the whole document is rendered in one pass, because each chunk
restarts the page counter.
overlay
A transparent numbering layer with the same page size is rendered once and
merged onto the finished PDF. This is the only option for a chunked document
and for Chromium, which has no usable page counters.
"""
from __future__ import annotations
import logging
from pathlib import Path
from ..errors import EngineUnavailableError
from ..models import PageNumbers, PageSettings
from .styles import build_overlay_css
logger = logging.getLogger(__name__)
def build_overlay(
total_pages: int,
width_pt: float,
height_pt: float,
page: PageSettings,
page_numbers: PageNumbers,
output_path: Path,
) -> Path:
"""Render the numbering layer, one empty page per page of the document."""
try:
from weasyprint import HTML
except ImportError as exc:
raise EngineUnavailableError(
"Cislovani stranek vyzaduje nainstalovany WeasyPrint, ktery kresli cislovaci vrstvu.",
) from exc
css = build_overlay_css(width_pt, height_pt, page, page_numbers, total_pages)
slots = '<div class="pdf-page-slot"></div>' * total_pages
html = (
"<!DOCTYPE html><html><head><meta charset=\"utf-8\">"
f"<style>{css}</style></head><body>{slots}</body></html>"
)
HTML(string=html).write_pdf(target=str(output_path))
logger.info("Numbering overlay rendered", extra={"pages": total_pages})
return output_path
def apply_overlay(document_path: Path, overlay_path: Path, output_path: Path) -> None:
"""Stamp the numbering layer onto every page of the document."""
from pypdf import PdfReader, PdfWriter
writer = PdfWriter(clone_from=str(document_path))
try:
with overlay_path.open("rb") as handle:
overlay = PdfReader(handle)
available = len(overlay.pages)
if available < len(writer.pages):
logger.warning(
"Numbering overlay has fewer pages than the document, tail will stay unnumbered",
extra={"overlay_pages": available, "document_pages": len(writer.pages)},
)
for index, page in enumerate(writer.pages):
if index >= available:
break
page.merge_page(overlay.pages[index])
with output_path.open("wb") as target:
writer.write(target)
finally:
writer.close()
+115
View File
@@ -0,0 +1,115 @@
"""Generation of the page stylesheet injected into the source document."""
from __future__ import annotations
import re
from ..models import PageNumbers, PageSettings
NAMED_SIZE = re.compile(r"^[A-Za-z][A-Za-z0-9]*$")
MARGIN_BOXES = {
"top-left": "@top-left",
"top-center": "@top-center",
"top-right": "@top-right",
"bottom-left": "@bottom-left",
"bottom-center": "@bottom-center",
"bottom-right": "@bottom-right",
}
PLACEHOLDER = re.compile(r"(\{page\}|\{pages\})")
def page_size_value(page: PageSettings) -> str:
fmt = page.format.strip()
if NAMED_SIZE.match(fmt):
return f"{fmt} {page.orientation}"
# Explicit dimensions already carry the orientation.
return fmt
def margin_shorthand(page: PageSettings) -> str:
m = page.margin
return f"{m.top} {m.right} {m.bottom} {m.left}"
def css_content_value(fmt: str, total_pages: int | None) -> str:
"""Turn "{page} / {pages}" into a CSS content value.
When total_pages is known the total is written as a literal, otherwise the
CSS counter(pages) is used.
"""
parts: list[str] = []
for token in PLACEHOLDER.split(fmt):
if token == "{page}":
parts.append("counter(page)")
elif token == "{pages}":
parts.append(str(total_pages) if total_pages is not None else "counter(pages)")
elif token:
escaped = token.replace("\\", "\\\\").replace('"', '\\"')
parts.append(f'"{escaped}"')
return " ".join(parts) if parts else '""'
def build_page_css(
page: PageSettings,
page_numbers: PageNumbers | None = None,
total_pages: int | None = None,
outline: bool = True,
) -> str:
"""Stylesheet applied on top of the document styles."""
rules = [
"@page {",
f" size: {page_size_value(page)};",
f" margin: {margin_shorthand(page)};",
]
if page_numbers is not None and page_numbers.enabled:
box = MARGIN_BOXES[page_numbers.position]
rules.append(f" {box} {{")
rules.append(f" content: {css_content_value(page_numbers.format, total_pages)};")
rules.append(" font-size: 9pt;")
rules.append(" color: #444;")
rules.append(" }")
rules.append("}")
if not outline:
rules.append("h1, h2, h3, h4, h5, h6 { bookmark-level: none; }")
return "\n".join(rules)
def build_overlay_css(
width_pt: float,
height_pt: float,
page: PageSettings,
page_numbers: PageNumbers,
total_pages: int,
) -> str:
"""Stylesheet for the transparent numbering layer merged onto the final PDF.
The size comes from the produced PDF itself, so the overlay always matches
even when the source document declares its own @page size.
"""
box = MARGIN_BOXES[page_numbers.position]
reset = ""
if page_numbers.start_at != 1:
reset = f"body {{ counter-reset: page {page_numbers.start_at - 1}; }}\n"
return (
f"@page {{\n"
f" size: {width_pt:.2f}pt {height_pt:.2f}pt;\n"
f" margin: {margin_shorthand(page)};\n"
f" {box} {{\n"
f" content: {css_content_value(page_numbers.format, total_pages)};\n"
f" font-size: 9pt;\n"
f" color: #444;\n"
f" }}\n"
f"}}\n"
f"{reset}"
f"body {{ margin: 0; }}\n"
f".pdf-page-slot {{ height: 1px; break-after: page; }}\n"
f".pdf-page-slot:last-child {{ break-after: auto; }}\n"
)