Sluzba prijme adresu HTML dokumentu nebo HTML v tele requestu a vrati PDF. Navrzena pro dokumenty o stovkach az tisicich stranek. Rendering: - WeasyPrint jako vychozi engine, spravne CSS Paged Media, nizka pametova narocnost, bez JavaScriptu - Chromium pres Playwright pro dokumenty dokreslovane skripty - rezim auto s detekci skriptu a fallbackem pri selhani WeasyPrintu Velke dokumenty: - deleni na casti na strukturalnich hranicich, rez nikdy uvnitr tabulky nebo odstavce - dvoupruchodovy render obsahu se skutecnymi cisly stranek, pozice nadpisu se ctou z kotev hlasenych u kazde stranky - cislovani stranek bud pres CSS countery, nebo pres cislovaci vrstvu nastampovanou na hotove PDF, rozmer stranky se cte z vysledneho souboru - Chromium se restartuje po N jobech, nikdy vsak behem beziciho renderu API: - POST /convert synchronne, POST /jobs asynchronne se sledovanim stavu, stahovanim vysledku, rusenim a volitelnym callbackem - GET /health s overenim dostupnosti obou enginu a stavem fronty - OpenAPI respektuje prefix reverse proxy pres root_path Bezpecnost a provoz: - SSRF kontrola po DNS resolvu, na kazdem presmerovani a u vsech pozadavku prohlizece - nedostupne assety render nezastavi, ale hlasi se v odpovedi i v logu - fronta s omezenym poctem workeru, rozpracovane joby se pri ukonceni oznaci jako failed, nezmizi potichu - strukturovane JSON logovani s job_id - vsechny limity vypnute ve vychozim stavu Dockerfile je dvoufazovy, obsahuje zavislosti WeasyPrintu, Chromium a fonty s ceskou diakritikou. Autentizace zamerne neni implementovana, zpusob predavani neni domluveny. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
187 lines
6.4 KiB
Python
187 lines
6.4 KiB
Python
"""Parsing of the source document, heading extraction and table of contents."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import re
|
|
from dataclasses import dataclass
|
|
|
|
from lxml import html as lxml_html
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
HEADING_TAGS = ("h1", "h2", "h3", "h4", "h5", "h6")
|
|
ID_SAFE = re.compile(r"[^a-zA-Z0-9_-]+")
|
|
|
|
TOC_PLACEHOLDER = "\u2007\u2007\u2007" # figure spaces, keeps the pass 1 layout stable
|
|
|
|
|
|
@dataclass
|
|
class Heading:
|
|
level: int
|
|
text: str
|
|
anchor: str
|
|
page: int | None = None
|
|
|
|
|
|
class SourceDocument:
|
|
"""Thin wrapper over the parsed document with the operations we need."""
|
|
|
|
def __init__(self, html: str) -> None:
|
|
self.tree = lxml_html.document_fromstring(html)
|
|
self.head = self.tree.find("head")
|
|
if self.head is None:
|
|
self.head = lxml_html.Element("head")
|
|
self.tree.insert(0, self.head)
|
|
self.body = self.tree.find("body")
|
|
if self.body is None:
|
|
raise ValueError("Dokument neobsahuje element body.")
|
|
self.container = self._find_container()
|
|
self._toc_css_added = False
|
|
|
|
def _find_container(self):
|
|
"""Element whose children are the natural split points.
|
|
|
|
Many documents wrap everything in a single <main> or <div>. Splitting the
|
|
body would then produce a single chunk, so we descend through up to two
|
|
single child wrappers.
|
|
"""
|
|
container = self.body
|
|
for _ in range(2):
|
|
children = [child for child in container if isinstance(child.tag, str)]
|
|
if len(children) == 1 and len(list(children[0])) > 1:
|
|
container = children[0]
|
|
continue
|
|
break
|
|
return container
|
|
|
|
# -- styles ---------------------------------------------------------
|
|
def append_stylesheet(self, css: str) -> None:
|
|
"""Append a stylesheet as the last element of head so it wins on ties."""
|
|
style = lxml_html.Element("style")
|
|
style.set("type", "text/css")
|
|
style.text = css
|
|
self.head.append(style)
|
|
|
|
def set_base_url(self, base_url: str | None) -> None:
|
|
if not base_url or self.head.find("base") is not None:
|
|
return
|
|
base = lxml_html.Element("base")
|
|
base.set("href", base_url)
|
|
self.head.insert(0, base)
|
|
|
|
# -- headings -------------------------------------------------------
|
|
def collect_headings(self, max_depth: int) -> list[Heading]:
|
|
"""Assign ids to headings that lack one and return them in document order."""
|
|
headings: list[Heading] = []
|
|
used: set[str] = {el.get("id") for el in self.tree.iter() if el.get("id")}
|
|
|
|
for element in self.body.iter(*HEADING_TAGS):
|
|
level = int(element.tag[1])
|
|
if level > max_depth:
|
|
continue
|
|
|
|
text = " ".join(element.text_content().split())
|
|
if not text:
|
|
continue
|
|
|
|
anchor = element.get("id")
|
|
if not anchor:
|
|
anchor = self._unique_anchor(text, used)
|
|
element.set("id", anchor)
|
|
used.add(anchor)
|
|
|
|
headings.append(Heading(level=level, text=text, anchor=anchor))
|
|
|
|
return headings
|
|
|
|
@staticmethod
|
|
def _unique_anchor(text: str, used: set[str]) -> str:
|
|
base = ID_SAFE.sub("-", text.strip().lower()).strip("-") or "nadpis"
|
|
base = f"htp-{base[:60]}"
|
|
candidate = base
|
|
counter = 2
|
|
while candidate in used:
|
|
candidate = f"{base}-{counter}"
|
|
counter += 1
|
|
return candidate
|
|
|
|
# -- table of contents ----------------------------------------------
|
|
def insert_toc(self, headings: list[Heading], title: str, pages: dict[str, int] | None) -> None:
|
|
"""Insert the table of contents as the first block of the body.
|
|
|
|
Pass 1 uses a fixed width placeholder instead of the page number so the
|
|
table keeps exactly the same layout in pass 2.
|
|
"""
|
|
container = lxml_html.Element("nav")
|
|
container.set("id", "htp-toc")
|
|
container.set("class", "htp-toc")
|
|
|
|
heading = lxml_html.Element("h1")
|
|
heading.set("class", "htp-toc-title")
|
|
heading.text = title
|
|
container.append(heading)
|
|
|
|
table = lxml_html.Element("table")
|
|
table.set("class", "htp-toc-table")
|
|
tbody = lxml_html.Element("tbody")
|
|
|
|
for item in headings:
|
|
if item.anchor == "htp-toc-title":
|
|
continue
|
|
row = lxml_html.Element("tr")
|
|
row.set("class", f"htp-toc-level-{item.level}")
|
|
|
|
label_cell = lxml_html.Element("td")
|
|
label_cell.set("class", "htp-toc-label")
|
|
link = lxml_html.Element("a")
|
|
link.set("href", f"#{item.anchor}")
|
|
link.text = item.text
|
|
label_cell.append(link)
|
|
|
|
page_cell = lxml_html.Element("td")
|
|
page_cell.set("class", "htp-toc-page")
|
|
if pages is None:
|
|
page_cell.text = TOC_PLACEHOLDER
|
|
else:
|
|
page_cell.text = str(pages.get(item.anchor, "")) or TOC_PLACEHOLDER
|
|
|
|
row.append(label_cell)
|
|
row.append(page_cell)
|
|
tbody.append(row)
|
|
|
|
table.append(tbody)
|
|
container.append(table)
|
|
|
|
spacer = lxml_html.Element("div")
|
|
spacer.set("class", "htp-toc-break")
|
|
container.append(spacer)
|
|
|
|
self.container.insert(0, container)
|
|
if not self._toc_css_added:
|
|
self.append_stylesheet(TOC_CSS)
|
|
self._toc_css_added = True
|
|
|
|
def remove_toc(self) -> None:
|
|
existing = self.tree.find(".//nav[@id='htp-toc']")
|
|
if existing is not None:
|
|
existing.getparent().remove(existing)
|
|
|
|
# -- serialization ---------------------------------------------------
|
|
def to_html(self) -> str:
|
|
return "<!DOCTYPE html>\n" + lxml_html.tostring(self.tree, encoding="unicode")
|
|
|
|
|
|
TOC_CSS = """
|
|
.htp-toc { break-after: page; }
|
|
.htp-toc-table { width: 100%; border-collapse: collapse; }
|
|
.htp-toc-table td { padding: 2pt 0; vertical-align: bottom; }
|
|
.htp-toc-label a { text-decoration: none; color: inherit; }
|
|
.htp-toc-page { text-align: right; width: 4em; font-variant-numeric: tabular-nums; white-space: nowrap; }
|
|
.htp-toc-level-2 .htp-toc-label { padding-left: 1.2em; }
|
|
.htp-toc-level-3 .htp-toc-label { padding-left: 2.4em; }
|
|
.htp-toc-level-4 .htp-toc-label { padding-left: 3.6em; }
|
|
.htp-toc-level-5 .htp-toc-label { padding-left: 4.8em; }
|
|
.htp-toc-level-6 .htp-toc-label { padding-left: 6em; }
|
|
"""
|