Implementace prevodu HTML na PDF
Sluzba prijme adresu HTML dokumentu nebo HTML v tele requestu a vrati PDF. Navrzena pro dokumenty o stovkach az tisicich stranek. Rendering: - WeasyPrint jako vychozi engine, spravne CSS Paged Media, nizka pametova narocnost, bez JavaScriptu - Chromium pres Playwright pro dokumenty dokreslovane skripty - rezim auto s detekci skriptu a fallbackem pri selhani WeasyPrintu Velke dokumenty: - deleni na casti na strukturalnich hranicich, rez nikdy uvnitr tabulky nebo odstavce - dvoupruchodovy render obsahu se skutecnymi cisly stranek, pozice nadpisu se ctou z kotev hlasenych u kazde stranky - cislovani stranek bud pres CSS countery, nebo pres cislovaci vrstvu nastampovanou na hotove PDF, rozmer stranky se cte z vysledneho souboru - Chromium se restartuje po N jobech, nikdy vsak behem beziciho renderu API: - POST /convert synchronne, POST /jobs asynchronne se sledovanim stavu, stahovanim vysledku, rusenim a volitelnym callbackem - GET /health s overenim dostupnosti obou enginu a stavem fronty - OpenAPI respektuje prefix reverse proxy pres root_path Bezpecnost a provoz: - SSRF kontrola po DNS resolvu, na kazdem presmerovani a u vsech pozadavku prohlizece - nedostupne assety render nezastavi, ale hlasi se v odpovedi i v logu - fronta s omezenym poctem workeru, rozpracovane joby se pri ukonceni oznaci jako failed, nezmizi potichu - strukturovane JSON logovani s job_id - vsechny limity vypnute ve vychozim stavu Dockerfile je dvoufazovy, obsahuje zavislosti WeasyPrintu, Chromium a fonty s ceskou diakritikou. Autentizace zamerne neni implementovana, zpusob predavani neni domluveny. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
e3cc8f418b
commit
156289fe2d
@@ -0,0 +1,186 @@
|
||||
"""Parsing of the source document, heading extraction and table of contents."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
|
||||
from lxml import html as lxml_html
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
HEADING_TAGS = ("h1", "h2", "h3", "h4", "h5", "h6")
|
||||
ID_SAFE = re.compile(r"[^a-zA-Z0-9_-]+")
|
||||
|
||||
TOC_PLACEHOLDER = "\u2007\u2007\u2007" # figure spaces, keeps the pass 1 layout stable
|
||||
|
||||
|
||||
@dataclass
|
||||
class Heading:
|
||||
level: int
|
||||
text: str
|
||||
anchor: str
|
||||
page: int | None = None
|
||||
|
||||
|
||||
class SourceDocument:
|
||||
"""Thin wrapper over the parsed document with the operations we need."""
|
||||
|
||||
def __init__(self, html: str) -> None:
|
||||
self.tree = lxml_html.document_fromstring(html)
|
||||
self.head = self.tree.find("head")
|
||||
if self.head is None:
|
||||
self.head = lxml_html.Element("head")
|
||||
self.tree.insert(0, self.head)
|
||||
self.body = self.tree.find("body")
|
||||
if self.body is None:
|
||||
raise ValueError("Dokument neobsahuje element body.")
|
||||
self.container = self._find_container()
|
||||
self._toc_css_added = False
|
||||
|
||||
def _find_container(self):
|
||||
"""Element whose children are the natural split points.
|
||||
|
||||
Many documents wrap everything in a single <main> or <div>. Splitting the
|
||||
body would then produce a single chunk, so we descend through up to two
|
||||
single child wrappers.
|
||||
"""
|
||||
container = self.body
|
||||
for _ in range(2):
|
||||
children = [child for child in container if isinstance(child.tag, str)]
|
||||
if len(children) == 1 and len(list(children[0])) > 1:
|
||||
container = children[0]
|
||||
continue
|
||||
break
|
||||
return container
|
||||
|
||||
# -- styles ---------------------------------------------------------
|
||||
def append_stylesheet(self, css: str) -> None:
|
||||
"""Append a stylesheet as the last element of head so it wins on ties."""
|
||||
style = lxml_html.Element("style")
|
||||
style.set("type", "text/css")
|
||||
style.text = css
|
||||
self.head.append(style)
|
||||
|
||||
def set_base_url(self, base_url: str | None) -> None:
|
||||
if not base_url or self.head.find("base") is not None:
|
||||
return
|
||||
base = lxml_html.Element("base")
|
||||
base.set("href", base_url)
|
||||
self.head.insert(0, base)
|
||||
|
||||
# -- headings -------------------------------------------------------
|
||||
def collect_headings(self, max_depth: int) -> list[Heading]:
|
||||
"""Assign ids to headings that lack one and return them in document order."""
|
||||
headings: list[Heading] = []
|
||||
used: set[str] = {el.get("id") for el in self.tree.iter() if el.get("id")}
|
||||
|
||||
for element in self.body.iter(*HEADING_TAGS):
|
||||
level = int(element.tag[1])
|
||||
if level > max_depth:
|
||||
continue
|
||||
|
||||
text = " ".join(element.text_content().split())
|
||||
if not text:
|
||||
continue
|
||||
|
||||
anchor = element.get("id")
|
||||
if not anchor:
|
||||
anchor = self._unique_anchor(text, used)
|
||||
element.set("id", anchor)
|
||||
used.add(anchor)
|
||||
|
||||
headings.append(Heading(level=level, text=text, anchor=anchor))
|
||||
|
||||
return headings
|
||||
|
||||
@staticmethod
|
||||
def _unique_anchor(text: str, used: set[str]) -> str:
|
||||
base = ID_SAFE.sub("-", text.strip().lower()).strip("-") or "nadpis"
|
||||
base = f"htp-{base[:60]}"
|
||||
candidate = base
|
||||
counter = 2
|
||||
while candidate in used:
|
||||
candidate = f"{base}-{counter}"
|
||||
counter += 1
|
||||
return candidate
|
||||
|
||||
# -- table of contents ----------------------------------------------
|
||||
def insert_toc(self, headings: list[Heading], title: str, pages: dict[str, int] | None) -> None:
|
||||
"""Insert the table of contents as the first block of the body.
|
||||
|
||||
Pass 1 uses a fixed width placeholder instead of the page number so the
|
||||
table keeps exactly the same layout in pass 2.
|
||||
"""
|
||||
container = lxml_html.Element("nav")
|
||||
container.set("id", "htp-toc")
|
||||
container.set("class", "htp-toc")
|
||||
|
||||
heading = lxml_html.Element("h1")
|
||||
heading.set("class", "htp-toc-title")
|
||||
heading.text = title
|
||||
container.append(heading)
|
||||
|
||||
table = lxml_html.Element("table")
|
||||
table.set("class", "htp-toc-table")
|
||||
tbody = lxml_html.Element("tbody")
|
||||
|
||||
for item in headings:
|
||||
if item.anchor == "htp-toc-title":
|
||||
continue
|
||||
row = lxml_html.Element("tr")
|
||||
row.set("class", f"htp-toc-level-{item.level}")
|
||||
|
||||
label_cell = lxml_html.Element("td")
|
||||
label_cell.set("class", "htp-toc-label")
|
||||
link = lxml_html.Element("a")
|
||||
link.set("href", f"#{item.anchor}")
|
||||
link.text = item.text
|
||||
label_cell.append(link)
|
||||
|
||||
page_cell = lxml_html.Element("td")
|
||||
page_cell.set("class", "htp-toc-page")
|
||||
if pages is None:
|
||||
page_cell.text = TOC_PLACEHOLDER
|
||||
else:
|
||||
page_cell.text = str(pages.get(item.anchor, "")) or TOC_PLACEHOLDER
|
||||
|
||||
row.append(label_cell)
|
||||
row.append(page_cell)
|
||||
tbody.append(row)
|
||||
|
||||
table.append(tbody)
|
||||
container.append(table)
|
||||
|
||||
spacer = lxml_html.Element("div")
|
||||
spacer.set("class", "htp-toc-break")
|
||||
container.append(spacer)
|
||||
|
||||
self.container.insert(0, container)
|
||||
if not self._toc_css_added:
|
||||
self.append_stylesheet(TOC_CSS)
|
||||
self._toc_css_added = True
|
||||
|
||||
def remove_toc(self) -> None:
|
||||
existing = self.tree.find(".//nav[@id='htp-toc']")
|
||||
if existing is not None:
|
||||
existing.getparent().remove(existing)
|
||||
|
||||
# -- serialization ---------------------------------------------------
|
||||
def to_html(self) -> str:
|
||||
return "<!DOCTYPE html>\n" + lxml_html.tostring(self.tree, encoding="unicode")
|
||||
|
||||
|
||||
TOC_CSS = """
|
||||
.htp-toc { break-after: page; }
|
||||
.htp-toc-table { width: 100%; border-collapse: collapse; }
|
||||
.htp-toc-table td { padding: 2pt 0; vertical-align: bottom; }
|
||||
.htp-toc-label a { text-decoration: none; color: inherit; }
|
||||
.htp-toc-page { text-align: right; width: 4em; font-variant-numeric: tabular-nums; white-space: nowrap; }
|
||||
.htp-toc-level-2 .htp-toc-label { padding-left: 1.2em; }
|
||||
.htp-toc-level-3 .htp-toc-label { padding-left: 2.4em; }
|
||||
.htp-toc-level-4 .htp-toc-label { padding-left: 3.6em; }
|
||||
.htp-toc-level-5 .htp-toc-label { padding-left: 4.8em; }
|
||||
.htp-toc-level-6 .htp-toc-label { padding-left: 6em; }
|
||||
"""
|
||||
Reference in New Issue
Block a user