Files
html-to-pdf/app/pdf/document.py
T
2026-09-03 10:03:10 +02:00

239 lines
8.4 KiB
Python

"""Parsing of the source document, heading extraction and table of contents."""
from __future__ import annotations
import logging
import re
from dataclasses import dataclass
from lxml import html as lxml_html
logger = logging.getLogger(__name__)
HEADING_TAGS = ("h1", "h2", "h3", "h4", "h5", "h6")
ID_SAFE = re.compile(r"[^a-zA-Z0-9_-]+")
# Matches a width declaration in percent, but never max-width or min-width.
PERCENT_WIDTH = re.compile(r"(?<![\w-])width\s*:\s*\d+(?:\.\d+)?%", re.IGNORECASE)
TOC_PLACEHOLDER = "\u2007\u2007\u2007" # figure spaces, keeps the pass 1 layout stable
@dataclass
class Heading:
level: int
text: str
anchor: str
page: int | None = None
class SourceDocument:
"""Thin wrapper over the parsed document with the operations we need."""
def __init__(self, html: str) -> None:
self.tree = lxml_html.document_fromstring(html)
self.head = self.tree.find("head")
if self.head is None:
self.head = lxml_html.Element("head")
self.tree.insert(0, self.head)
self.body = self.tree.find("body")
if self.body is None:
raise ValueError("Dokument neobsahuje element body.")
self.container = self._find_container()
self._toc_css_added = False
def _find_container(self):
"""Element whose children are the natural split points.
Many documents wrap everything in a single <main> or <div>. Splitting the
body would then produce a single chunk, so we descend through up to two
single child wrappers.
"""
container = self.body
for _ in range(2):
children = [child for child in container if isinstance(child.tag, str)]
if len(children) == 1 and len(list(children[0])) > 1:
container = children[0]
continue
break
return container
# -- styles ---------------------------------------------------------
def append_stylesheet(self, css: str) -> None:
"""Append a stylesheet as the last element of head so it wins on ties."""
style = lxml_html.Element("style")
style.set("type", "text/css")
style.text = css
self.head.append(style)
def set_base_url(self, base_url: str | None) -> None:
if not base_url or self.head.find("base") is not None:
return
base = lxml_html.Element("base")
base.set("href", base_url)
self.head.insert(0, base)
# -- images ---------------------------------------------------------
def relax_percentage_image_widths(self) -> int:
"""Replace percentage widths of images inside table cells by auto.
WeasyPrint resolves such a percentage against a cell width that is not
known yet while the table is being laid out. The image comes out zero
wide and disappears from the PDF without any error, so it is not even
reported as a missing asset. Rendering the image at its intrinsic size
capped by max-width gives the result the document intended.
Only relevant for WeasyPrint, Chromium lays these images out correctly.
Returns the number of images that were changed.
"""
changed = 0
for cell in self.body.iter("td", "th"):
for image in cell.iter("img"):
if self._relax_image_width(image):
changed += 1
if changed:
logger.info(
"Percentage width of images inside table cells replaced by auto",
extra={"images": changed},
)
return changed
@staticmethod
def _relax_image_width(image) -> bool:
touched = False
if (image.get("width") or "").strip().endswith("%"):
del image.attrib["width"]
touched = True
style = image.get("style") or ""
if PERCENT_WIDTH.search(style):
style = PERCENT_WIDTH.sub("width: auto", style)
touched = True
if not touched:
return False
if "max-width" not in style.lower():
stripped = style.strip().rstrip(";")
style = f"{stripped}; max-width: 100%" if stripped else "max-width: 100%"
image.set("style", style)
return True
# -- headings -------------------------------------------------------
def collect_headings(self, max_depth: int) -> list[Heading]:
"""Assign ids to headings that lack one and return them in document order."""
headings: list[Heading] = []
used: set[str] = {el.get("id") for el in self.tree.iter() if el.get("id")}
for element in self.body.iter(*HEADING_TAGS):
level = int(element.tag[1])
if level > max_depth:
continue
text = " ".join(element.text_content().split())
if not text:
continue
anchor = element.get("id")
if not anchor:
anchor = self._unique_anchor(text, used)
element.set("id", anchor)
used.add(anchor)
headings.append(Heading(level=level, text=text, anchor=anchor))
return headings
@staticmethod
def _unique_anchor(text: str, used: set[str]) -> str:
base = ID_SAFE.sub("-", text.strip().lower()).strip("-") or "nadpis"
base = f"htp-{base[:60]}"
candidate = base
counter = 2
while candidate in used:
candidate = f"{base}-{counter}"
counter += 1
return candidate
# -- table of contents ----------------------------------------------
def insert_toc(self, headings: list[Heading], title: str, pages: dict[str, int] | None) -> None:
"""Insert the table of contents as the first block of the body.
Pass 1 uses a fixed width placeholder instead of the page number so the
table keeps exactly the same layout in pass 2.
"""
container = lxml_html.Element("nav")
container.set("id", "htp-toc")
container.set("class", "htp-toc")
heading = lxml_html.Element("h1")
heading.set("class", "htp-toc-title")
heading.text = title
container.append(heading)
table = lxml_html.Element("table")
table.set("class", "htp-toc-table")
tbody = lxml_html.Element("tbody")
for item in headings:
if item.anchor == "htp-toc-title":
continue
row = lxml_html.Element("tr")
row.set("class", f"htp-toc-level-{item.level}")
label_cell = lxml_html.Element("td")
label_cell.set("class", "htp-toc-label")
link = lxml_html.Element("a")
link.set("href", f"#{item.anchor}")
link.text = item.text
label_cell.append(link)
page_cell = lxml_html.Element("td")
page_cell.set("class", "htp-toc-page")
if pages is None:
page_cell.text = TOC_PLACEHOLDER
else:
page_cell.text = str(pages.get(item.anchor, "")) or TOC_PLACEHOLDER
row.append(label_cell)
row.append(page_cell)
tbody.append(row)
table.append(tbody)
container.append(table)
spacer = lxml_html.Element("div")
spacer.set("class", "htp-toc-break")
container.append(spacer)
self.container.insert(0, container)
if not self._toc_css_added:
self.append_stylesheet(TOC_CSS)
self._toc_css_added = True
def remove_toc(self) -> None:
existing = self.tree.find(".//nav[@id='htp-toc']")
if existing is not None:
existing.getparent().remove(existing)
# -- serialization ---------------------------------------------------
def to_html(self) -> str:
return "<!DOCTYPE html>\n" + lxml_html.tostring(self.tree, encoding="unicode")
TOC_CSS = """
.htp-toc { break-after: page; }
.htp-toc-table { width: 100%; border-collapse: collapse; }
.htp-toc-table td { padding: 2pt 0; vertical-align: bottom; }
.htp-toc-label a { text-decoration: none; color: inherit; }
.htp-toc-page { text-align: right; width: 4em; font-variant-numeric: tabular-nums; white-space: nowrap; }
.htp-toc-level-2 .htp-toc-label { padding-left: 1.2em; }
.htp-toc-level-3 .htp-toc-label { padding-left: 2.4em; }
.htp-toc-level-4 .htp-toc-label { padding-left: 3.6em; }
.htp-toc-level-5 .htp-toc-label { padding-left: 4.8em; }
.htp-toc-level-6 .htp-toc-label { padding-left: 6em; }
"""