239 lines
8.4 KiB
Python
239 lines
8.4 KiB
Python
"""Parsing of the source document, heading extraction and table of contents."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import re
|
|
from dataclasses import dataclass
|
|
|
|
from lxml import html as lxml_html
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
HEADING_TAGS = ("h1", "h2", "h3", "h4", "h5", "h6")
|
|
ID_SAFE = re.compile(r"[^a-zA-Z0-9_-]+")
|
|
|
|
# Matches a width declaration in percent, but never max-width or min-width.
|
|
PERCENT_WIDTH = re.compile(r"(?<![\w-])width\s*:\s*\d+(?:\.\d+)?%", re.IGNORECASE)
|
|
|
|
TOC_PLACEHOLDER = "\u2007\u2007\u2007" # figure spaces, keeps the pass 1 layout stable
|
|
|
|
|
|
@dataclass
|
|
class Heading:
|
|
level: int
|
|
text: str
|
|
anchor: str
|
|
page: int | None = None
|
|
|
|
|
|
class SourceDocument:
|
|
"""Thin wrapper over the parsed document with the operations we need."""
|
|
|
|
def __init__(self, html: str) -> None:
|
|
self.tree = lxml_html.document_fromstring(html)
|
|
self.head = self.tree.find("head")
|
|
if self.head is None:
|
|
self.head = lxml_html.Element("head")
|
|
self.tree.insert(0, self.head)
|
|
self.body = self.tree.find("body")
|
|
if self.body is None:
|
|
raise ValueError("Dokument neobsahuje element body.")
|
|
self.container = self._find_container()
|
|
self._toc_css_added = False
|
|
|
|
def _find_container(self):
|
|
"""Element whose children are the natural split points.
|
|
|
|
Many documents wrap everything in a single <main> or <div>. Splitting the
|
|
body would then produce a single chunk, so we descend through up to two
|
|
single child wrappers.
|
|
"""
|
|
container = self.body
|
|
for _ in range(2):
|
|
children = [child for child in container if isinstance(child.tag, str)]
|
|
if len(children) == 1 and len(list(children[0])) > 1:
|
|
container = children[0]
|
|
continue
|
|
break
|
|
return container
|
|
|
|
# -- styles ---------------------------------------------------------
|
|
def append_stylesheet(self, css: str) -> None:
|
|
"""Append a stylesheet as the last element of head so it wins on ties."""
|
|
style = lxml_html.Element("style")
|
|
style.set("type", "text/css")
|
|
style.text = css
|
|
self.head.append(style)
|
|
|
|
def set_base_url(self, base_url: str | None) -> None:
|
|
if not base_url or self.head.find("base") is not None:
|
|
return
|
|
base = lxml_html.Element("base")
|
|
base.set("href", base_url)
|
|
self.head.insert(0, base)
|
|
|
|
# -- images ---------------------------------------------------------
|
|
def relax_percentage_image_widths(self) -> int:
|
|
"""Replace percentage widths of images inside table cells by auto.
|
|
|
|
WeasyPrint resolves such a percentage against a cell width that is not
|
|
known yet while the table is being laid out. The image comes out zero
|
|
wide and disappears from the PDF without any error, so it is not even
|
|
reported as a missing asset. Rendering the image at its intrinsic size
|
|
capped by max-width gives the result the document intended.
|
|
|
|
Only relevant for WeasyPrint, Chromium lays these images out correctly.
|
|
Returns the number of images that were changed.
|
|
"""
|
|
changed = 0
|
|
for cell in self.body.iter("td", "th"):
|
|
for image in cell.iter("img"):
|
|
if self._relax_image_width(image):
|
|
changed += 1
|
|
|
|
if changed:
|
|
logger.info(
|
|
"Percentage width of images inside table cells replaced by auto",
|
|
extra={"images": changed},
|
|
)
|
|
return changed
|
|
|
|
@staticmethod
|
|
def _relax_image_width(image) -> bool:
|
|
touched = False
|
|
|
|
if (image.get("width") or "").strip().endswith("%"):
|
|
del image.attrib["width"]
|
|
touched = True
|
|
|
|
style = image.get("style") or ""
|
|
if PERCENT_WIDTH.search(style):
|
|
style = PERCENT_WIDTH.sub("width: auto", style)
|
|
touched = True
|
|
|
|
if not touched:
|
|
return False
|
|
|
|
if "max-width" not in style.lower():
|
|
stripped = style.strip().rstrip(";")
|
|
style = f"{stripped}; max-width: 100%" if stripped else "max-width: 100%"
|
|
|
|
image.set("style", style)
|
|
return True
|
|
|
|
# -- headings -------------------------------------------------------
|
|
def collect_headings(self, max_depth: int) -> list[Heading]:
|
|
"""Assign ids to headings that lack one and return them in document order."""
|
|
headings: list[Heading] = []
|
|
used: set[str] = {el.get("id") for el in self.tree.iter() if el.get("id")}
|
|
|
|
for element in self.body.iter(*HEADING_TAGS):
|
|
level = int(element.tag[1])
|
|
if level > max_depth:
|
|
continue
|
|
|
|
text = " ".join(element.text_content().split())
|
|
if not text:
|
|
continue
|
|
|
|
anchor = element.get("id")
|
|
if not anchor:
|
|
anchor = self._unique_anchor(text, used)
|
|
element.set("id", anchor)
|
|
used.add(anchor)
|
|
|
|
headings.append(Heading(level=level, text=text, anchor=anchor))
|
|
|
|
return headings
|
|
|
|
@staticmethod
|
|
def _unique_anchor(text: str, used: set[str]) -> str:
|
|
base = ID_SAFE.sub("-", text.strip().lower()).strip("-") or "nadpis"
|
|
base = f"htp-{base[:60]}"
|
|
candidate = base
|
|
counter = 2
|
|
while candidate in used:
|
|
candidate = f"{base}-{counter}"
|
|
counter += 1
|
|
return candidate
|
|
|
|
# -- table of contents ----------------------------------------------
|
|
def insert_toc(self, headings: list[Heading], title: str, pages: dict[str, int] | None) -> None:
|
|
"""Insert the table of contents as the first block of the body.
|
|
|
|
Pass 1 uses a fixed width placeholder instead of the page number so the
|
|
table keeps exactly the same layout in pass 2.
|
|
"""
|
|
container = lxml_html.Element("nav")
|
|
container.set("id", "htp-toc")
|
|
container.set("class", "htp-toc")
|
|
|
|
heading = lxml_html.Element("h1")
|
|
heading.set("class", "htp-toc-title")
|
|
heading.text = title
|
|
container.append(heading)
|
|
|
|
table = lxml_html.Element("table")
|
|
table.set("class", "htp-toc-table")
|
|
tbody = lxml_html.Element("tbody")
|
|
|
|
for item in headings:
|
|
if item.anchor == "htp-toc-title":
|
|
continue
|
|
row = lxml_html.Element("tr")
|
|
row.set("class", f"htp-toc-level-{item.level}")
|
|
|
|
label_cell = lxml_html.Element("td")
|
|
label_cell.set("class", "htp-toc-label")
|
|
link = lxml_html.Element("a")
|
|
link.set("href", f"#{item.anchor}")
|
|
link.text = item.text
|
|
label_cell.append(link)
|
|
|
|
page_cell = lxml_html.Element("td")
|
|
page_cell.set("class", "htp-toc-page")
|
|
if pages is None:
|
|
page_cell.text = TOC_PLACEHOLDER
|
|
else:
|
|
page_cell.text = str(pages.get(item.anchor, "")) or TOC_PLACEHOLDER
|
|
|
|
row.append(label_cell)
|
|
row.append(page_cell)
|
|
tbody.append(row)
|
|
|
|
table.append(tbody)
|
|
container.append(table)
|
|
|
|
spacer = lxml_html.Element("div")
|
|
spacer.set("class", "htp-toc-break")
|
|
container.append(spacer)
|
|
|
|
self.container.insert(0, container)
|
|
if not self._toc_css_added:
|
|
self.append_stylesheet(TOC_CSS)
|
|
self._toc_css_added = True
|
|
|
|
def remove_toc(self) -> None:
|
|
existing = self.tree.find(".//nav[@id='htp-toc']")
|
|
if existing is not None:
|
|
existing.getparent().remove(existing)
|
|
|
|
# -- serialization ---------------------------------------------------
|
|
def to_html(self) -> str:
|
|
return "<!DOCTYPE html>\n" + lxml_html.tostring(self.tree, encoding="unicode")
|
|
|
|
|
|
TOC_CSS = """
|
|
.htp-toc { break-after: page; }
|
|
.htp-toc-table { width: 100%; border-collapse: collapse; }
|
|
.htp-toc-table td { padding: 2pt 0; vertical-align: bottom; }
|
|
.htp-toc-label a { text-decoration: none; color: inherit; }
|
|
.htp-toc-page { text-align: right; width: 4em; font-variant-numeric: tabular-nums; white-space: nowrap; }
|
|
.htp-toc-level-2 .htp-toc-label { padding-left: 1.2em; }
|
|
.htp-toc-level-3 .htp-toc-label { padding-left: 2.4em; }
|
|
.htp-toc-level-4 .htp-toc-label { padding-left: 3.6em; }
|
|
.htp-toc-level-5 .htp-toc-label { padding-left: 4.8em; }
|
|
.htp-toc-level-6 .htp-toc-label { padding-left: 6em; }
|
|
"""
|