"""Parsing of the source document, heading extraction and table of contents.""" from __future__ import annotations import logging import re from dataclasses import dataclass from lxml import html as lxml_html logger = logging.getLogger(__name__) HEADING_TAGS = ("h1", "h2", "h3", "h4", "h5", "h6") ID_SAFE = re.compile(r"[^a-zA-Z0-9_-]+") # Matches a width declaration in percent, but never max-width or min-width. PERCENT_WIDTH = re.compile(r"(? None: self.tree = lxml_html.document_fromstring(html) self.head = self.tree.find("head") if self.head is None: self.head = lxml_html.Element("head") self.tree.insert(0, self.head) self.body = self.tree.find("body") if self.body is None: raise ValueError("Dokument neobsahuje element body.") self.container = self._find_container() self._toc_css_added = False def _find_container(self): """Element whose children are the natural split points. Many documents wrap everything in a single
or
. Splitting the body would then produce a single chunk, so we descend through up to two single child wrappers. """ container = self.body for _ in range(2): children = [child for child in container if isinstance(child.tag, str)] if len(children) == 1 and len(list(children[0])) > 1: container = children[0] continue break return container # -- styles --------------------------------------------------------- def append_stylesheet(self, css: str) -> None: """Append a stylesheet as the last element of head so it wins on ties.""" style = lxml_html.Element("style") style.set("type", "text/css") style.text = css self.head.append(style) def set_base_url(self, base_url: str | None) -> None: if not base_url or self.head.find("base") is not None: return base = lxml_html.Element("base") base.set("href", base_url) self.head.insert(0, base) # -- images --------------------------------------------------------- def relax_percentage_image_widths(self) -> int: """Replace percentage widths of images inside table cells by auto. WeasyPrint resolves such a percentage against a cell width that is not known yet while the table is being laid out. The image comes out zero wide and disappears from the PDF without any error, so it is not even reported as a missing asset. Rendering the image at its intrinsic size capped by max-width gives the result the document intended. Only relevant for WeasyPrint, Chromium lays these images out correctly. Returns the number of images that were changed. """ changed = 0 for cell in self.body.iter("td", "th"): for image in cell.iter("img"): if self._relax_image_width(image): changed += 1 if changed: logger.info( "Percentage width of images inside table cells replaced by auto", extra={"images": changed}, ) return changed @staticmethod def _relax_image_width(image) -> bool: touched = False if (image.get("width") or "").strip().endswith("%"): del image.attrib["width"] touched = True style = image.get("style") or "" if PERCENT_WIDTH.search(style): style = PERCENT_WIDTH.sub("width: auto", style) touched = True if not touched: return False if "max-width" not in style.lower(): stripped = style.strip().rstrip(";") style = f"{stripped}; max-width: 100%" if stripped else "max-width: 100%" image.set("style", style) return True # -- headings ------------------------------------------------------- def collect_headings(self, max_depth: int) -> list[Heading]: """Assign ids to headings that lack one and return them in document order.""" headings: list[Heading] = [] used: set[str] = {el.get("id") for el in self.tree.iter() if el.get("id")} for element in self.body.iter(*HEADING_TAGS): level = int(element.tag[1]) if level > max_depth: continue text = " ".join(element.text_content().split()) if not text: continue anchor = element.get("id") if not anchor: anchor = self._unique_anchor(text, used) element.set("id", anchor) used.add(anchor) headings.append(Heading(level=level, text=text, anchor=anchor)) return headings @staticmethod def _unique_anchor(text: str, used: set[str]) -> str: base = ID_SAFE.sub("-", text.strip().lower()).strip("-") or "nadpis" base = f"htp-{base[:60]}" candidate = base counter = 2 while candidate in used: candidate = f"{base}-{counter}" counter += 1 return candidate # -- table of contents ---------------------------------------------- def insert_toc(self, headings: list[Heading], title: str, pages: dict[str, int] | None) -> None: """Insert the table of contents as the first block of the body. Pass 1 uses a fixed width placeholder instead of the page number so the table keeps exactly the same layout in pass 2. """ container = lxml_html.Element("nav") container.set("id", "htp-toc") container.set("class", "htp-toc") heading = lxml_html.Element("h1") heading.set("class", "htp-toc-title") heading.text = title container.append(heading) table = lxml_html.Element("table") table.set("class", "htp-toc-table") tbody = lxml_html.Element("tbody") for item in headings: if item.anchor == "htp-toc-title": continue row = lxml_html.Element("tr") row.set("class", f"htp-toc-level-{item.level}") label_cell = lxml_html.Element("td") label_cell.set("class", "htp-toc-label") link = lxml_html.Element("a") link.set("href", f"#{item.anchor}") link.text = item.text label_cell.append(link) page_cell = lxml_html.Element("td") page_cell.set("class", "htp-toc-page") if pages is None: page_cell.text = TOC_PLACEHOLDER else: page_cell.text = str(pages.get(item.anchor, "")) or TOC_PLACEHOLDER row.append(label_cell) row.append(page_cell) tbody.append(row) table.append(tbody) container.append(table) spacer = lxml_html.Element("div") spacer.set("class", "htp-toc-break") container.append(spacer) self.container.insert(0, container) if not self._toc_css_added: self.append_stylesheet(TOC_CSS) self._toc_css_added = True def remove_toc(self) -> None: existing = self.tree.find(".//nav[@id='htp-toc']") if existing is not None: existing.getparent().remove(existing) # -- serialization --------------------------------------------------- def to_html(self) -> str: return "\n" + lxml_html.tostring(self.tree, encoding="unicode") TOC_CSS = """ .htp-toc { break-after: page; } .htp-toc-table { width: 100%; border-collapse: collapse; } .htp-toc-table td { padding: 2pt 0; vertical-align: bottom; } .htp-toc-label a { text-decoration: none; color: inherit; } .htp-toc-page { text-align: right; width: 4em; font-variant-numeric: tabular-nums; white-space: nowrap; } .htp-toc-level-2 .htp-toc-label { padding-left: 1.2em; } .htp-toc-level-3 .htp-toc-label { padding-left: 2.4em; } .htp-toc-level-4 .htp-toc-label { padding-left: 3.6em; } .htp-toc-level-5 .htp-toc-label { padding-left: 4.8em; } .htp-toc-level-6 .htp-toc-label { padding-left: 6em; } """