"""Parsing of the source document, heading extraction and table of contents.""" from __future__ import annotations import logging import re from dataclasses import dataclass from lxml import html as lxml_html logger = logging.getLogger(__name__) HEADING_TAGS = ("h1", "h2", "h3", "h4", "h5", "h6") ID_SAFE = re.compile(r"[^a-zA-Z0-9_-]+") TOC_PLACEHOLDER = "\u2007\u2007\u2007" # figure spaces, keeps the pass 1 layout stable @dataclass class Heading: level: int text: str anchor: str page: int | None = None class SourceDocument: """Thin wrapper over the parsed document with the operations we need.""" def __init__(self, html: str) -> None: self.tree = lxml_html.document_fromstring(html) self.head = self.tree.find("head") if self.head is None: self.head = lxml_html.Element("head") self.tree.insert(0, self.head) self.body = self.tree.find("body") if self.body is None: raise ValueError("Dokument neobsahuje element body.") self.container = self._find_container() self._toc_css_added = False def _find_container(self): """Element whose children are the natural split points. Many documents wrap everything in a single
or
. Splitting the body would then produce a single chunk, so we descend through up to two single child wrappers. """ container = self.body for _ in range(2): children = [child for child in container if isinstance(child.tag, str)] if len(children) == 1 and len(list(children[0])) > 1: container = children[0] continue break return container # -- styles --------------------------------------------------------- def append_stylesheet(self, css: str) -> None: """Append a stylesheet as the last element of head so it wins on ties.""" style = lxml_html.Element("style") style.set("type", "text/css") style.text = css self.head.append(style) def set_base_url(self, base_url: str | None) -> None: if not base_url or self.head.find("base") is not None: return base = lxml_html.Element("base") base.set("href", base_url) self.head.insert(0, base) # -- headings ------------------------------------------------------- def collect_headings(self, max_depth: int) -> list[Heading]: """Assign ids to headings that lack one and return them in document order.""" headings: list[Heading] = [] used: set[str] = {el.get("id") for el in self.tree.iter() if el.get("id")} for element in self.body.iter(*HEADING_TAGS): level = int(element.tag[1]) if level > max_depth: continue text = " ".join(element.text_content().split()) if not text: continue anchor = element.get("id") if not anchor: anchor = self._unique_anchor(text, used) element.set("id", anchor) used.add(anchor) headings.append(Heading(level=level, text=text, anchor=anchor)) return headings @staticmethod def _unique_anchor(text: str, used: set[str]) -> str: base = ID_SAFE.sub("-", text.strip().lower()).strip("-") or "nadpis" base = f"htp-{base[:60]}" candidate = base counter = 2 while candidate in used: candidate = f"{base}-{counter}" counter += 1 return candidate # -- table of contents ---------------------------------------------- def insert_toc(self, headings: list[Heading], title: str, pages: dict[str, int] | None) -> None: """Insert the table of contents as the first block of the body. Pass 1 uses a fixed width placeholder instead of the page number so the table keeps exactly the same layout in pass 2. """ container = lxml_html.Element("nav") container.set("id", "htp-toc") container.set("class", "htp-toc") heading = lxml_html.Element("h1") heading.set("class", "htp-toc-title") heading.text = title container.append(heading) table = lxml_html.Element("table") table.set("class", "htp-toc-table") tbody = lxml_html.Element("tbody") for item in headings: if item.anchor == "htp-toc-title": continue row = lxml_html.Element("tr") row.set("class", f"htp-toc-level-{item.level}") label_cell = lxml_html.Element("td") label_cell.set("class", "htp-toc-label") link = lxml_html.Element("a") link.set("href", f"#{item.anchor}") link.text = item.text label_cell.append(link) page_cell = lxml_html.Element("td") page_cell.set("class", "htp-toc-page") if pages is None: page_cell.text = TOC_PLACEHOLDER else: page_cell.text = str(pages.get(item.anchor, "")) or TOC_PLACEHOLDER row.append(label_cell) row.append(page_cell) tbody.append(row) table.append(tbody) container.append(table) spacer = lxml_html.Element("div") spacer.set("class", "htp-toc-break") container.append(spacer) self.container.insert(0, container) if not self._toc_css_added: self.append_stylesheet(TOC_CSS) self._toc_css_added = True def remove_toc(self) -> None: existing = self.tree.find(".//nav[@id='htp-toc']") if existing is not None: existing.getparent().remove(existing) # -- serialization --------------------------------------------------- def to_html(self) -> str: return "\n" + lxml_html.tostring(self.tree, encoding="unicode") TOC_CSS = """ .htp-toc { break-after: page; } .htp-toc-table { width: 100%; border-collapse: collapse; } .htp-toc-table td { padding: 2pt 0; vertical-align: bottom; } .htp-toc-label a { text-decoration: none; color: inherit; } .htp-toc-page { text-align: right; width: 4em; font-variant-numeric: tabular-nums; white-space: nowrap; } .htp-toc-level-2 .htp-toc-label { padding-left: 1.2em; } .htp-toc-level-3 .htp-toc-label { padding-left: 2.4em; } .htp-toc-level-4 .htp-toc-label { padding-left: 3.6em; } .htp-toc-level-5 .htp-toc-label { padding-left: 4.8em; } .htp-toc-level-6 .htp-toc-label { padding-left: 6em; } """