"""Page numbering and table of contents of a merged document. This is where the chunking approach usually breaks, so the numbers are read back out of the produced PDF instead of being trusted. """ from __future__ import annotations import re import pytest pytest.importorskip("weasyprint") pytest.importorskip("pypdf") from app.config import Settings # noqa: E402 from app.engines.weasy import WeasyPrintEngine # noqa: E402 from app.models import ChunkSettings, ConvertRequest, PageNumbers, Source, TocSettings # noqa: E402 from app.services.pipeline import ConversionPipeline # noqa: E402 from tests.conftest import build_document # noqa: E402 pytestmark = pytest.mark.asyncio def page_texts(path) -> list[str]: from pypdf import PdfReader with open(path, "rb") as handle: return [page.extract_text() or "" for page in PdfReader(handle).pages] def make_pipeline() -> ConversionPipeline: return ConversionPipeline({"weasyprint": WeasyPrintEngine()}, Settings()) async def test_page_numbers_are_continuous_across_chunks(tmp_path) -> None: request = ConvertRequest( source=Source(html=build_document(sections=12, paragraphs_per_section=8)), engine="weasyprint", page_numbers=PageNumbers(enabled=True, format="{page} / {pages}"), chunking=ChunkSettings(enabled=True, pages_per_chunk=2), ) result = await make_pipeline().run(request, tmp_path) texts = page_texts(result.path) assert result.page_count == len(texts) assert result.page_count > 3, "dokument musi mit vic stranek, jinak test nic neoveruje" for index, text in enumerate(texts, start=1): assert f"{index} / {result.page_count}" in text.replace("\n", " ") async def test_table_of_contents_points_to_the_real_pages(tmp_path) -> None: request = ConvertRequest( source=Source(html=build_document(sections=10, paragraphs_per_section=8)), engine="weasyprint", toc=TocSettings(enabled=True, depth=1, title="Obsah"), chunking=ChunkSettings(enabled=True, pages_per_chunk=2), ) result = await make_pipeline().run(request, tmp_path) texts = page_texts(result.path) toc_text = " ".join(texts[:2]).replace("\n", " ") for index in range(10): heading = f"Kapitola {index + 1}" match = re.search(re.escape(heading) + r"\s+(\d+)", toc_text) assert match, f"v obsahu chybi polozka {heading}" declared_page = int(match.group(1)) actual_pages = [ number for number, text in enumerate(texts, start=1) if heading in text.replace("\n", " ") ] # The first occurrence after the table of contents is the heading itself. assert declared_page in actual_pages, f"{heading} deklaruje stranku {declared_page}" async def test_unchunked_document_uses_css_counters(tmp_path) -> None: request = ConvertRequest( source=Source(html=build_document(sections=4, paragraphs_per_section=6)), engine="weasyprint", page_numbers=PageNumbers(enabled=True, format="{page} / {pages}"), chunking=ChunkSettings(enabled=False), ) result = await make_pipeline().run(request, tmp_path) texts = page_texts(result.path) assert f"1 / {result.page_count}" in texts[0].replace("\n", " ")