Implementace prevodu HTML na PDF
Sluzba prijme adresu HTML dokumentu nebo HTML v tele requestu a vrati PDF. Navrzena pro dokumenty o stovkach az tisicich stranek. Rendering: - WeasyPrint jako vychozi engine, spravne CSS Paged Media, nizka pametova narocnost, bez JavaScriptu - Chromium pres Playwright pro dokumenty dokreslovane skripty - rezim auto s detekci skriptu a fallbackem pri selhani WeasyPrintu Velke dokumenty: - deleni na casti na strukturalnich hranicich, rez nikdy uvnitr tabulky nebo odstavce - dvoupruchodovy render obsahu se skutecnymi cisly stranek, pozice nadpisu se ctou z kotev hlasenych u kazde stranky - cislovani stranek bud pres CSS countery, nebo pres cislovaci vrstvu nastampovanou na hotove PDF, rozmer stranky se cte z vysledneho souboru - Chromium se restartuje po N jobech, nikdy vsak behem beziciho renderu API: - POST /convert synchronne, POST /jobs asynchronne se sledovanim stavu, stahovanim vysledku, rusenim a volitelnym callbackem - GET /health s overenim dostupnosti obou enginu a stavem fronty - OpenAPI respektuje prefix reverse proxy pres root_path Bezpecnost a provoz: - SSRF kontrola po DNS resolvu, na kazdem presmerovani a u vsech pozadavku prohlizece - nedostupne assety render nezastavi, ale hlasi se v odpovedi i v logu - fronta s omezenym poctem workeru, rozpracovane joby se pri ukonceni oznaci jako failed, nezmizi potichu - strukturovane JSON logovani s job_id - vsechny limity vypnute ve vychozim stavu Dockerfile je dvoufazovy, obsahuje zavislosti WeasyPrintu, Chromium a fonty s ceskou diakritikou. Autentizace zamerne neni implementovana, zpusob predavani neni domluveny. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
e3cc8f418b
commit
156289fe2d
@@ -0,0 +1,90 @@
|
||||
"""Page numbering and table of contents of a merged document.
|
||||
|
||||
This is where the chunking approach usually breaks, so the numbers are read back
|
||||
out of the produced PDF instead of being trusted.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
import pytest
|
||||
|
||||
pytest.importorskip("weasyprint")
|
||||
pytest.importorskip("pypdf")
|
||||
|
||||
from app.config import Settings # noqa: E402
|
||||
from app.engines.weasy import WeasyPrintEngine # noqa: E402
|
||||
from app.models import ChunkSettings, ConvertRequest, PageNumbers, Source, TocSettings # noqa: E402
|
||||
from app.services.pipeline import ConversionPipeline # noqa: E402
|
||||
from tests.conftest import build_document # noqa: E402
|
||||
|
||||
pytestmark = pytest.mark.asyncio
|
||||
|
||||
|
||||
def page_texts(path) -> list[str]:
|
||||
from pypdf import PdfReader
|
||||
|
||||
with open(path, "rb") as handle:
|
||||
return [page.extract_text() or "" for page in PdfReader(handle).pages]
|
||||
|
||||
|
||||
def make_pipeline() -> ConversionPipeline:
|
||||
return ConversionPipeline({"weasyprint": WeasyPrintEngine()}, Settings())
|
||||
|
||||
|
||||
async def test_page_numbers_are_continuous_across_chunks(tmp_path) -> None:
|
||||
request = ConvertRequest(
|
||||
source=Source(html=build_document(sections=12, paragraphs_per_section=8)),
|
||||
engine="weasyprint",
|
||||
page_numbers=PageNumbers(enabled=True, format="{page} / {pages}"),
|
||||
chunking=ChunkSettings(enabled=True, pages_per_chunk=2),
|
||||
)
|
||||
|
||||
result = await make_pipeline().run(request, tmp_path)
|
||||
texts = page_texts(result.path)
|
||||
|
||||
assert result.page_count == len(texts)
|
||||
assert result.page_count > 3, "dokument musi mit vic stranek, jinak test nic neoveruje"
|
||||
|
||||
for index, text in enumerate(texts, start=1):
|
||||
assert f"{index} / {result.page_count}" in text.replace("\n", " ")
|
||||
|
||||
|
||||
async def test_table_of_contents_points_to_the_real_pages(tmp_path) -> None:
|
||||
request = ConvertRequest(
|
||||
source=Source(html=build_document(sections=10, paragraphs_per_section=8)),
|
||||
engine="weasyprint",
|
||||
toc=TocSettings(enabled=True, depth=1, title="Obsah"),
|
||||
chunking=ChunkSettings(enabled=True, pages_per_chunk=2),
|
||||
)
|
||||
|
||||
result = await make_pipeline().run(request, tmp_path)
|
||||
texts = page_texts(result.path)
|
||||
toc_text = " ".join(texts[:2]).replace("\n", " ")
|
||||
|
||||
for index in range(10):
|
||||
heading = f"Kapitola {index + 1}"
|
||||
match = re.search(re.escape(heading) + r"\s+(\d+)", toc_text)
|
||||
assert match, f"v obsahu chybi polozka {heading}"
|
||||
|
||||
declared_page = int(match.group(1))
|
||||
actual_pages = [
|
||||
number for number, text in enumerate(texts, start=1) if heading in text.replace("\n", " ")
|
||||
]
|
||||
# The first occurrence after the table of contents is the heading itself.
|
||||
assert declared_page in actual_pages, f"{heading} deklaruje stranku {declared_page}"
|
||||
|
||||
|
||||
async def test_unchunked_document_uses_css_counters(tmp_path) -> None:
|
||||
request = ConvertRequest(
|
||||
source=Source(html=build_document(sections=4, paragraphs_per_section=6)),
|
||||
engine="weasyprint",
|
||||
page_numbers=PageNumbers(enabled=True, format="{page} / {pages}"),
|
||||
chunking=ChunkSettings(enabled=False),
|
||||
)
|
||||
|
||||
result = await make_pipeline().run(request, tmp_path)
|
||||
texts = page_texts(result.path)
|
||||
|
||||
assert f"1 / {result.page_count}" in texts[0].replace("\n", " ")
|
||||
Reference in New Issue
Block a user