Implementace prevodu HTML na PDF

Sluzba prijme adresu HTML dokumentu nebo HTML v tele requestu a vrati PDF.
Navrzena pro dokumenty o stovkach az tisicich stranek.

Rendering:
- WeasyPrint jako vychozi engine, spravne CSS Paged Media, nizka pametova
  narocnost, bez JavaScriptu
- Chromium pres Playwright pro dokumenty dokreslovane skripty
- rezim auto s detekci skriptu a fallbackem pri selhani WeasyPrintu

Velke dokumenty:
- deleni na casti na strukturalnich hranicich, rez nikdy uvnitr tabulky
  nebo odstavce
- dvoupruchodovy render obsahu se skutecnymi cisly stranek, pozice nadpisu
  se ctou z kotev hlasenych u kazde stranky
- cislovani stranek bud pres CSS countery, nebo pres cislovaci vrstvu
  nastampovanou na hotove PDF, rozmer stranky se cte z vysledneho souboru
- Chromium se restartuje po N jobech, nikdy vsak behem beziciho renderu

API:
- POST /convert synchronne, POST /jobs asynchronne se sledovanim stavu,
  stahovanim vysledku, rusenim a volitelnym callbackem
- GET /health s overenim dostupnosti obou enginu a stavem fronty
- OpenAPI respektuje prefix reverse proxy pres root_path

Bezpecnost a provoz:
- SSRF kontrola po DNS resolvu, na kazdem presmerovani a u vsech pozadavku
  prohlizece
- nedostupne assety render nezastavi, ale hlasi se v odpovedi i v logu
- fronta s omezenym poctem workeru, rozpracovane joby se pri ukonceni
  oznaci jako failed, nezmizi potichu
- strukturovane JSON logovani s job_id
- vsechny limity vypnute ve vychozim stavu

Dockerfile je dvoufazovy, obsahuje zavislosti WeasyPrintu, Chromium
a fonty s ceskou diakritikou.

Autentizace zamerne neni implementovana, zpusob predavani neni domluveny.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
JiriUhlir
2026-08-27 14:50:10 +02:00
co-authored by Claude Opus 5
parent e3cc8f418b
commit 156289fe2d
48 changed files with 4043 additions and 24 deletions
View File
+46
View File
@@ -0,0 +1,46 @@
"""Shared fixtures.
Nothing here starts a real browser. Engine level tests skip themselves when the
engine is not installed in the current environment.
"""
from __future__ import annotations
import pytest
def build_document(sections: int, paragraphs_per_section: int = 12) -> str:
"""Synthetic document with predictable structure and size."""
parts = [
"<!DOCTYPE html><html><head><meta charset='utf-8'>",
"<style>body{font-family:sans-serif;font-size:11pt}</style>",
"</head><body>",
]
for index in range(sections):
parts.append(f"<section id='sekce-{index}'><h1>Kapitola {index + 1}</h1>")
for paragraph in range(paragraphs_per_section):
parts.append(
f"<p>Odstavec {paragraph + 1} kapitoly {index + 1}. "
+ ("Text s ceskou diakritikou pro overeni fontu. " * 8)
+ "</p>"
)
parts.append("</section>")
parts.append("</body></html>")
return "".join(parts)
@pytest.fixture
def small_document() -> str:
return build_document(sections=3, paragraphs_per_section=2)
@pytest.fixture
def large_document() -> str:
"""Roughly twelve hundred pages, used for memory and timing measurements."""
return build_document(sections=400, paragraphs_per_section=14)
@pytest.fixture
def unsplittable_document() -> str:
body = "".join(f"<p>Odstavec {index}. " + ("Text. " * 40) + "</p>" for index in range(200))
return f"<!DOCTYPE html><html><head><meta charset='utf-8'></head><body>{body}</body></html>"
+69
View File
@@ -0,0 +1,69 @@
"""HTTP surface of the service."""
from __future__ import annotations
import pytest
from fastapi.testclient import TestClient
from app.main import app
@pytest.fixture
def client():
with TestClient(app) as test_client:
yield test_client
def test_health_is_available(client) -> None:
response = client.get("/health")
assert response.status_code == 200
body = response.json()
assert body["status"] in ("ok", "degraded")
assert "engines" in body
def test_version_reports_the_application(client) -> None:
response = client.get("/version")
assert response.status_code == 200
assert response.json()["language"] == "python"
def test_openapi_is_generated(client) -> None:
response = client.get("/openapi.json")
assert response.status_code == 200
paths = response.json()["paths"]
assert "/convert" in paths
assert "/jobs" in paths
assert "/jobs/{job_id}/result" in paths
def test_source_requires_exactly_one_input(client) -> None:
response = client.post(
"/convert",
json={"source": {"url": "https://example.com/a.html", "html": "<html></html>"}},
)
assert response.status_code == 422
def test_empty_source_is_rejected(client) -> None:
response = client.post("/convert", json={"source": {}})
assert response.status_code == 422
def test_private_address_is_refused(client) -> None:
response = client.post("/convert", json={"source": {"url": "http://127.0.0.1:8000/a.html"}})
assert response.status_code == 400
assert response.json()["error_code"] == "blocked_target"
def test_unknown_job_returns_404(client) -> None:
response = client.get("/jobs/00000000-0000-0000-0000-000000000000")
assert response.status_code == 404
assert response.json()["error_code"] == "job_not_found"
+42
View File
@@ -0,0 +1,42 @@
"""Behaviour when an asset cannot be loaded.
A missing image must not abort the render, but it must never disappear
silently either.
"""
from __future__ import annotations
import pytest
pytest.importorskip("weasyprint")
from app.config import Settings # noqa: E402
from app.engines.weasy import WeasyPrintEngine # noqa: E402
from app.models import ConvertRequest, Source # noqa: E402
from app.services.pipeline import ConversionPipeline # noqa: E402
pytestmark = pytest.mark.asyncio
HTML_WITH_BLOCKED_IMAGE = """
<!DOCTYPE html><html><head><meta charset="utf-8"></head>
<body>
<h1>Dokument s nedostupnym obrazkem</h1>
<p>Text pokracuje i kdyz se obrazek nenacte.</p>
<img src="http://127.0.0.1:9/neexistuje.png" alt="chybi">
</body></html>
"""
async def test_blocked_asset_is_reported_and_render_continues(tmp_path) -> None:
pipeline = ConversionPipeline({"weasyprint": WeasyPrintEngine()}, Settings())
request = ConvertRequest(
source=Source(html=HTML_WITH_BLOCKED_IMAGE),
engine="weasyprint",
)
result = await pipeline.run(request, tmp_path)
assert result.page_count >= 1
assert result.path.exists()
assert result.missing_assets, "nedostupny asset musi byt hlaseny v odpovedi"
assert "127.0.0.1" in result.missing_assets[0].url
+56
View File
@@ -0,0 +1,56 @@
"""Splitting of the document into chunks."""
from __future__ import annotations
from app.pdf.chunker import split_document
from app.pdf.document import SourceDocument
from tests.conftest import build_document
def test_document_with_sections_is_split() -> None:
document = SourceDocument(build_document(sections=40, paragraphs_per_section=10))
chunks = split_document(document, pages_per_chunk=5)
assert len(chunks) > 1
for chunk in chunks:
assert chunk.startswith("<!DOCTYPE html>")
assert "<body" in chunk
assert "</html>" in chunk
def test_every_section_appears_exactly_once() -> None:
document = SourceDocument(build_document(sections=30, paragraphs_per_section=8))
chunks = split_document(document, pages_per_chunk=3)
joined = "".join(chunks)
for index in range(30):
assert joined.count(f"id=\"sekce-{index}\"") == 1
def test_head_is_repeated_in_every_chunk() -> None:
document = SourceDocument(build_document(sections=20, paragraphs_per_section=10))
chunks = split_document(document, pages_per_chunk=2)
assert len(chunks) > 1
for chunk in chunks:
assert "font-family:sans-serif" in chunk
def test_document_without_split_points_stays_whole(unsplittable_document: str) -> None:
document = SourceDocument(unsplittable_document)
chunks = split_document(document, pages_per_chunk=1)
assert len(chunks) == 1
def test_single_wrapper_is_traversed() -> None:
sections = "".join(f"<section><h1>Kapitola {i}</h1><p>Text</p></section>" for i in range(10))
html = f"<!DOCTYPE html><html><head></head><body><main class='obal'>{sections}</main></body></html>"
document = SourceDocument(html)
assert document.container.tag == "main"
chunks = split_document(document, pages_per_chunk=1)
assert len(chunks) > 1
for chunk in chunks:
assert "class=\"obal\"" in chunk
+133
View File
@@ -0,0 +1,133 @@
"""Job queue behaviour."""
from __future__ import annotations
import asyncio
from pathlib import Path
import pytest
from app.config import Settings
from app.errors import LimitExceededError, QueueFullError
from app.models import ConvertRequest, Source
from app.services.jobs import JobManager
from app.services.pipeline import ConversionResult
from app.services.storage import Storage
pytestmark = pytest.mark.asyncio
class StubPipeline:
def __init__(self, behaviour="ok", delay=0.0) -> None:
self.behaviour = behaviour
self.delay = delay
async def run(self, request, workdir: Path, progress=None) -> ConversionResult:
if progress is not None:
progress(1, 1, 1, 1)
if self.delay:
await asyncio.sleep(self.delay)
if self.behaviour == "conversion_error":
raise LimitExceededError("Prekrocen limit stranek.", {"limit": 10})
if self.behaviour == "crash":
raise RuntimeError("necekana chyba enginu")
workdir.mkdir(parents=True, exist_ok=True)
produced = workdir / "out.pdf"
produced.write_bytes(b"%PDF-1.7\n%fake\n")
return ConversionResult(path=produced, page_count=1, engine_used="stub")
def make_manager(tmp_path, behaviour="ok", delay=0.0, **overrides) -> JobManager:
settings = Settings()
for key, value in overrides.items():
object.__setattr__(settings, key, value)
object.__setattr__(settings, "storage_dir", str(tmp_path))
storage = Storage(str(tmp_path))
return JobManager(StubPipeline(behaviour, delay), storage, settings)
def simple_request() -> ConvertRequest:
return ConvertRequest(source=Source(html="<html><body><p>ahoj</p></body></html>"))
async def test_successful_job_publishes_a_result(tmp_path) -> None:
manager = make_manager(tmp_path)
await manager.start()
try:
job = manager.submit(simple_request())
await asyncio.wait_for(job.done.wait(), timeout=5)
assert job.state.status == "done"
assert job.state.page_count == 1
assert (Path(tmp_path) / job.id / "result.pdf").exists()
finally:
await manager.stop()
async def test_conversion_error_keeps_its_code(tmp_path) -> None:
manager = make_manager(tmp_path, behaviour="conversion_error")
await manager.start()
try:
job = manager.submit(simple_request())
await asyncio.wait_for(job.done.wait(), timeout=5)
assert job.state.status == "failed"
assert job.state.error is not None
assert job.state.error.error_code == "limit_exceeded"
finally:
await manager.stop()
async def test_unexpected_error_is_reported_not_swallowed(tmp_path) -> None:
manager = make_manager(tmp_path, behaviour="crash")
await manager.start()
try:
job = manager.submit(simple_request())
await asyncio.wait_for(job.done.wait(), timeout=5)
assert job.state.status == "failed"
assert job.state.error.error_code == "internal_error"
finally:
await manager.stop()
async def test_running_job_can_be_cancelled(tmp_path) -> None:
manager = make_manager(tmp_path, delay=5.0)
await manager.start()
try:
job = manager.submit(simple_request())
await asyncio.sleep(0.2)
manager.cancel(job.id)
await asyncio.wait_for(job.done.wait(), timeout=5)
assert job.state.status == "cancelled"
assert not (Path(tmp_path) / job.id).exists()
finally:
await manager.stop()
async def test_full_queue_is_rejected(tmp_path) -> None:
manager = make_manager(tmp_path, delay=5.0, queue_max_size=1, workers=1)
await manager.start()
try:
manager.submit(simple_request())
await asyncio.sleep(0.1)
manager.submit(simple_request())
with pytest.raises(QueueFullError):
manager.submit(simple_request())
finally:
await manager.stop()
async def test_shutdown_marks_pending_jobs_as_failed(tmp_path) -> None:
manager = make_manager(tmp_path, delay=5.0, workers=1)
await manager.start()
job = manager.submit(simple_request())
await asyncio.sleep(0.1)
await manager.stop()
assert job.state.status == "failed"
assert job.state.error.error_code == "service_restarted"
+39
View File
@@ -0,0 +1,39 @@
"""Measurement fixture for a document of roughly twelve hundred pages.
Marked slow on purpose. It is here to measure memory and wall clock time of the
chunked pipeline, not to assert exact numbers.
"""
from __future__ import annotations
import time
import pytest
pytest.importorskip("weasyprint")
from app.config import Settings # noqa: E402
from app.engines.weasy import WeasyPrintEngine # noqa: E402
from app.models import ChunkSettings, ConvertRequest, PageNumbers, Source # noqa: E402
from app.services.pipeline import ConversionPipeline # noqa: E402
pytestmark = [pytest.mark.asyncio, pytest.mark.slow]
async def test_large_document_renders_in_chunks(tmp_path, large_document: str) -> None:
pipeline = ConversionPipeline({"weasyprint": WeasyPrintEngine()}, Settings())
request = ConvertRequest(
source=Source(html=large_document),
engine="weasyprint",
page_numbers=PageNumbers(enabled=True),
chunking=ChunkSettings(enabled=True, pages_per_chunk=50),
)
started = time.monotonic()
result = await pipeline.run(request, tmp_path)
duration = time.monotonic() - started
print(f"stranek: {result.page_count}, cas: {duration:.1f} s")
assert result.page_count > 1000
assert result.path.exists()
+90
View File
@@ -0,0 +1,90 @@
"""Page numbering and table of contents of a merged document.
This is where the chunking approach usually breaks, so the numbers are read back
out of the produced PDF instead of being trusted.
"""
from __future__ import annotations
import re
import pytest
pytest.importorskip("weasyprint")
pytest.importorskip("pypdf")
from app.config import Settings # noqa: E402
from app.engines.weasy import WeasyPrintEngine # noqa: E402
from app.models import ChunkSettings, ConvertRequest, PageNumbers, Source, TocSettings # noqa: E402
from app.services.pipeline import ConversionPipeline # noqa: E402
from tests.conftest import build_document # noqa: E402
pytestmark = pytest.mark.asyncio
def page_texts(path) -> list[str]:
from pypdf import PdfReader
with open(path, "rb") as handle:
return [page.extract_text() or "" for page in PdfReader(handle).pages]
def make_pipeline() -> ConversionPipeline:
return ConversionPipeline({"weasyprint": WeasyPrintEngine()}, Settings())
async def test_page_numbers_are_continuous_across_chunks(tmp_path) -> None:
request = ConvertRequest(
source=Source(html=build_document(sections=12, paragraphs_per_section=8)),
engine="weasyprint",
page_numbers=PageNumbers(enabled=True, format="{page} / {pages}"),
chunking=ChunkSettings(enabled=True, pages_per_chunk=2),
)
result = await make_pipeline().run(request, tmp_path)
texts = page_texts(result.path)
assert result.page_count == len(texts)
assert result.page_count > 3, "dokument musi mit vic stranek, jinak test nic neoveruje"
for index, text in enumerate(texts, start=1):
assert f"{index} / {result.page_count}" in text.replace("\n", " ")
async def test_table_of_contents_points_to_the_real_pages(tmp_path) -> None:
request = ConvertRequest(
source=Source(html=build_document(sections=10, paragraphs_per_section=8)),
engine="weasyprint",
toc=TocSettings(enabled=True, depth=1, title="Obsah"),
chunking=ChunkSettings(enabled=True, pages_per_chunk=2),
)
result = await make_pipeline().run(request, tmp_path)
texts = page_texts(result.path)
toc_text = " ".join(texts[:2]).replace("\n", " ")
for index in range(10):
heading = f"Kapitola {index + 1}"
match = re.search(re.escape(heading) + r"\s+(\d+)", toc_text)
assert match, f"v obsahu chybi polozka {heading}"
declared_page = int(match.group(1))
actual_pages = [
number for number, text in enumerate(texts, start=1) if heading in text.replace("\n", " ")
]
# The first occurrence after the table of contents is the heading itself.
assert declared_page in actual_pages, f"{heading} deklaruje stranku {declared_page}"
async def test_unchunked_document_uses_css_counters(tmp_path) -> None:
request = ConvertRequest(
source=Source(html=build_document(sections=4, paragraphs_per_section=6)),
engine="weasyprint",
page_numbers=PageNumbers(enabled=True, format="{page} / {pages}"),
chunking=ChunkSettings(enabled=False),
)
result = await make_pipeline().run(request, tmp_path)
texts = page_texts(result.path)
assert f"1 / {result.page_count}" in texts[0].replace("\n", " ")
+65
View File
@@ -0,0 +1,65 @@
"""SSRF protection.
The important case is a public hostname that resolves to a loopback address.
Checking the URL string alone would let it through.
"""
from __future__ import annotations
import socket
import pytest
from app.config import Settings
from app.errors import BlockedTargetError
from app.services.security import UrlGuard
@pytest.fixture
def guard() -> UrlGuard:
return UrlGuard(Settings())
def test_literal_loopback_is_blocked(guard: UrlGuard) -> None:
with pytest.raises(BlockedTargetError):
guard.check("http://127.0.0.1:8000/dokument.html")
def test_private_range_is_blocked(guard: UrlGuard) -> None:
with pytest.raises(BlockedTargetError):
guard.check("http://192.168.1.10/dokument.html")
def test_link_local_metadata_endpoint_is_blocked(guard: UrlGuard) -> None:
with pytest.raises(BlockedTargetError):
guard.check("http://169.254.169.254/latest/meta-data/")
def test_hostname_resolving_to_loopback_is_blocked(guard: UrlGuard, monkeypatch) -> None:
def fake_getaddrinfo(host, *args, **kwargs): # noqa: ARG001
return [(socket.AF_INET, socket.SOCK_STREAM, 6, "", ("127.0.0.1", 80))]
monkeypatch.setattr(socket, "getaddrinfo", fake_getaddrinfo)
with pytest.raises(BlockedTargetError):
guard.check("http://vlastni-domena.example.com/dokument.html")
def test_file_scheme_is_blocked(guard: UrlGuard) -> None:
with pytest.raises(BlockedTargetError):
guard.check("file:///etc/passwd")
def test_public_address_passes(guard: UrlGuard, monkeypatch) -> None:
def fake_getaddrinfo(host, *args, **kwargs): # noqa: ARG001
return [(socket.AF_INET, socket.SOCK_STREAM, 6, "", ("93.184.216.34", 80))]
monkeypatch.setattr(socket, "getaddrinfo", fake_getaddrinfo)
assert guard.check("https://example.com/dokument.html") == "example.com"
def test_allowlisted_host_skips_the_check() -> None:
settings = Settings()
object.__setattr__(settings, "ssrf_allowed_hosts", ["localhost"])
guard = UrlGuard(settings)
assert guard.check("http://localhost:9000/dokument.html") == "localhost"
+37
View File
@@ -0,0 +1,37 @@
"""Building of the page stylesheet."""
from __future__ import annotations
from app.models import PageNumbers, PageSettings
from app.pdf.styles import build_page_css, css_content_value, page_size_value
def test_named_format_keeps_orientation() -> None:
assert page_size_value(PageSettings(format="A4", orientation="landscape")) == "A4 landscape"
def test_explicit_dimensions_are_used_as_is() -> None:
assert page_size_value(PageSettings(format="210mm 297mm")) == "210mm 297mm"
def test_content_uses_counters_when_the_total_is_unknown() -> None:
assert css_content_value("{page} / {pages}", None) == 'counter(page) " / " counter(pages)'
def test_content_uses_a_literal_when_the_total_is_known() -> None:
assert css_content_value("Strana {page} z {pages}", 120) == '"Strana " counter(page) " z " "120"'
def test_page_numbers_are_absent_when_disabled() -> None:
css = build_page_css(PageSettings(), PageNumbers(enabled=False))
assert "@bottom-center" not in css
def test_page_numbers_land_in_the_requested_box() -> None:
css = build_page_css(PageSettings(), PageNumbers(enabled=True, position="top-right"))
assert "@top-right" in css
def test_bookmarks_can_be_switched_off() -> None:
css = build_page_css(PageSettings(), None, outline=False)
assert "bookmark-level: none" in css