Implementace prevodu HTML na PDF
Sluzba prijme adresu HTML dokumentu nebo HTML v tele requestu a vrati PDF. Navrzena pro dokumenty o stovkach az tisicich stranek. Rendering: - WeasyPrint jako vychozi engine, spravne CSS Paged Media, nizka pametova narocnost, bez JavaScriptu - Chromium pres Playwright pro dokumenty dokreslovane skripty - rezim auto s detekci skriptu a fallbackem pri selhani WeasyPrintu Velke dokumenty: - deleni na casti na strukturalnich hranicich, rez nikdy uvnitr tabulky nebo odstavce - dvoupruchodovy render obsahu se skutecnymi cisly stranek, pozice nadpisu se ctou z kotev hlasenych u kazde stranky - cislovani stranek bud pres CSS countery, nebo pres cislovaci vrstvu nastampovanou na hotove PDF, rozmer stranky se cte z vysledneho souboru - Chromium se restartuje po N jobech, nikdy vsak behem beziciho renderu API: - POST /convert synchronne, POST /jobs asynchronne se sledovanim stavu, stahovanim vysledku, rusenim a volitelnym callbackem - GET /health s overenim dostupnosti obou enginu a stavem fronty - OpenAPI respektuje prefix reverse proxy pres root_path Bezpecnost a provoz: - SSRF kontrola po DNS resolvu, na kazdem presmerovani a u vsech pozadavku prohlizece - nedostupne assety render nezastavi, ale hlasi se v odpovedi i v logu - fronta s omezenym poctem workeru, rozpracovane joby se pri ukonceni oznaci jako failed, nezmizi potichu - strukturovane JSON logovani s job_id - vsechny limity vypnute ve vychozim stavu Dockerfile je dvoufazovy, obsahuje zavislosti WeasyPrintu, Chromium a fonty s ceskou diakritikou. Autentizace zamerne neni implementovana, zpusob predavani neni domluveny. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
e3cc8f418b
commit
156289fe2d
@@ -0,0 +1,48 @@
|
||||
"""Common interface of the render engines."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from abc import ABC, abstractmethod
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
from ..models import ConvertRequest
|
||||
|
||||
|
||||
@dataclass
|
||||
class ChunkRender:
|
||||
"""Result of rendering a single chunk."""
|
||||
|
||||
path: Path
|
||||
page_count: int
|
||||
# Anchor name to zero based page index inside this chunk.
|
||||
anchor_pages: dict[str, int] = field(default_factory=dict)
|
||||
|
||||
|
||||
class RenderEngine(ABC):
|
||||
name: str
|
||||
|
||||
@abstractmethod
|
||||
async def available(self) -> bool:
|
||||
"""True when the engine can render right now."""
|
||||
|
||||
@abstractmethod
|
||||
async def render_chunk(
|
||||
self,
|
||||
html: str,
|
||||
base_url: str | None,
|
||||
request: ConvertRequest,
|
||||
output_path: Path,
|
||||
total_pages: int | None = None,
|
||||
asset_gate=None,
|
||||
) -> ChunkRender:
|
||||
"""Render one chunk into output_path."""
|
||||
|
||||
async def shutdown(self) -> None:
|
||||
"""Release engine resources. Default is a no-op."""
|
||||
return None
|
||||
|
||||
@property
|
||||
def supports_anchor_pages(self) -> bool:
|
||||
"""True when the engine can report on which page an anchor landed."""
|
||||
return False
|
||||
@@ -0,0 +1,211 @@
|
||||
"""Headless Chromium engine driven by Playwright.
|
||||
|
||||
Used for foreign URLs and for documents that are completed by JavaScript.
|
||||
Chromium has no usable CSS page counters, so page numbering for this engine is
|
||||
always done by the overlay in the post processing step.
|
||||
|
||||
The browser leaks memory over time, therefore the whole instance is restarted
|
||||
after a configurable number of jobs.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
from ..config import get_settings
|
||||
from ..errors import EngineUnavailableError, RenderTimeoutError
|
||||
from ..models import ConvertRequest
|
||||
from ..pdf.styles import NAMED_SIZE
|
||||
from .base import ChunkRender, RenderEngine
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
LAUNCH_ARGS = [
|
||||
"--no-sandbox",
|
||||
"--disable-dev-shm-usage",
|
||||
"--disable-gpu",
|
||||
"--hide-scrollbars",
|
||||
]
|
||||
|
||||
|
||||
class ChromiumEngine(RenderEngine):
|
||||
name = "chromium"
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._settings = get_settings()
|
||||
self._lock = asyncio.Lock()
|
||||
self._playwright = None
|
||||
self._browser = None
|
||||
self._jobs_since_launch = 0
|
||||
self._active_renders = 0
|
||||
|
||||
async def available(self) -> bool:
|
||||
if not self._settings.chromium_enabled:
|
||||
return False
|
||||
try:
|
||||
await self._ensure_browser(reserve=False)
|
||||
except Exception as exc: # noqa: BLE001 - reported, never silent
|
||||
logger.error("Chromium is not available", exc_info=exc)
|
||||
return False
|
||||
return True
|
||||
|
||||
async def _ensure_browser(self, reserve: bool = True):
|
||||
"""Return a running browser, restarting it when that is safe.
|
||||
|
||||
The restart must never happen while a render is in flight, otherwise the
|
||||
health check could close the browser under a running job.
|
||||
"""
|
||||
async with self._lock:
|
||||
due_for_restart = self._jobs_since_launch >= self._settings.chromium_restart_after_jobs
|
||||
if self._browser is not None and self._active_renders == 0 and due_for_restart:
|
||||
logger.info(
|
||||
"Restarting Chromium to release memory",
|
||||
extra={"jobs_since_launch": self._jobs_since_launch},
|
||||
)
|
||||
await self._close_browser()
|
||||
|
||||
if self._browser is None:
|
||||
try:
|
||||
from playwright.async_api import async_playwright
|
||||
except ImportError as exc:
|
||||
raise EngineUnavailableError(
|
||||
"Engine chromium neni k dispozici, chybi balicek playwright.",
|
||||
) from exc
|
||||
|
||||
self._playwright = await async_playwright().start()
|
||||
self._browser = await self._playwright.chromium.launch(args=LAUNCH_ARGS)
|
||||
self._jobs_since_launch = 0
|
||||
logger.info("Chromium launched")
|
||||
|
||||
if reserve:
|
||||
self._active_renders += 1
|
||||
|
||||
return self._browser
|
||||
|
||||
async def _close_browser(self) -> None:
|
||||
if self._browser is not None:
|
||||
try:
|
||||
await self._browser.close()
|
||||
except Exception as exc: # noqa: BLE001 - reported, never silent
|
||||
logger.warning("Closing Chromium failed", exc_info=exc)
|
||||
self._browser = None
|
||||
if self._playwright is not None:
|
||||
try:
|
||||
await self._playwright.stop()
|
||||
except Exception as exc: # noqa: BLE001 - reported, never silent
|
||||
logger.warning("Stopping Playwright failed", exc_info=exc)
|
||||
self._playwright = None
|
||||
|
||||
async def shutdown(self) -> None:
|
||||
async with self._lock:
|
||||
await self._close_browser()
|
||||
|
||||
def on_job_finished(self) -> None:
|
||||
self._jobs_since_launch += 1
|
||||
|
||||
async def render_chunk(
|
||||
self,
|
||||
html: str | None,
|
||||
base_url: str | None,
|
||||
request: ConvertRequest,
|
||||
output_path: Path,
|
||||
total_pages: int | None = None,
|
||||
asset_gate=None,
|
||||
) -> ChunkRender:
|
||||
browser = await self._ensure_browser()
|
||||
try:
|
||||
context = await browser.new_context()
|
||||
except Exception:
|
||||
self._active_renders -= 1
|
||||
raise
|
||||
|
||||
try:
|
||||
if asset_gate is not None:
|
||||
await context.route("**/*", _make_route_handler(asset_gate))
|
||||
|
||||
page = await context.new_page()
|
||||
timeout_ms = request.wait_for.timeout_seconds * 1000
|
||||
|
||||
try:
|
||||
if html is None:
|
||||
if not base_url:
|
||||
raise EngineUnavailableError("Chybi adresa dokumentu pro engine chromium.")
|
||||
await page.goto(base_url, wait_until=request.wait_for.state, timeout=timeout_ms)
|
||||
else:
|
||||
await page.set_content(html, wait_until=request.wait_for.state, timeout=timeout_ms)
|
||||
|
||||
if request.wait_for.selector:
|
||||
await page.wait_for_selector(request.wait_for.selector, timeout=timeout_ms)
|
||||
except Exception as exc: # noqa: BLE001 - converted to a typed error
|
||||
if "Timeout" in type(exc).__name__ or "timeout" in str(exc).lower():
|
||||
raise RenderTimeoutError(
|
||||
"Chromium nestihl nacist dokument v zadanem casovem limitu.",
|
||||
{"timeout_seconds": request.wait_for.timeout_seconds},
|
||||
) from exc
|
||||
raise
|
||||
|
||||
await page.emulate_media(media="print")
|
||||
pdf_kwargs = _pdf_kwargs(request)
|
||||
|
||||
try:
|
||||
pdf_bytes = await page.pdf(outline=request.outline, **pdf_kwargs)
|
||||
except TypeError:
|
||||
logger.warning("Installed Playwright does not support the outline option, continuing without bookmarks")
|
||||
pdf_bytes = await page.pdf(**pdf_kwargs)
|
||||
|
||||
output_path.write_bytes(pdf_bytes)
|
||||
finally:
|
||||
self._active_renders -= 1
|
||||
try:
|
||||
await context.close()
|
||||
except Exception as exc:
|
||||
logger.warning("Closing the browser context failed", exc_info=exc)
|
||||
|
||||
return ChunkRender(path=output_path, page_count=_count_pages(output_path))
|
||||
|
||||
|
||||
def _pdf_kwargs(request: ConvertRequest) -> dict:
|
||||
margin = request.page.margin
|
||||
kwargs: dict = {
|
||||
"print_background": True,
|
||||
"prefer_css_page_size": True,
|
||||
"margin": {
|
||||
"top": margin.top,
|
||||
"right": margin.right,
|
||||
"bottom": margin.bottom,
|
||||
"left": margin.left,
|
||||
},
|
||||
}
|
||||
fmt = request.page.format.strip()
|
||||
if NAMED_SIZE.match(fmt):
|
||||
kwargs["format"] = fmt
|
||||
else:
|
||||
width, _, height = fmt.partition(" ")
|
||||
if width and height:
|
||||
kwargs["width"] = width.strip()
|
||||
kwargs["height"] = height.strip()
|
||||
else:
|
||||
kwargs["format"] = fmt
|
||||
kwargs["landscape"] = request.page.orientation == "landscape"
|
||||
return kwargs
|
||||
|
||||
|
||||
def _make_route_handler(asset_gate):
|
||||
async def handler(route, playwright_request):
|
||||
allowed, reason = asset_gate.allowed(playwright_request.url)
|
||||
if allowed:
|
||||
await route.continue_()
|
||||
return
|
||||
asset_gate.report.add(playwright_request.url, reason)
|
||||
await route.abort()
|
||||
|
||||
return handler
|
||||
|
||||
|
||||
def _count_pages(path: Path) -> int:
|
||||
from pypdf import PdfReader
|
||||
|
||||
with path.open("rb") as handle:
|
||||
return len(PdfReader(handle).pages)
|
||||
@@ -0,0 +1,107 @@
|
||||
"""WeasyPrint engine.
|
||||
|
||||
Default engine for documents we generate ourselves. It implements CSS Paged
|
||||
Media properly, which is what makes counters, running headers and repeated table
|
||||
headers work, and it uses far less memory than a headless browser.
|
||||
|
||||
It does not execute JavaScript. That is intentional.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
from ..models import ConvertRequest
|
||||
from ..pdf.styles import build_page_css
|
||||
from .base import ChunkRender, RenderEngine
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class WeasyPrintEngine(RenderEngine):
|
||||
name = "weasyprint"
|
||||
|
||||
@property
|
||||
def supports_anchor_pages(self) -> bool:
|
||||
return True
|
||||
|
||||
async def available(self) -> bool:
|
||||
try:
|
||||
import weasyprint # noqa: F401
|
||||
except Exception as exc: # noqa: BLE001 - reported, never silent
|
||||
logger.error("WeasyPrint is not importable", exc_info=exc)
|
||||
return False
|
||||
return True
|
||||
|
||||
async def render_chunk(
|
||||
self,
|
||||
html: str,
|
||||
base_url: str | None,
|
||||
request: ConvertRequest,
|
||||
output_path: Path,
|
||||
total_pages: int | None = None,
|
||||
asset_gate=None,
|
||||
) -> ChunkRender:
|
||||
return await asyncio.to_thread(
|
||||
self._render_blocking, html, base_url, request, output_path, total_pages, asset_gate
|
||||
)
|
||||
|
||||
def _render_blocking(
|
||||
self,
|
||||
html: str,
|
||||
base_url: str | None,
|
||||
request: ConvertRequest,
|
||||
output_path: Path,
|
||||
total_pages: int | None,
|
||||
asset_gate,
|
||||
) -> ChunkRender:
|
||||
from weasyprint import CSS, HTML
|
||||
|
||||
kwargs = {"string": html, "base_url": base_url}
|
||||
if asset_gate is not None:
|
||||
kwargs["url_fetcher"] = asset_gate.weasy_fetcher()
|
||||
|
||||
stylesheets = []
|
||||
if total_pages is not None:
|
||||
# Only used when the page total is known up front, which happens on
|
||||
# the second pass of an unchunked document.
|
||||
stylesheets.append(
|
||||
CSS(
|
||||
string=build_page_css(
|
||||
request.page,
|
||||
request.page_numbers,
|
||||
total_pages=total_pages,
|
||||
outline=request.outline,
|
||||
)
|
||||
)
|
||||
)
|
||||
|
||||
document = HTML(**kwargs).render(stylesheets=stylesheets or None)
|
||||
|
||||
write_kwargs = {}
|
||||
if request.pdf_profile:
|
||||
write_kwargs["pdf_variant"] = request.pdf_profile
|
||||
|
||||
document.write_pdf(target=str(output_path), **write_kwargs)
|
||||
|
||||
return ChunkRender(
|
||||
path=output_path,
|
||||
page_count=len(document.pages),
|
||||
anchor_pages=self._anchor_pages(document),
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _anchor_pages(document) -> dict[str, int]:
|
||||
"""Map anchor name to the zero based page index it landed on."""
|
||||
anchors: dict[str, int] = {}
|
||||
for index, page in enumerate(document.pages):
|
||||
page_anchors = getattr(page, "anchors", None)
|
||||
if not page_anchors:
|
||||
continue
|
||||
for name in page_anchors:
|
||||
anchors.setdefault(name, index)
|
||||
if not anchors:
|
||||
logger.warning("WeasyPrint returned no anchors, table of contents page numbers may be missing")
|
||||
return anchors
|
||||
Reference in New Issue
Block a user