Sluzba prijme adresu HTML dokumentu nebo HTML v tele requestu a vrati PDF. Navrzena pro dokumenty o stovkach az tisicich stranek. Rendering: - WeasyPrint jako vychozi engine, spravne CSS Paged Media, nizka pametova narocnost, bez JavaScriptu - Chromium pres Playwright pro dokumenty dokreslovane skripty - rezim auto s detekci skriptu a fallbackem pri selhani WeasyPrintu Velke dokumenty: - deleni na casti na strukturalnich hranicich, rez nikdy uvnitr tabulky nebo odstavce - dvoupruchodovy render obsahu se skutecnymi cisly stranek, pozice nadpisu se ctou z kotev hlasenych u kazde stranky - cislovani stranek bud pres CSS countery, nebo pres cislovaci vrstvu nastampovanou na hotove PDF, rozmer stranky se cte z vysledneho souboru - Chromium se restartuje po N jobech, nikdy vsak behem beziciho renderu API: - POST /convert synchronne, POST /jobs asynchronne se sledovanim stavu, stahovanim vysledku, rusenim a volitelnym callbackem - GET /health s overenim dostupnosti obou enginu a stavem fronty - OpenAPI respektuje prefix reverse proxy pres root_path Bezpecnost a provoz: - SSRF kontrola po DNS resolvu, na kazdem presmerovani a u vsech pozadavku prohlizece - nedostupne assety render nezastavi, ale hlasi se v odpovedi i v logu - fronta s omezenym poctem workeru, rozpracovane joby se pri ukonceni oznaci jako failed, nezmizi potichu - strukturovane JSON logovani s job_id - vsechny limity vypnute ve vychozim stavu Dockerfile je dvoufazovy, obsahuje zavislosti WeasyPrintu, Chromium a fonty s ceskou diakritikou. Autentizace zamerne neni implementovana, zpusob predavani neni domluveny. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
108 lines
3.3 KiB
Python
108 lines
3.3 KiB
Python
"""WeasyPrint engine.
|
|
|
|
Default engine for documents we generate ourselves. It implements CSS Paged
|
|
Media properly, which is what makes counters, running headers and repeated table
|
|
headers work, and it uses far less memory than a headless browser.
|
|
|
|
It does not execute JavaScript. That is intentional.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import logging
|
|
from pathlib import Path
|
|
|
|
from ..models import ConvertRequest
|
|
from ..pdf.styles import build_page_css
|
|
from .base import ChunkRender, RenderEngine
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
class WeasyPrintEngine(RenderEngine):
|
|
name = "weasyprint"
|
|
|
|
@property
|
|
def supports_anchor_pages(self) -> bool:
|
|
return True
|
|
|
|
async def available(self) -> bool:
|
|
try:
|
|
import weasyprint # noqa: F401
|
|
except Exception as exc: # noqa: BLE001 - reported, never silent
|
|
logger.error("WeasyPrint is not importable", exc_info=exc)
|
|
return False
|
|
return True
|
|
|
|
async def render_chunk(
|
|
self,
|
|
html: str,
|
|
base_url: str | None,
|
|
request: ConvertRequest,
|
|
output_path: Path,
|
|
total_pages: int | None = None,
|
|
asset_gate=None,
|
|
) -> ChunkRender:
|
|
return await asyncio.to_thread(
|
|
self._render_blocking, html, base_url, request, output_path, total_pages, asset_gate
|
|
)
|
|
|
|
def _render_blocking(
|
|
self,
|
|
html: str,
|
|
base_url: str | None,
|
|
request: ConvertRequest,
|
|
output_path: Path,
|
|
total_pages: int | None,
|
|
asset_gate,
|
|
) -> ChunkRender:
|
|
from weasyprint import CSS, HTML
|
|
|
|
kwargs = {"string": html, "base_url": base_url}
|
|
if asset_gate is not None:
|
|
kwargs["url_fetcher"] = asset_gate.weasy_fetcher()
|
|
|
|
stylesheets = []
|
|
if total_pages is not None:
|
|
# Only used when the page total is known up front, which happens on
|
|
# the second pass of an unchunked document.
|
|
stylesheets.append(
|
|
CSS(
|
|
string=build_page_css(
|
|
request.page,
|
|
request.page_numbers,
|
|
total_pages=total_pages,
|
|
outline=request.outline,
|
|
)
|
|
)
|
|
)
|
|
|
|
document = HTML(**kwargs).render(stylesheets=stylesheets or None)
|
|
|
|
write_kwargs = {}
|
|
if request.pdf_profile:
|
|
write_kwargs["pdf_variant"] = request.pdf_profile
|
|
|
|
document.write_pdf(target=str(output_path), **write_kwargs)
|
|
|
|
return ChunkRender(
|
|
path=output_path,
|
|
page_count=len(document.pages),
|
|
anchor_pages=self._anchor_pages(document),
|
|
)
|
|
|
|
@staticmethod
|
|
def _anchor_pages(document) -> dict[str, int]:
|
|
"""Map anchor name to the zero based page index it landed on."""
|
|
anchors: dict[str, int] = {}
|
|
for index, page in enumerate(document.pages):
|
|
page_anchors = getattr(page, "anchors", None)
|
|
if not page_anchors:
|
|
continue
|
|
for name in page_anchors:
|
|
anchors.setdefault(name, index)
|
|
if not anchors:
|
|
logger.warning("WeasyPrint returned no anchors, table of contents page numbers may be missing")
|
|
return anchors
|