Implementace prevodu HTML na PDF
Sluzba prijme adresu HTML dokumentu nebo HTML v tele requestu a vrati PDF. Navrzena pro dokumenty o stovkach az tisicich stranek. Rendering: - WeasyPrint jako vychozi engine, spravne CSS Paged Media, nizka pametova narocnost, bez JavaScriptu - Chromium pres Playwright pro dokumenty dokreslovane skripty - rezim auto s detekci skriptu a fallbackem pri selhani WeasyPrintu Velke dokumenty: - deleni na casti na strukturalnich hranicich, rez nikdy uvnitr tabulky nebo odstavce - dvoupruchodovy render obsahu se skutecnymi cisly stranek, pozice nadpisu se ctou z kotev hlasenych u kazde stranky - cislovani stranek bud pres CSS countery, nebo pres cislovaci vrstvu nastampovanou na hotove PDF, rozmer stranky se cte z vysledneho souboru - Chromium se restartuje po N jobech, nikdy vsak behem beziciho renderu API: - POST /convert synchronne, POST /jobs asynchronne se sledovanim stavu, stahovanim vysledku, rusenim a volitelnym callbackem - GET /health s overenim dostupnosti obou enginu a stavem fronty - OpenAPI respektuje prefix reverse proxy pres root_path Bezpecnost a provoz: - SSRF kontrola po DNS resolvu, na kazdem presmerovani a u vsech pozadavku prohlizece - nedostupne assety render nezastavi, ale hlasi se v odpovedi i v logu - fronta s omezenym poctem workeru, rozpracovane joby se pri ukonceni oznaci jako failed, nezmizi potichu - strukturovane JSON logovani s job_id - vsechny limity vypnute ve vychozim stavu Dockerfile je dvoufazovy, obsahuje zavislosti WeasyPrintu, Chromium a fonty s ceskou diakritikou. Autentizace zamerne neni implementovana, zpusob predavani neni domluveny. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
e3cc8f418b
commit
156289fe2d
@@ -0,0 +1,107 @@
|
||||
"""WeasyPrint engine.
|
||||
|
||||
Default engine for documents we generate ourselves. It implements CSS Paged
|
||||
Media properly, which is what makes counters, running headers and repeated table
|
||||
headers work, and it uses far less memory than a headless browser.
|
||||
|
||||
It does not execute JavaScript. That is intentional.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
from ..models import ConvertRequest
|
||||
from ..pdf.styles import build_page_css
|
||||
from .base import ChunkRender, RenderEngine
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class WeasyPrintEngine(RenderEngine):
|
||||
name = "weasyprint"
|
||||
|
||||
@property
|
||||
def supports_anchor_pages(self) -> bool:
|
||||
return True
|
||||
|
||||
async def available(self) -> bool:
|
||||
try:
|
||||
import weasyprint # noqa: F401
|
||||
except Exception as exc: # noqa: BLE001 - reported, never silent
|
||||
logger.error("WeasyPrint is not importable", exc_info=exc)
|
||||
return False
|
||||
return True
|
||||
|
||||
async def render_chunk(
|
||||
self,
|
||||
html: str,
|
||||
base_url: str | None,
|
||||
request: ConvertRequest,
|
||||
output_path: Path,
|
||||
total_pages: int | None = None,
|
||||
asset_gate=None,
|
||||
) -> ChunkRender:
|
||||
return await asyncio.to_thread(
|
||||
self._render_blocking, html, base_url, request, output_path, total_pages, asset_gate
|
||||
)
|
||||
|
||||
def _render_blocking(
|
||||
self,
|
||||
html: str,
|
||||
base_url: str | None,
|
||||
request: ConvertRequest,
|
||||
output_path: Path,
|
||||
total_pages: int | None,
|
||||
asset_gate,
|
||||
) -> ChunkRender:
|
||||
from weasyprint import CSS, HTML
|
||||
|
||||
kwargs = {"string": html, "base_url": base_url}
|
||||
if asset_gate is not None:
|
||||
kwargs["url_fetcher"] = asset_gate.weasy_fetcher()
|
||||
|
||||
stylesheets = []
|
||||
if total_pages is not None:
|
||||
# Only used when the page total is known up front, which happens on
|
||||
# the second pass of an unchunked document.
|
||||
stylesheets.append(
|
||||
CSS(
|
||||
string=build_page_css(
|
||||
request.page,
|
||||
request.page_numbers,
|
||||
total_pages=total_pages,
|
||||
outline=request.outline,
|
||||
)
|
||||
)
|
||||
)
|
||||
|
||||
document = HTML(**kwargs).render(stylesheets=stylesheets or None)
|
||||
|
||||
write_kwargs = {}
|
||||
if request.pdf_profile:
|
||||
write_kwargs["pdf_variant"] = request.pdf_profile
|
||||
|
||||
document.write_pdf(target=str(output_path), **write_kwargs)
|
||||
|
||||
return ChunkRender(
|
||||
path=output_path,
|
||||
page_count=len(document.pages),
|
||||
anchor_pages=self._anchor_pages(document),
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _anchor_pages(document) -> dict[str, int]:
|
||||
"""Map anchor name to the zero based page index it landed on."""
|
||||
anchors: dict[str, int] = {}
|
||||
for index, page in enumerate(document.pages):
|
||||
page_anchors = getattr(page, "anchors", None)
|
||||
if not page_anchors:
|
||||
continue
|
||||
for name in page_anchors:
|
||||
anchors.setdefault(name, index)
|
||||
if not anchors:
|
||||
logger.warning("WeasyPrint returned no anchors, table of contents page numbers may be missing")
|
||||
return anchors
|
||||
Reference in New Issue
Block a user