Files
2026-09-03 10:03:10 +02:00

354 lines
13 KiB
Python

"""The conversion pipeline.
Order of operations:
1. load the source HTML, validating the target address
2. pick the engine
3. decide how page numbers will be produced
4. build the document, inject the page stylesheet, optionally insert the table
of contents placeholder
5. split into chunks
6. first render pass, which yields the real page count and anchor positions
7. second render pass when a table of contents needs real page numbers
8. merge the chunks
9. stamp the numbering overlay when CSS counters cannot be used
"""
from __future__ import annotations
import asyncio
import logging
import re
from dataclasses import dataclass, field
from pathlib import Path
from typing import Callable
from ..config import Settings, get_settings
from ..errors import (
ConversionError,
LimitExceededError,
RenderTimeoutError,
UnsupportedCombinationError,
)
from ..models import ConvertRequest, MissingAsset
from ..pdf import merger, paginator
from ..pdf.chunker import split_document
from ..pdf.document import SourceDocument
from ..pdf.styles import build_page_css
from .fetcher import AssetGate, AssetReport, fetch_document
from .security import UrlGuard
logger = logging.getLogger(__name__)
SCRIPT_TAG = re.compile(r"<script\b[^>]*>(.*?)</script>", re.IGNORECASE | re.DOTALL)
SCRIPT_SRC = re.compile(r"<script\b[^>]*\bsrc\s*=", re.IGNORECASE)
ProgressCallback = Callable[[int, int, int, int], None]
@dataclass
class ConversionResult:
path: Path
page_count: int
engine_used: str
missing_assets: list[MissingAsset] = field(default_factory=list)
warnings: list[str] = field(default_factory=list)
class ConversionPipeline:
def __init__(self, engines: dict, settings: Settings | None = None) -> None:
self._engines = engines
self._settings = settings or get_settings()
self._guard = UrlGuard(self._settings)
async def run(
self,
request: ConvertRequest,
workdir: Path,
progress: ProgressCallback | None = None,
) -> ConversionResult:
workdir.mkdir(parents=True, exist_ok=True)
timeout = self._settings.max_render_seconds
coroutine = self._run_with_fallback(request, workdir, progress)
if timeout:
try:
return await asyncio.wait_for(coroutine, timeout=timeout)
except asyncio.TimeoutError as exc:
raise RenderTimeoutError(
"Render prekrocil nakonfigurovany limit MAX_RENDER_SECONDS.",
{"limit_seconds": timeout},
) from exc
return await coroutine
async def _run_with_fallback(
self, request: ConvertRequest, workdir: Path, progress: ProgressCallback | None
) -> ConversionResult:
raw_html, base_url = await self._load_source(request)
requested = request.engine
engine_name = self._select_engine(requested, raw_html)
try:
return await self._execute(request, raw_html, base_url, engine_name, workdir, progress)
except ConversionError:
raise
except Exception as exc:
if requested != "auto" or engine_name != "weasyprint" or "chromium" not in self._engines:
raise
logger.warning(
"WeasyPrint render failed, falling back to Chromium",
exc_info=exc,
extra={"engine": engine_name},
)
result = await self._execute(request, raw_html, base_url, "chromium", workdir, progress)
result.warnings.append(
"Render pres WeasyPrint selhal, dokument byl vygenerovan pres Chromium."
)
return result
def _select_engine(self, requested: str, raw_html: str) -> str:
if requested != "auto":
if requested not in self._engines:
raise UnsupportedCombinationError(
f"Engine {requested} neni v teto instanci k dispozici.",
{"available": sorted(self._engines)},
)
return requested
if self._has_active_scripts(raw_html) and "chromium" in self._engines:
logger.info("Auto engine selected chromium because the document contains scripts")
return "chromium"
if "weasyprint" in self._engines:
return "weasyprint"
return next(iter(self._engines))
@staticmethod
def _has_active_scripts(raw_html: str) -> bool:
if SCRIPT_SRC.search(raw_html):
return True
return any(body.strip() for body in SCRIPT_TAG.findall(raw_html))
async def _load_source(self, request: ConvertRequest) -> tuple:
if request.source.html is not None:
html = request.source.html
limit = self._settings.max_html_bytes
if limit and len(html.encode("utf-8")) > limit:
raise LimitExceededError(
"Zdrojove HTML je vetsi nez nakonfigurovany limit MAX_HTML_BYTES.",
{"limit_bytes": limit},
)
return html, request.source.base_url
fetched = await asyncio.to_thread(
fetch_document, request.source.url, self._guard, self._settings
)
return fetched.html, request.source.base_url or fetched.base_url
async def _execute(
self,
request: ConvertRequest,
raw_html: str,
base_url: str | None,
engine_name: str,
workdir: Path,
progress: ProgressCallback | None,
) -> ConversionResult:
engine = self._engines[engine_name]
report = AssetReport()
gate = AssetGate(
guard=self._guard,
report=report,
allow_remote=request.assets.allow_remote,
timeout_seconds=request.assets.timeout_seconds,
)
warnings: list[str] = []
numbering_mode = self._numbering_mode(request, engine_name)
if request.toc.enabled and not getattr(engine, "supports_anchor_pages", False):
raise UnsupportedCombinationError(
"Generovani obsahu s cisly stranek podporuje pouze engine weasyprint.",
{"engine": engine_name},
)
document = SourceDocument(raw_html)
if engine_name == "weasyprint":
# WeasyPrint drops such images without a word, see the method docstring.
document.relax_percentage_image_widths()
document.set_base_url(base_url)
document.append_stylesheet(
build_page_css(
request.page,
request.page_numbers if numbering_mode == "css" else None,
total_pages=None,
outline=request.outline,
)
)
headings = document.collect_headings(request.toc.depth) if request.toc.enabled else []
if request.toc.enabled:
if not headings:
warnings.append("Dokument neobsahuje zadne nadpisy, obsah nebyl vygenerovan.")
request.toc.enabled = False
else:
document.insert_toc(headings, request.toc.title, pages=None)
chunks = self._split(document, request)
direct_navigation = self._can_navigate_directly(request, engine_name, chunks)
if direct_navigation:
logger.info("Chromium will navigate to the source URL directly so its scripts run in context")
renders = await self._render_pass(
engine, chunks, base_url, request, workdir, gate, progress, 1, direct_navigation
)
if request.toc.enabled:
anchor_pages = self._absolute_anchor_pages(renders)
missing = [item.anchor for item in headings if item.anchor not in anchor_pages]
if missing:
logger.warning(
"Some headings have no anchor position, their page numbers stay empty",
extra={"missing_anchors": len(missing)},
)
document.remove_toc()
document.insert_toc(headings, request.toc.title, pages=anchor_pages)
chunks = self._split(document, request)
renders = await self._render_pass(
engine, chunks, base_url, request, workdir, gate, progress, 2
)
if len(chunks) > 1:
warnings.append(
"Dokument byl rozdelen na casti, odkazy v obsahu proto nejsou klikatelne. "
"Cisla stranek jsou spravna."
)
merged_path = workdir / "merged.pdf"
total_pages = merger.merge([item.path for item in renders], merged_path)
self._check_page_limit(total_pages)
final_path = merged_path
if request.page_numbers.enabled and numbering_mode == "overlay":
final_path = await asyncio.to_thread(
self._stamp_numbers, merged_path, workdir, request, total_pages
)
on_job_finished = getattr(engine, "on_job_finished", None)
if callable(on_job_finished):
on_job_finished()
return ConversionResult(
path=final_path,
page_count=total_pages,
engine_used=engine_name,
missing_assets=report.missing,
warnings=warnings,
)
@staticmethod
def _can_navigate_directly(request: ConvertRequest, engine_name: str, chunks: list) -> bool:
"""Chromium renders a foreign page best when it loads the URL itself.
Only possible when the document is not split and needs no injected
markup, otherwise the modified HTML has to be pushed into the page.
"""
return (
engine_name == "chromium"
and request.source.url is not None
and not request.toc.enabled
and len(chunks) == 1
)
def _numbering_mode(self, request: ConvertRequest, engine_name: str) -> str:
mode = request.page_numbers.mode
if mode == "auto":
if engine_name == "weasyprint" and not request.chunking.enabled:
return "css"
return "overlay"
if mode == "css" and engine_name != "weasyprint":
raise UnsupportedCombinationError(
"Rezim cislovani css funguje pouze s enginem weasyprint. Pouzijte overlay nebo auto.",
{"engine": engine_name},
)
if mode == "css" and request.chunking.enabled:
raise UnsupportedCombinationError(
"Rezim cislovani css nelze kombinovat s chunkovanim, protoze citac stranek se v kazde "
"casti restartuje. Vypnete chunking nebo pouzijte overlay.",
)
return mode
def _split(self, document: SourceDocument, request: ConvertRequest) -> list:
if not request.chunking.enabled:
return [document.to_html()]
return split_document(document, request.chunking.pages_per_chunk)
async def _render_pass(
self,
engine,
chunks: list,
base_url: str | None,
request: ConvertRequest,
workdir: Path,
gate: AssetGate,
progress: ProgressCallback | None,
pass_number: int,
direct_navigation: bool = False,
) -> list:
renders = []
pages_rendered = 0
for index, chunk_html in enumerate(chunks):
output = workdir / f"pass{pass_number}-chunk{index:04d}.pdf"
render = await engine.render_chunk(
None if direct_navigation else chunk_html,
base_url,
request,
output,
total_pages=None,
asset_gate=gate,
)
renders.append(render)
pages_rendered += render.page_count
if progress is not None:
progress(pages_rendered, index + 1, len(chunks), pass_number)
self._check_page_limit(pages_rendered)
logger.info(
"Render pass finished",
extra={"pass_number": pass_number, "chunks": len(chunks), "pages": pages_rendered},
)
return renders
@staticmethod
def _absolute_anchor_pages(renders: list) -> dict:
pages: dict = {}
offset = 0
for render in renders:
for anchor, local_page in render.anchor_pages.items():
pages.setdefault(anchor, offset + local_page + 1)
offset += render.page_count
return pages
def _check_page_limit(self, pages: int) -> None:
if self._settings.max_pages and pages > self._settings.max_pages:
raise LimitExceededError(
"Dokument ma vice stranek nez nakonfigurovany limit MAX_PAGES.",
{"pages": pages, "limit": self._settings.max_pages},
)
@staticmethod
def _stamp_numbers(merged_path: Path, workdir: Path, request: ConvertRequest, total_pages: int) -> Path:
width, height = merger.first_page_size(merged_path)
overlay_path = workdir / "overlay.pdf"
paginator.build_overlay(
total_pages, width, height, request.page, request.page_numbers, overlay_path
)
numbered_path = workdir / "numbered.pdf"
paginator.apply_overlay(merged_path, overlay_path, numbered_path)
return numbered_path