"""The conversion pipeline. Order of operations: 1. load the source HTML, validating the target address 2. pick the engine 3. decide how page numbers will be produced 4. build the document, inject the page stylesheet, optionally insert the table of contents placeholder 5. split into chunks 6. first render pass, which yields the real page count and anchor positions 7. second render pass when a table of contents needs real page numbers 8. merge the chunks 9. stamp the numbering overlay when CSS counters cannot be used """ from __future__ import annotations import asyncio import logging import re from dataclasses import dataclass, field from pathlib import Path from typing import Callable from ..config import Settings, get_settings from ..errors import ( ConversionError, LimitExceededError, RenderTimeoutError, UnsupportedCombinationError, ) from ..models import ConvertRequest, MissingAsset from ..pdf import merger, paginator from ..pdf.chunker import split_document from ..pdf.document import SourceDocument from ..pdf.styles import build_page_css from .fetcher import AssetGate, AssetReport, fetch_document from .security import UrlGuard logger = logging.getLogger(__name__) SCRIPT_TAG = re.compile(r"]*>(.*?)", re.IGNORECASE | re.DOTALL) SCRIPT_SRC = re.compile(r"]*\bsrc\s*=", re.IGNORECASE) ProgressCallback = Callable[[int, int, int, int], None] @dataclass class ConversionResult: path: Path page_count: int engine_used: str missing_assets: list[MissingAsset] = field(default_factory=list) warnings: list[str] = field(default_factory=list) class ConversionPipeline: def __init__(self, engines: dict, settings: Settings | None = None) -> None: self._engines = engines self._settings = settings or get_settings() self._guard = UrlGuard(self._settings) async def run( self, request: ConvertRequest, workdir: Path, progress: ProgressCallback | None = None, ) -> ConversionResult: workdir.mkdir(parents=True, exist_ok=True) timeout = self._settings.max_render_seconds coroutine = self._run_with_fallback(request, workdir, progress) if timeout: try: return await asyncio.wait_for(coroutine, timeout=timeout) except asyncio.TimeoutError as exc: raise RenderTimeoutError( "Render prekrocil nakonfigurovany limit MAX_RENDER_SECONDS.", {"limit_seconds": timeout}, ) from exc return await coroutine async def _run_with_fallback( self, request: ConvertRequest, workdir: Path, progress: ProgressCallback | None ) -> ConversionResult: raw_html, base_url = await self._load_source(request) requested = request.engine engine_name = self._select_engine(requested, raw_html) try: return await self._execute(request, raw_html, base_url, engine_name, workdir, progress) except ConversionError: raise except Exception as exc: if requested != "auto" or engine_name != "weasyprint" or "chromium" not in self._engines: raise logger.warning( "WeasyPrint render failed, falling back to Chromium", exc_info=exc, extra={"engine": engine_name}, ) result = await self._execute(request, raw_html, base_url, "chromium", workdir, progress) result.warnings.append( "Render pres WeasyPrint selhal, dokument byl vygenerovan pres Chromium." ) return result def _select_engine(self, requested: str, raw_html: str) -> str: if requested != "auto": if requested not in self._engines: raise UnsupportedCombinationError( f"Engine {requested} neni v teto instanci k dispozici.", {"available": sorted(self._engines)}, ) return requested if self._has_active_scripts(raw_html) and "chromium" in self._engines: logger.info("Auto engine selected chromium because the document contains scripts") return "chromium" if "weasyprint" in self._engines: return "weasyprint" return next(iter(self._engines)) @staticmethod def _has_active_scripts(raw_html: str) -> bool: if SCRIPT_SRC.search(raw_html): return True return any(body.strip() for body in SCRIPT_TAG.findall(raw_html)) async def _load_source(self, request: ConvertRequest) -> tuple: if request.source.html is not None: html = request.source.html limit = self._settings.max_html_bytes if limit and len(html.encode("utf-8")) > limit: raise LimitExceededError( "Zdrojove HTML je vetsi nez nakonfigurovany limit MAX_HTML_BYTES.", {"limit_bytes": limit}, ) return html, request.source.base_url fetched = await asyncio.to_thread( fetch_document, request.source.url, self._guard, self._settings ) return fetched.html, request.source.base_url or fetched.base_url async def _execute( self, request: ConvertRequest, raw_html: str, base_url: str | None, engine_name: str, workdir: Path, progress: ProgressCallback | None, ) -> ConversionResult: engine = self._engines[engine_name] report = AssetReport() gate = AssetGate( guard=self._guard, report=report, allow_remote=request.assets.allow_remote, timeout_seconds=request.assets.timeout_seconds, ) warnings: list[str] = [] numbering_mode = self._numbering_mode(request, engine_name) if request.toc.enabled and not getattr(engine, "supports_anchor_pages", False): raise UnsupportedCombinationError( "Generovani obsahu s cisly stranek podporuje pouze engine weasyprint.", {"engine": engine_name}, ) document = SourceDocument(raw_html) if engine_name == "weasyprint": # WeasyPrint drops such images without a word, see the method docstring. document.relax_percentage_image_widths() document.set_base_url(base_url) document.append_stylesheet( build_page_css( request.page, request.page_numbers if numbering_mode == "css" else None, total_pages=None, outline=request.outline, ) ) headings = document.collect_headings(request.toc.depth) if request.toc.enabled else [] if request.toc.enabled: if not headings: warnings.append("Dokument neobsahuje zadne nadpisy, obsah nebyl vygenerovan.") request.toc.enabled = False else: document.insert_toc(headings, request.toc.title, pages=None) chunks = self._split(document, request) direct_navigation = self._can_navigate_directly(request, engine_name, chunks) if direct_navigation: logger.info("Chromium will navigate to the source URL directly so its scripts run in context") renders = await self._render_pass( engine, chunks, base_url, request, workdir, gate, progress, 1, direct_navigation ) if request.toc.enabled: anchor_pages = self._absolute_anchor_pages(renders) missing = [item.anchor for item in headings if item.anchor not in anchor_pages] if missing: logger.warning( "Some headings have no anchor position, their page numbers stay empty", extra={"missing_anchors": len(missing)}, ) document.remove_toc() document.insert_toc(headings, request.toc.title, pages=anchor_pages) chunks = self._split(document, request) renders = await self._render_pass( engine, chunks, base_url, request, workdir, gate, progress, 2 ) if len(chunks) > 1: warnings.append( "Dokument byl rozdelen na casti, odkazy v obsahu proto nejsou klikatelne. " "Cisla stranek jsou spravna." ) merged_path = workdir / "merged.pdf" total_pages = merger.merge([item.path for item in renders], merged_path) self._check_page_limit(total_pages) final_path = merged_path if request.page_numbers.enabled and numbering_mode == "overlay": final_path = await asyncio.to_thread( self._stamp_numbers, merged_path, workdir, request, total_pages ) on_job_finished = getattr(engine, "on_job_finished", None) if callable(on_job_finished): on_job_finished() return ConversionResult( path=final_path, page_count=total_pages, engine_used=engine_name, missing_assets=report.missing, warnings=warnings, ) @staticmethod def _can_navigate_directly(request: ConvertRequest, engine_name: str, chunks: list) -> bool: """Chromium renders a foreign page best when it loads the URL itself. Only possible when the document is not split and needs no injected markup, otherwise the modified HTML has to be pushed into the page. """ return ( engine_name == "chromium" and request.source.url is not None and not request.toc.enabled and len(chunks) == 1 ) def _numbering_mode(self, request: ConvertRequest, engine_name: str) -> str: mode = request.page_numbers.mode if mode == "auto": if engine_name == "weasyprint" and not request.chunking.enabled: return "css" return "overlay" if mode == "css" and engine_name != "weasyprint": raise UnsupportedCombinationError( "Rezim cislovani css funguje pouze s enginem weasyprint. Pouzijte overlay nebo auto.", {"engine": engine_name}, ) if mode == "css" and request.chunking.enabled: raise UnsupportedCombinationError( "Rezim cislovani css nelze kombinovat s chunkovanim, protoze citac stranek se v kazde " "casti restartuje. Vypnete chunking nebo pouzijte overlay.", ) return mode def _split(self, document: SourceDocument, request: ConvertRequest) -> list: if not request.chunking.enabled: return [document.to_html()] return split_document(document, request.chunking.pages_per_chunk) async def _render_pass( self, engine, chunks: list, base_url: str | None, request: ConvertRequest, workdir: Path, gate: AssetGate, progress: ProgressCallback | None, pass_number: int, direct_navigation: bool = False, ) -> list: renders = [] pages_rendered = 0 for index, chunk_html in enumerate(chunks): output = workdir / f"pass{pass_number}-chunk{index:04d}.pdf" render = await engine.render_chunk( None if direct_navigation else chunk_html, base_url, request, output, total_pages=None, asset_gate=gate, ) renders.append(render) pages_rendered += render.page_count if progress is not None: progress(pages_rendered, index + 1, len(chunks), pass_number) self._check_page_limit(pages_rendered) logger.info( "Render pass finished", extra={"pass_number": pass_number, "chunks": len(chunks), "pages": pages_rendered}, ) return renders @staticmethod def _absolute_anchor_pages(renders: list) -> dict: pages: dict = {} offset = 0 for render in renders: for anchor, local_page in render.anchor_pages.items(): pages.setdefault(anchor, offset + local_page + 1) offset += render.page_count return pages def _check_page_limit(self, pages: int) -> None: if self._settings.max_pages and pages > self._settings.max_pages: raise LimitExceededError( "Dokument ma vice stranek nez nakonfigurovany limit MAX_PAGES.", {"pages": pages, "limit": self._settings.max_pages}, ) @staticmethod def _stamp_numbers(merged_path: Path, workdir: Path, request: ConvertRequest, total_pages: int) -> Path: width, height = merger.first_page_size(merged_path) overlay_path = workdir / "overlay.pdf" paginator.build_overlay( total_pages, width, height, request.page, request.page_numbers, overlay_path ) numbered_path = workdir / "numbered.pdf" paginator.apply_overlay(merged_path, overlay_path, numbered_path) return numbered_path