354 lines
13 KiB
Python
354 lines
13 KiB
Python
"""The conversion pipeline.
|
|
|
|
Order of operations:
|
|
|
|
1. load the source HTML, validating the target address
|
|
2. pick the engine
|
|
3. decide how page numbers will be produced
|
|
4. build the document, inject the page stylesheet, optionally insert the table
|
|
of contents placeholder
|
|
5. split into chunks
|
|
6. first render pass, which yields the real page count and anchor positions
|
|
7. second render pass when a table of contents needs real page numbers
|
|
8. merge the chunks
|
|
9. stamp the numbering overlay when CSS counters cannot be used
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import logging
|
|
import re
|
|
from dataclasses import dataclass, field
|
|
from pathlib import Path
|
|
from typing import Callable
|
|
|
|
from ..config import Settings, get_settings
|
|
from ..errors import (
|
|
ConversionError,
|
|
LimitExceededError,
|
|
RenderTimeoutError,
|
|
UnsupportedCombinationError,
|
|
)
|
|
from ..models import ConvertRequest, MissingAsset
|
|
from ..pdf import merger, paginator
|
|
from ..pdf.chunker import split_document
|
|
from ..pdf.document import SourceDocument
|
|
from ..pdf.styles import build_page_css
|
|
from .fetcher import AssetGate, AssetReport, fetch_document
|
|
from .security import UrlGuard
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
SCRIPT_TAG = re.compile(r"<script\b[^>]*>(.*?)</script>", re.IGNORECASE | re.DOTALL)
|
|
SCRIPT_SRC = re.compile(r"<script\b[^>]*\bsrc\s*=", re.IGNORECASE)
|
|
|
|
ProgressCallback = Callable[[int, int, int, int], None]
|
|
|
|
|
|
@dataclass
|
|
class ConversionResult:
|
|
path: Path
|
|
page_count: int
|
|
engine_used: str
|
|
missing_assets: list[MissingAsset] = field(default_factory=list)
|
|
warnings: list[str] = field(default_factory=list)
|
|
|
|
|
|
class ConversionPipeline:
|
|
def __init__(self, engines: dict, settings: Settings | None = None) -> None:
|
|
self._engines = engines
|
|
self._settings = settings or get_settings()
|
|
self._guard = UrlGuard(self._settings)
|
|
|
|
async def run(
|
|
self,
|
|
request: ConvertRequest,
|
|
workdir: Path,
|
|
progress: ProgressCallback | None = None,
|
|
) -> ConversionResult:
|
|
workdir.mkdir(parents=True, exist_ok=True)
|
|
timeout = self._settings.max_render_seconds
|
|
|
|
coroutine = self._run_with_fallback(request, workdir, progress)
|
|
if timeout:
|
|
try:
|
|
return await asyncio.wait_for(coroutine, timeout=timeout)
|
|
except asyncio.TimeoutError as exc:
|
|
raise RenderTimeoutError(
|
|
"Render prekrocil nakonfigurovany limit MAX_RENDER_SECONDS.",
|
|
{"limit_seconds": timeout},
|
|
) from exc
|
|
return await coroutine
|
|
|
|
async def _run_with_fallback(
|
|
self, request: ConvertRequest, workdir: Path, progress: ProgressCallback | None
|
|
) -> ConversionResult:
|
|
raw_html, base_url = await self._load_source(request)
|
|
requested = request.engine
|
|
engine_name = self._select_engine(requested, raw_html)
|
|
|
|
try:
|
|
return await self._execute(request, raw_html, base_url, engine_name, workdir, progress)
|
|
except ConversionError:
|
|
raise
|
|
except Exception as exc:
|
|
if requested != "auto" or engine_name != "weasyprint" or "chromium" not in self._engines:
|
|
raise
|
|
logger.warning(
|
|
"WeasyPrint render failed, falling back to Chromium",
|
|
exc_info=exc,
|
|
extra={"engine": engine_name},
|
|
)
|
|
result = await self._execute(request, raw_html, base_url, "chromium", workdir, progress)
|
|
result.warnings.append(
|
|
"Render pres WeasyPrint selhal, dokument byl vygenerovan pres Chromium."
|
|
)
|
|
return result
|
|
|
|
def _select_engine(self, requested: str, raw_html: str) -> str:
|
|
if requested != "auto":
|
|
if requested not in self._engines:
|
|
raise UnsupportedCombinationError(
|
|
f"Engine {requested} neni v teto instanci k dispozici.",
|
|
{"available": sorted(self._engines)},
|
|
)
|
|
return requested
|
|
|
|
if self._has_active_scripts(raw_html) and "chromium" in self._engines:
|
|
logger.info("Auto engine selected chromium because the document contains scripts")
|
|
return "chromium"
|
|
|
|
if "weasyprint" in self._engines:
|
|
return "weasyprint"
|
|
|
|
return next(iter(self._engines))
|
|
|
|
@staticmethod
|
|
def _has_active_scripts(raw_html: str) -> bool:
|
|
if SCRIPT_SRC.search(raw_html):
|
|
return True
|
|
return any(body.strip() for body in SCRIPT_TAG.findall(raw_html))
|
|
|
|
async def _load_source(self, request: ConvertRequest) -> tuple:
|
|
if request.source.html is not None:
|
|
html = request.source.html
|
|
limit = self._settings.max_html_bytes
|
|
if limit and len(html.encode("utf-8")) > limit:
|
|
raise LimitExceededError(
|
|
"Zdrojove HTML je vetsi nez nakonfigurovany limit MAX_HTML_BYTES.",
|
|
{"limit_bytes": limit},
|
|
)
|
|
return html, request.source.base_url
|
|
|
|
fetched = await asyncio.to_thread(
|
|
fetch_document, request.source.url, self._guard, self._settings
|
|
)
|
|
return fetched.html, request.source.base_url or fetched.base_url
|
|
|
|
async def _execute(
|
|
self,
|
|
request: ConvertRequest,
|
|
raw_html: str,
|
|
base_url: str | None,
|
|
engine_name: str,
|
|
workdir: Path,
|
|
progress: ProgressCallback | None,
|
|
) -> ConversionResult:
|
|
engine = self._engines[engine_name]
|
|
report = AssetReport()
|
|
gate = AssetGate(
|
|
guard=self._guard,
|
|
report=report,
|
|
allow_remote=request.assets.allow_remote,
|
|
timeout_seconds=request.assets.timeout_seconds,
|
|
)
|
|
warnings: list[str] = []
|
|
|
|
numbering_mode = self._numbering_mode(request, engine_name)
|
|
|
|
if request.toc.enabled and not getattr(engine, "supports_anchor_pages", False):
|
|
raise UnsupportedCombinationError(
|
|
"Generovani obsahu s cisly stranek podporuje pouze engine weasyprint.",
|
|
{"engine": engine_name},
|
|
)
|
|
|
|
document = SourceDocument(raw_html)
|
|
if engine_name == "weasyprint":
|
|
# WeasyPrint drops such images without a word, see the method docstring.
|
|
document.relax_percentage_image_widths()
|
|
document.set_base_url(base_url)
|
|
document.append_stylesheet(
|
|
build_page_css(
|
|
request.page,
|
|
request.page_numbers if numbering_mode == "css" else None,
|
|
total_pages=None,
|
|
outline=request.outline,
|
|
)
|
|
)
|
|
|
|
headings = document.collect_headings(request.toc.depth) if request.toc.enabled else []
|
|
if request.toc.enabled:
|
|
if not headings:
|
|
warnings.append("Dokument neobsahuje zadne nadpisy, obsah nebyl vygenerovan.")
|
|
request.toc.enabled = False
|
|
else:
|
|
document.insert_toc(headings, request.toc.title, pages=None)
|
|
|
|
chunks = self._split(document, request)
|
|
direct_navigation = self._can_navigate_directly(request, engine_name, chunks)
|
|
if direct_navigation:
|
|
logger.info("Chromium will navigate to the source URL directly so its scripts run in context")
|
|
|
|
renders = await self._render_pass(
|
|
engine, chunks, base_url, request, workdir, gate, progress, 1, direct_navigation
|
|
)
|
|
|
|
if request.toc.enabled:
|
|
anchor_pages = self._absolute_anchor_pages(renders)
|
|
missing = [item.anchor for item in headings if item.anchor not in anchor_pages]
|
|
if missing:
|
|
logger.warning(
|
|
"Some headings have no anchor position, their page numbers stay empty",
|
|
extra={"missing_anchors": len(missing)},
|
|
)
|
|
document.remove_toc()
|
|
document.insert_toc(headings, request.toc.title, pages=anchor_pages)
|
|
chunks = self._split(document, request)
|
|
renders = await self._render_pass(
|
|
engine, chunks, base_url, request, workdir, gate, progress, 2
|
|
)
|
|
|
|
if len(chunks) > 1:
|
|
warnings.append(
|
|
"Dokument byl rozdelen na casti, odkazy v obsahu proto nejsou klikatelne. "
|
|
"Cisla stranek jsou spravna."
|
|
)
|
|
|
|
merged_path = workdir / "merged.pdf"
|
|
total_pages = merger.merge([item.path for item in renders], merged_path)
|
|
self._check_page_limit(total_pages)
|
|
|
|
final_path = merged_path
|
|
if request.page_numbers.enabled and numbering_mode == "overlay":
|
|
final_path = await asyncio.to_thread(
|
|
self._stamp_numbers, merged_path, workdir, request, total_pages
|
|
)
|
|
|
|
on_job_finished = getattr(engine, "on_job_finished", None)
|
|
if callable(on_job_finished):
|
|
on_job_finished()
|
|
|
|
return ConversionResult(
|
|
path=final_path,
|
|
page_count=total_pages,
|
|
engine_used=engine_name,
|
|
missing_assets=report.missing,
|
|
warnings=warnings,
|
|
)
|
|
|
|
@staticmethod
|
|
def _can_navigate_directly(request: ConvertRequest, engine_name: str, chunks: list) -> bool:
|
|
"""Chromium renders a foreign page best when it loads the URL itself.
|
|
|
|
Only possible when the document is not split and needs no injected
|
|
markup, otherwise the modified HTML has to be pushed into the page.
|
|
"""
|
|
return (
|
|
engine_name == "chromium"
|
|
and request.source.url is not None
|
|
and not request.toc.enabled
|
|
and len(chunks) == 1
|
|
)
|
|
|
|
def _numbering_mode(self, request: ConvertRequest, engine_name: str) -> str:
|
|
mode = request.page_numbers.mode
|
|
if mode == "auto":
|
|
if engine_name == "weasyprint" and not request.chunking.enabled:
|
|
return "css"
|
|
return "overlay"
|
|
|
|
if mode == "css" and engine_name != "weasyprint":
|
|
raise UnsupportedCombinationError(
|
|
"Rezim cislovani css funguje pouze s enginem weasyprint. Pouzijte overlay nebo auto.",
|
|
{"engine": engine_name},
|
|
)
|
|
if mode == "css" and request.chunking.enabled:
|
|
raise UnsupportedCombinationError(
|
|
"Rezim cislovani css nelze kombinovat s chunkovanim, protoze citac stranek se v kazde "
|
|
"casti restartuje. Vypnete chunking nebo pouzijte overlay.",
|
|
)
|
|
return mode
|
|
|
|
def _split(self, document: SourceDocument, request: ConvertRequest) -> list:
|
|
if not request.chunking.enabled:
|
|
return [document.to_html()]
|
|
return split_document(document, request.chunking.pages_per_chunk)
|
|
|
|
async def _render_pass(
|
|
self,
|
|
engine,
|
|
chunks: list,
|
|
base_url: str | None,
|
|
request: ConvertRequest,
|
|
workdir: Path,
|
|
gate: AssetGate,
|
|
progress: ProgressCallback | None,
|
|
pass_number: int,
|
|
direct_navigation: bool = False,
|
|
) -> list:
|
|
renders = []
|
|
pages_rendered = 0
|
|
|
|
for index, chunk_html in enumerate(chunks):
|
|
output = workdir / f"pass{pass_number}-chunk{index:04d}.pdf"
|
|
render = await engine.render_chunk(
|
|
None if direct_navigation else chunk_html,
|
|
base_url,
|
|
request,
|
|
output,
|
|
total_pages=None,
|
|
asset_gate=gate,
|
|
)
|
|
renders.append(render)
|
|
pages_rendered += render.page_count
|
|
|
|
if progress is not None:
|
|
progress(pages_rendered, index + 1, len(chunks), pass_number)
|
|
|
|
self._check_page_limit(pages_rendered)
|
|
|
|
logger.info(
|
|
"Render pass finished",
|
|
extra={"pass_number": pass_number, "chunks": len(chunks), "pages": pages_rendered},
|
|
)
|
|
return renders
|
|
|
|
@staticmethod
|
|
def _absolute_anchor_pages(renders: list) -> dict:
|
|
pages: dict = {}
|
|
offset = 0
|
|
for render in renders:
|
|
for anchor, local_page in render.anchor_pages.items():
|
|
pages.setdefault(anchor, offset + local_page + 1)
|
|
offset += render.page_count
|
|
return pages
|
|
|
|
def _check_page_limit(self, pages: int) -> None:
|
|
if self._settings.max_pages and pages > self._settings.max_pages:
|
|
raise LimitExceededError(
|
|
"Dokument ma vice stranek nez nakonfigurovany limit MAX_PAGES.",
|
|
{"pages": pages, "limit": self._settings.max_pages},
|
|
)
|
|
|
|
@staticmethod
|
|
def _stamp_numbers(merged_path: Path, workdir: Path, request: ConvertRequest, total_pages: int) -> Path:
|
|
width, height = merger.first_page_size(merged_path)
|
|
overlay_path = workdir / "overlay.pdf"
|
|
paginator.build_overlay(
|
|
total_pages, width, height, request.page, request.page_numbers, overlay_path
|
|
)
|
|
numbered_path = workdir / "numbered.pdf"
|
|
paginator.apply_overlay(merged_path, overlay_path, numbered_path)
|
|
return numbered_path
|