Implementace prevodu HTML na PDF
Sluzba prijme adresu HTML dokumentu nebo HTML v tele requestu a vrati PDF. Navrzena pro dokumenty o stovkach az tisicich stranek. Rendering: - WeasyPrint jako vychozi engine, spravne CSS Paged Media, nizka pametova narocnost, bez JavaScriptu - Chromium pres Playwright pro dokumenty dokreslovane skripty - rezim auto s detekci skriptu a fallbackem pri selhani WeasyPrintu Velke dokumenty: - deleni na casti na strukturalnich hranicich, rez nikdy uvnitr tabulky nebo odstavce - dvoupruchodovy render obsahu se skutecnymi cisly stranek, pozice nadpisu se ctou z kotev hlasenych u kazde stranky - cislovani stranek bud pres CSS countery, nebo pres cislovaci vrstvu nastampovanou na hotove PDF, rozmer stranky se cte z vysledneho souboru - Chromium se restartuje po N jobech, nikdy vsak behem beziciho renderu API: - POST /convert synchronne, POST /jobs asynchronne se sledovanim stavu, stahovanim vysledku, rusenim a volitelnym callbackem - GET /health s overenim dostupnosti obou enginu a stavem fronty - OpenAPI respektuje prefix reverse proxy pres root_path Bezpecnost a provoz: - SSRF kontrola po DNS resolvu, na kazdem presmerovani a u vsech pozadavku prohlizece - nedostupne assety render nezastavi, ale hlasi se v odpovedi i v logu - fronta s omezenym poctem workeru, rozpracovane joby se pri ukonceni oznaci jako failed, nezmizi potichu - strukturovane JSON logovani s job_id - vsechny limity vypnute ve vychozim stavu Dockerfile je dvoufazovy, obsahuje zavislosti WeasyPrintu, Chromium a fonty s ceskou diakritikou. Autentizace zamerne neni implementovana, zpusob predavani neni domluveny. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
e3cc8f418b
commit
156289fe2d
@@ -0,0 +1 @@
|
||||
"""html-to-pdf service package."""
|
||||
@@ -0,0 +1,79 @@
|
||||
"""Runtime configuration read from environment variables.
|
||||
|
||||
AppFactory injects variables through the generated runtime .env file, so every
|
||||
option below has a safe default and the service starts without any variable set.
|
||||
"""
|
||||
|
||||
import os
|
||||
from dataclasses import dataclass, field
|
||||
from functools import lru_cache
|
||||
|
||||
|
||||
def _bool(name: str, default: bool) -> bool:
|
||||
raw = os.getenv(name)
|
||||
if raw is None or raw.strip() == "":
|
||||
return default
|
||||
return raw.strip().lower() in ("1", "true", "yes", "on")
|
||||
|
||||
|
||||
def _int(name: str, default: int) -> int:
|
||||
raw = os.getenv(name)
|
||||
if raw is None or raw.strip() == "":
|
||||
return default
|
||||
try:
|
||||
return int(raw)
|
||||
except ValueError:
|
||||
return default
|
||||
|
||||
|
||||
def _list(name: str) -> list[str]:
|
||||
raw = os.getenv(name, "")
|
||||
return [item.strip() for item in raw.split(",") if item.strip()]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Settings:
|
||||
app_name: str = field(default_factory=lambda: os.getenv("APP_NAME", "html-to-pdf"))
|
||||
app_version: str = field(default_factory=lambda: os.getenv("APP_VERSION", "1.0.0"))
|
||||
root_path: str = field(
|
||||
default_factory=lambda: os.getenv("ROOT_PATH") or os.getenv("BASE_PATH") or ""
|
||||
)
|
||||
log_level: str = field(default_factory=lambda: os.getenv("LOG_LEVEL", "INFO").upper())
|
||||
|
||||
# Job queue
|
||||
workers: int = field(default_factory=lambda: _int("WORKERS", 2))
|
||||
queue_max_size: int = field(default_factory=lambda: _int("QUEUE_MAX_SIZE", 100))
|
||||
sync_timeout_seconds: int = field(default_factory=lambda: _int("SYNC_TIMEOUT_SECONDS", 60))
|
||||
job_result_ttl_seconds: int = field(default_factory=lambda: _int("JOB_RESULT_TTL_SECONDS", 3600))
|
||||
storage_dir: str = field(default_factory=lambda: os.getenv("STORAGE_DIR", "/tmp/html-to-pdf"))
|
||||
|
||||
# Engines
|
||||
default_engine: str = field(default_factory=lambda: os.getenv("DEFAULT_ENGINE", "auto"))
|
||||
chromium_enabled: bool = field(default_factory=lambda: _bool("CHROMIUM_ENABLED", True))
|
||||
chromium_restart_after_jobs: int = field(
|
||||
default_factory=lambda: _int("CHROMIUM_RESTART_AFTER_JOBS", 50)
|
||||
)
|
||||
|
||||
# Network
|
||||
fetch_timeout_seconds: int = field(default_factory=lambda: _int("FETCH_TIMEOUT_SECONDS", 30))
|
||||
asset_timeout_seconds: int = field(default_factory=lambda: _int("ASSET_TIMEOUT_SECONDS", 10))
|
||||
max_redirects: int = field(default_factory=lambda: _int("MAX_REDIRECTS", 5))
|
||||
|
||||
# SSRF protection. Blocking is on by default and can be widened or narrowed.
|
||||
ssrf_block_private: bool = field(default_factory=lambda: _bool("SSRF_BLOCK_PRIVATE", True))
|
||||
ssrf_extra_blocked_cidrs: list[str] = field(default_factory=lambda: _list("SSRF_EXTRA_BLOCKED_CIDRS"))
|
||||
ssrf_allowed_hosts: list[str] = field(default_factory=lambda: _list("SSRF_ALLOWED_HOSTS"))
|
||||
|
||||
# Optional limits. Zero means no limit, which is the default on purpose.
|
||||
max_pages: int = field(default_factory=lambda: _int("MAX_PAGES", 0))
|
||||
max_html_bytes: int = field(default_factory=lambda: _int("MAX_HTML_BYTES", 0))
|
||||
max_render_seconds: int = field(default_factory=lambda: _int("MAX_RENDER_SECONDS", 0))
|
||||
|
||||
# Callback delivery
|
||||
callback_timeout_seconds: int = field(default_factory=lambda: _int("CALLBACK_TIMEOUT_SECONDS", 15))
|
||||
callback_retries: int = field(default_factory=lambda: _int("CALLBACK_RETRIES", 3))
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def get_settings() -> Settings:
|
||||
return Settings()
|
||||
+45
@@ -0,0 +1,45 @@
|
||||
"""Application container.
|
||||
|
||||
Built once during startup and read by the routers. Keeping it here avoids
|
||||
importing the FastAPI app from the routers.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
from .config import Settings
|
||||
from .services.jobs import JobManager
|
||||
from .services.pipeline import ConversionPipeline
|
||||
from .services.storage import Storage
|
||||
|
||||
|
||||
@dataclass
|
||||
class Container:
|
||||
settings: Settings
|
||||
storage: Storage
|
||||
pipeline: ConversionPipeline
|
||||
jobs: JobManager
|
||||
engines: dict
|
||||
|
||||
|
||||
_container: Container | None = None
|
||||
|
||||
|
||||
def set_container(container: Container) -> None:
|
||||
global _container
|
||||
_container = container
|
||||
|
||||
|
||||
def get_container() -> Container:
|
||||
if _container is None:
|
||||
raise RuntimeError("Aplikace jeste nebyla inicializovana.")
|
||||
return _container
|
||||
|
||||
|
||||
def get_jobs() -> JobManager:
|
||||
return get_container().jobs
|
||||
|
||||
|
||||
def get_settings_dep() -> Settings:
|
||||
return get_container().settings
|
||||
@@ -0,0 +1,48 @@
|
||||
"""Common interface of the render engines."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from abc import ABC, abstractmethod
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
from ..models import ConvertRequest
|
||||
|
||||
|
||||
@dataclass
|
||||
class ChunkRender:
|
||||
"""Result of rendering a single chunk."""
|
||||
|
||||
path: Path
|
||||
page_count: int
|
||||
# Anchor name to zero based page index inside this chunk.
|
||||
anchor_pages: dict[str, int] = field(default_factory=dict)
|
||||
|
||||
|
||||
class RenderEngine(ABC):
|
||||
name: str
|
||||
|
||||
@abstractmethod
|
||||
async def available(self) -> bool:
|
||||
"""True when the engine can render right now."""
|
||||
|
||||
@abstractmethod
|
||||
async def render_chunk(
|
||||
self,
|
||||
html: str,
|
||||
base_url: str | None,
|
||||
request: ConvertRequest,
|
||||
output_path: Path,
|
||||
total_pages: int | None = None,
|
||||
asset_gate=None,
|
||||
) -> ChunkRender:
|
||||
"""Render one chunk into output_path."""
|
||||
|
||||
async def shutdown(self) -> None:
|
||||
"""Release engine resources. Default is a no-op."""
|
||||
return None
|
||||
|
||||
@property
|
||||
def supports_anchor_pages(self) -> bool:
|
||||
"""True when the engine can report on which page an anchor landed."""
|
||||
return False
|
||||
@@ -0,0 +1,211 @@
|
||||
"""Headless Chromium engine driven by Playwright.
|
||||
|
||||
Used for foreign URLs and for documents that are completed by JavaScript.
|
||||
Chromium has no usable CSS page counters, so page numbering for this engine is
|
||||
always done by the overlay in the post processing step.
|
||||
|
||||
The browser leaks memory over time, therefore the whole instance is restarted
|
||||
after a configurable number of jobs.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
from ..config import get_settings
|
||||
from ..errors import EngineUnavailableError, RenderTimeoutError
|
||||
from ..models import ConvertRequest
|
||||
from ..pdf.styles import NAMED_SIZE
|
||||
from .base import ChunkRender, RenderEngine
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
LAUNCH_ARGS = [
|
||||
"--no-sandbox",
|
||||
"--disable-dev-shm-usage",
|
||||
"--disable-gpu",
|
||||
"--hide-scrollbars",
|
||||
]
|
||||
|
||||
|
||||
class ChromiumEngine(RenderEngine):
|
||||
name = "chromium"
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._settings = get_settings()
|
||||
self._lock = asyncio.Lock()
|
||||
self._playwright = None
|
||||
self._browser = None
|
||||
self._jobs_since_launch = 0
|
||||
self._active_renders = 0
|
||||
|
||||
async def available(self) -> bool:
|
||||
if not self._settings.chromium_enabled:
|
||||
return False
|
||||
try:
|
||||
await self._ensure_browser(reserve=False)
|
||||
except Exception as exc: # noqa: BLE001 - reported, never silent
|
||||
logger.error("Chromium is not available", exc_info=exc)
|
||||
return False
|
||||
return True
|
||||
|
||||
async def _ensure_browser(self, reserve: bool = True):
|
||||
"""Return a running browser, restarting it when that is safe.
|
||||
|
||||
The restart must never happen while a render is in flight, otherwise the
|
||||
health check could close the browser under a running job.
|
||||
"""
|
||||
async with self._lock:
|
||||
due_for_restart = self._jobs_since_launch >= self._settings.chromium_restart_after_jobs
|
||||
if self._browser is not None and self._active_renders == 0 and due_for_restart:
|
||||
logger.info(
|
||||
"Restarting Chromium to release memory",
|
||||
extra={"jobs_since_launch": self._jobs_since_launch},
|
||||
)
|
||||
await self._close_browser()
|
||||
|
||||
if self._browser is None:
|
||||
try:
|
||||
from playwright.async_api import async_playwright
|
||||
except ImportError as exc:
|
||||
raise EngineUnavailableError(
|
||||
"Engine chromium neni k dispozici, chybi balicek playwright.",
|
||||
) from exc
|
||||
|
||||
self._playwright = await async_playwright().start()
|
||||
self._browser = await self._playwright.chromium.launch(args=LAUNCH_ARGS)
|
||||
self._jobs_since_launch = 0
|
||||
logger.info("Chromium launched")
|
||||
|
||||
if reserve:
|
||||
self._active_renders += 1
|
||||
|
||||
return self._browser
|
||||
|
||||
async def _close_browser(self) -> None:
|
||||
if self._browser is not None:
|
||||
try:
|
||||
await self._browser.close()
|
||||
except Exception as exc: # noqa: BLE001 - reported, never silent
|
||||
logger.warning("Closing Chromium failed", exc_info=exc)
|
||||
self._browser = None
|
||||
if self._playwright is not None:
|
||||
try:
|
||||
await self._playwright.stop()
|
||||
except Exception as exc: # noqa: BLE001 - reported, never silent
|
||||
logger.warning("Stopping Playwright failed", exc_info=exc)
|
||||
self._playwright = None
|
||||
|
||||
async def shutdown(self) -> None:
|
||||
async with self._lock:
|
||||
await self._close_browser()
|
||||
|
||||
def on_job_finished(self) -> None:
|
||||
self._jobs_since_launch += 1
|
||||
|
||||
async def render_chunk(
|
||||
self,
|
||||
html: str | None,
|
||||
base_url: str | None,
|
||||
request: ConvertRequest,
|
||||
output_path: Path,
|
||||
total_pages: int | None = None,
|
||||
asset_gate=None,
|
||||
) -> ChunkRender:
|
||||
browser = await self._ensure_browser()
|
||||
try:
|
||||
context = await browser.new_context()
|
||||
except Exception:
|
||||
self._active_renders -= 1
|
||||
raise
|
||||
|
||||
try:
|
||||
if asset_gate is not None:
|
||||
await context.route("**/*", _make_route_handler(asset_gate))
|
||||
|
||||
page = await context.new_page()
|
||||
timeout_ms = request.wait_for.timeout_seconds * 1000
|
||||
|
||||
try:
|
||||
if html is None:
|
||||
if not base_url:
|
||||
raise EngineUnavailableError("Chybi adresa dokumentu pro engine chromium.")
|
||||
await page.goto(base_url, wait_until=request.wait_for.state, timeout=timeout_ms)
|
||||
else:
|
||||
await page.set_content(html, wait_until=request.wait_for.state, timeout=timeout_ms)
|
||||
|
||||
if request.wait_for.selector:
|
||||
await page.wait_for_selector(request.wait_for.selector, timeout=timeout_ms)
|
||||
except Exception as exc: # noqa: BLE001 - converted to a typed error
|
||||
if "Timeout" in type(exc).__name__ or "timeout" in str(exc).lower():
|
||||
raise RenderTimeoutError(
|
||||
"Chromium nestihl nacist dokument v zadanem casovem limitu.",
|
||||
{"timeout_seconds": request.wait_for.timeout_seconds},
|
||||
) from exc
|
||||
raise
|
||||
|
||||
await page.emulate_media(media="print")
|
||||
pdf_kwargs = _pdf_kwargs(request)
|
||||
|
||||
try:
|
||||
pdf_bytes = await page.pdf(outline=request.outline, **pdf_kwargs)
|
||||
except TypeError:
|
||||
logger.warning("Installed Playwright does not support the outline option, continuing without bookmarks")
|
||||
pdf_bytes = await page.pdf(**pdf_kwargs)
|
||||
|
||||
output_path.write_bytes(pdf_bytes)
|
||||
finally:
|
||||
self._active_renders -= 1
|
||||
try:
|
||||
await context.close()
|
||||
except Exception as exc:
|
||||
logger.warning("Closing the browser context failed", exc_info=exc)
|
||||
|
||||
return ChunkRender(path=output_path, page_count=_count_pages(output_path))
|
||||
|
||||
|
||||
def _pdf_kwargs(request: ConvertRequest) -> dict:
|
||||
margin = request.page.margin
|
||||
kwargs: dict = {
|
||||
"print_background": True,
|
||||
"prefer_css_page_size": True,
|
||||
"margin": {
|
||||
"top": margin.top,
|
||||
"right": margin.right,
|
||||
"bottom": margin.bottom,
|
||||
"left": margin.left,
|
||||
},
|
||||
}
|
||||
fmt = request.page.format.strip()
|
||||
if NAMED_SIZE.match(fmt):
|
||||
kwargs["format"] = fmt
|
||||
else:
|
||||
width, _, height = fmt.partition(" ")
|
||||
if width and height:
|
||||
kwargs["width"] = width.strip()
|
||||
kwargs["height"] = height.strip()
|
||||
else:
|
||||
kwargs["format"] = fmt
|
||||
kwargs["landscape"] = request.page.orientation == "landscape"
|
||||
return kwargs
|
||||
|
||||
|
||||
def _make_route_handler(asset_gate):
|
||||
async def handler(route, playwright_request):
|
||||
allowed, reason = asset_gate.allowed(playwright_request.url)
|
||||
if allowed:
|
||||
await route.continue_()
|
||||
return
|
||||
asset_gate.report.add(playwright_request.url, reason)
|
||||
await route.abort()
|
||||
|
||||
return handler
|
||||
|
||||
|
||||
def _count_pages(path: Path) -> int:
|
||||
from pypdf import PdfReader
|
||||
|
||||
with path.open("rb") as handle:
|
||||
return len(PdfReader(handle).pages)
|
||||
@@ -0,0 +1,107 @@
|
||||
"""WeasyPrint engine.
|
||||
|
||||
Default engine for documents we generate ourselves. It implements CSS Paged
|
||||
Media properly, which is what makes counters, running headers and repeated table
|
||||
headers work, and it uses far less memory than a headless browser.
|
||||
|
||||
It does not execute JavaScript. That is intentional.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
from ..models import ConvertRequest
|
||||
from ..pdf.styles import build_page_css
|
||||
from .base import ChunkRender, RenderEngine
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class WeasyPrintEngine(RenderEngine):
|
||||
name = "weasyprint"
|
||||
|
||||
@property
|
||||
def supports_anchor_pages(self) -> bool:
|
||||
return True
|
||||
|
||||
async def available(self) -> bool:
|
||||
try:
|
||||
import weasyprint # noqa: F401
|
||||
except Exception as exc: # noqa: BLE001 - reported, never silent
|
||||
logger.error("WeasyPrint is not importable", exc_info=exc)
|
||||
return False
|
||||
return True
|
||||
|
||||
async def render_chunk(
|
||||
self,
|
||||
html: str,
|
||||
base_url: str | None,
|
||||
request: ConvertRequest,
|
||||
output_path: Path,
|
||||
total_pages: int | None = None,
|
||||
asset_gate=None,
|
||||
) -> ChunkRender:
|
||||
return await asyncio.to_thread(
|
||||
self._render_blocking, html, base_url, request, output_path, total_pages, asset_gate
|
||||
)
|
||||
|
||||
def _render_blocking(
|
||||
self,
|
||||
html: str,
|
||||
base_url: str | None,
|
||||
request: ConvertRequest,
|
||||
output_path: Path,
|
||||
total_pages: int | None,
|
||||
asset_gate,
|
||||
) -> ChunkRender:
|
||||
from weasyprint import CSS, HTML
|
||||
|
||||
kwargs = {"string": html, "base_url": base_url}
|
||||
if asset_gate is not None:
|
||||
kwargs["url_fetcher"] = asset_gate.weasy_fetcher()
|
||||
|
||||
stylesheets = []
|
||||
if total_pages is not None:
|
||||
# Only used when the page total is known up front, which happens on
|
||||
# the second pass of an unchunked document.
|
||||
stylesheets.append(
|
||||
CSS(
|
||||
string=build_page_css(
|
||||
request.page,
|
||||
request.page_numbers,
|
||||
total_pages=total_pages,
|
||||
outline=request.outline,
|
||||
)
|
||||
)
|
||||
)
|
||||
|
||||
document = HTML(**kwargs).render(stylesheets=stylesheets or None)
|
||||
|
||||
write_kwargs = {}
|
||||
if request.pdf_profile:
|
||||
write_kwargs["pdf_variant"] = request.pdf_profile
|
||||
|
||||
document.write_pdf(target=str(output_path), **write_kwargs)
|
||||
|
||||
return ChunkRender(
|
||||
path=output_path,
|
||||
page_count=len(document.pages),
|
||||
anchor_pages=self._anchor_pages(document),
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _anchor_pages(document) -> dict[str, int]:
|
||||
"""Map anchor name to the zero based page index it landed on."""
|
||||
anchors: dict[str, int] = {}
|
||||
for index, page in enumerate(document.pages):
|
||||
page_anchors = getattr(page, "anchors", None)
|
||||
if not page_anchors:
|
||||
continue
|
||||
for name in page_anchors:
|
||||
anchors.setdefault(name, index)
|
||||
if not anchors:
|
||||
logger.warning("WeasyPrint returned no anchors, table of contents page numbers may be missing")
|
||||
return anchors
|
||||
+103
@@ -0,0 +1,103 @@
|
||||
"""Application errors.
|
||||
|
||||
Every failure surfaces a machine readable error_code and a message in Czech,
|
||||
because the message is shown to the caller.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
class ConversionError(Exception):
|
||||
"""Base class for every error the conversion pipeline can raise."""
|
||||
|
||||
error_code = "internal_error"
|
||||
status_code = 500
|
||||
|
||||
def __init__(self, message: str, detail: dict | None = None) -> None:
|
||||
super().__init__(message)
|
||||
self.message = message
|
||||
self.detail = detail or {}
|
||||
|
||||
def to_dict(self) -> dict:
|
||||
payload = {"error_code": self.error_code, "message": self.message}
|
||||
if self.detail:
|
||||
payload["detail"] = self.detail
|
||||
return payload
|
||||
|
||||
|
||||
class InvalidRequestError(ConversionError):
|
||||
error_code = "invalid_request"
|
||||
status_code = 400
|
||||
|
||||
|
||||
class BlockedTargetError(ConversionError):
|
||||
"""The requested URL points somewhere the service refuses to reach."""
|
||||
|
||||
error_code = "blocked_target"
|
||||
status_code = 400
|
||||
|
||||
|
||||
class SourceUnavailableError(ConversionError):
|
||||
error_code = "source_unavailable"
|
||||
status_code = 502
|
||||
|
||||
|
||||
class RenderTimeoutError(ConversionError):
|
||||
error_code = "render_timeout"
|
||||
status_code = 504
|
||||
|
||||
|
||||
class SyncTooLongError(ConversionError):
|
||||
"""Synchronous conversion exceeded its budget, caller should use /jobs."""
|
||||
|
||||
error_code = "sync_too_long"
|
||||
status_code = 413
|
||||
|
||||
|
||||
class EngineUnavailableError(ConversionError):
|
||||
error_code = "engine_unavailable"
|
||||
status_code = 503
|
||||
|
||||
|
||||
class UnsupportedCombinationError(ConversionError):
|
||||
error_code = "unsupported_combination"
|
||||
status_code = 400
|
||||
|
||||
|
||||
class LimitExceededError(ConversionError):
|
||||
error_code = "limit_exceeded"
|
||||
status_code = 400
|
||||
|
||||
|
||||
class JobNotFoundError(ConversionError):
|
||||
error_code = "job_not_found"
|
||||
status_code = 404
|
||||
|
||||
|
||||
class QueueFullError(ConversionError):
|
||||
error_code = "queue_full"
|
||||
status_code = 503
|
||||
|
||||
|
||||
ERROR_CLASSES = (
|
||||
InvalidRequestError,
|
||||
BlockedTargetError,
|
||||
SourceUnavailableError,
|
||||
RenderTimeoutError,
|
||||
SyncTooLongError,
|
||||
EngineUnavailableError,
|
||||
UnsupportedCombinationError,
|
||||
LimitExceededError,
|
||||
JobNotFoundError,
|
||||
QueueFullError,
|
||||
)
|
||||
|
||||
STATUS_BY_CODE = {cls.error_code: cls.status_code for cls in ERROR_CLASSES}
|
||||
|
||||
|
||||
def error_from_code(error_code: str, message: str, detail: dict | None = None) -> ConversionError:
|
||||
"""Rebuild a typed error from a stored job error."""
|
||||
error = ConversionError(message, detail)
|
||||
error.error_code = error_code
|
||||
error.status_code = STATUS_BY_CODE.get(error_code, 500)
|
||||
return error
|
||||
@@ -0,0 +1,58 @@
|
||||
"""Structured JSON logging.
|
||||
|
||||
Every log record is a single JSON line. Anything related to a job carries the
|
||||
job_id so the whole conversion can be reconstructed from the log.
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
import sys
|
||||
from contextvars import ContextVar
|
||||
|
||||
current_job_id: ContextVar[str | None] = ContextVar("current_job_id", default=None)
|
||||
|
||||
_RESERVED = {
|
||||
"args", "asctime", "created", "exc_info", "exc_text", "filename", "funcName",
|
||||
"levelname", "levelno", "lineno", "module", "msecs", "message", "msg", "name",
|
||||
"pathname", "process", "processName", "relativeCreated", "stack_info",
|
||||
"thread", "threadName", "taskName",
|
||||
}
|
||||
|
||||
|
||||
class JsonFormatter(logging.Formatter):
|
||||
def format(self, record: logging.LogRecord) -> str:
|
||||
payload: dict[str, object] = {
|
||||
"ts": self.formatTime(record, "%Y-%m-%dT%H:%M:%S%z"),
|
||||
"level": record.levelname,
|
||||
"logger": record.name,
|
||||
"message": record.getMessage(),
|
||||
}
|
||||
|
||||
job_id = getattr(record, "job_id", None) or current_job_id.get()
|
||||
if job_id:
|
||||
payload["job_id"] = job_id
|
||||
|
||||
for key, value in record.__dict__.items():
|
||||
if key not in _RESERVED and not key.startswith("_") and key != "job_id":
|
||||
payload[key] = value
|
||||
|
||||
if record.exc_info:
|
||||
payload["exception"] = self.formatException(record.exc_info)
|
||||
|
||||
return json.dumps(payload, ensure_ascii=False, default=str)
|
||||
|
||||
|
||||
def setup_logging(level: str) -> None:
|
||||
handler = logging.StreamHandler(sys.stdout)
|
||||
handler.setFormatter(JsonFormatter())
|
||||
|
||||
root = logging.getLogger()
|
||||
root.handlers.clear()
|
||||
root.addHandler(handler)
|
||||
root.setLevel(getattr(logging, level, logging.INFO))
|
||||
|
||||
# uvicorn keeps its own handlers, route them through ours as well
|
||||
for name in ("uvicorn", "uvicorn.access", "uvicorn.error"):
|
||||
logger = logging.getLogger(name)
|
||||
logger.handlers.clear()
|
||||
logger.propagate = True
|
||||
+128
-19
@@ -1,25 +1,134 @@
|
||||
import os
|
||||
from fastapi import FastAPI
|
||||
"""Service entry point.
|
||||
|
||||
APP_NAME = os.getenv("APP_NAME", "html-to-pdf")
|
||||
APP_VERSION = os.getenv("APP_VERSION", "1.0.0")
|
||||
ROOT_PATH = os.getenv("ROOT_PATH", "")
|
||||
The application runs behind the AppFactory reverse proxy under
|
||||
/apps/<app-id>. The prefix is stripped before the request reaches the
|
||||
container, so only the OpenAPI document has to know about it. That is what
|
||||
root_path does.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from contextlib import asynccontextmanager
|
||||
|
||||
from fastapi import FastAPI, Request
|
||||
from fastapi.responses import JSONResponse
|
||||
|
||||
from .config import get_settings
|
||||
from .deps import Container, set_container
|
||||
from .engines.chromium import ChromiumEngine
|
||||
from .engines.weasy import WeasyPrintEngine
|
||||
from .errors import ConversionError
|
||||
from .logging_setup import setup_logging
|
||||
from .routers import convert as convert_router
|
||||
from .routers import health as health_router
|
||||
from .routers import jobs as jobs_router
|
||||
from .services.jobs import JobManager
|
||||
from .services.pipeline import ConversionPipeline
|
||||
from .services.storage import Storage
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
DESCRIPTION = """
|
||||
Sluzba prevadi HTML dokument na PDF. Prijme adresu dokumentu nebo HTML primo
|
||||
v tele requestu a vrati soubor PDF.
|
||||
|
||||
Pro dokumenty o stovkach az tisicich stranek pouzijte asynchronni endpoint
|
||||
POST /jobs. Synchronni POST /convert je urceny pro mensi dokumenty.
|
||||
|
||||
Dva render enginy:
|
||||
|
||||
- weasyprint je vychozi, ma spravne strankovani a nizkou pametovou narocnost,
|
||||
nespousti JavaScript
|
||||
- chromium zvladne i dokumenty dokreslovane JavaScriptem, ale nema pouzitelne
|
||||
CSS countery, takze cisla stranek se dopisuji do hotoveho PDF
|
||||
"""
|
||||
|
||||
|
||||
@asynccontextmanager
|
||||
async def lifespan(app: FastAPI):
|
||||
settings = get_settings()
|
||||
setup_logging(settings.log_level)
|
||||
|
||||
engines: dict = {}
|
||||
|
||||
weasy = WeasyPrintEngine()
|
||||
if await weasy.available():
|
||||
engines["weasyprint"] = weasy
|
||||
else:
|
||||
logger.error("WeasyPrint engine is unavailable, the service will rely on Chromium only")
|
||||
|
||||
chromium = ChromiumEngine()
|
||||
if settings.chromium_enabled:
|
||||
engines["chromium"] = chromium
|
||||
else:
|
||||
logger.info("Chromium engine is disabled by configuration")
|
||||
|
||||
if not engines:
|
||||
logger.error("No render engine is available, conversion requests will fail")
|
||||
|
||||
storage = Storage(settings.storage_dir)
|
||||
pipeline = ConversionPipeline(engines, settings)
|
||||
manager = JobManager(pipeline, storage, settings)
|
||||
|
||||
set_container(
|
||||
Container(
|
||||
settings=settings,
|
||||
storage=storage,
|
||||
pipeline=pipeline,
|
||||
jobs=manager,
|
||||
engines=engines,
|
||||
)
|
||||
)
|
||||
|
||||
await manager.start()
|
||||
logger.info(
|
||||
"Service started",
|
||||
extra={"engines": sorted(engines), "root_path": settings.root_path},
|
||||
)
|
||||
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
await manager.stop()
|
||||
for engine in engines.values():
|
||||
await engine.shutdown()
|
||||
logger.info("Service stopped")
|
||||
|
||||
|
||||
settings = get_settings()
|
||||
setup_logging(settings.log_level)
|
||||
|
||||
app = FastAPI(
|
||||
title=APP_NAME,
|
||||
version=APP_VERSION,
|
||||
root_path=ROOT_PATH
|
||||
title=settings.app_name,
|
||||
version=settings.app_version,
|
||||
description=DESCRIPTION,
|
||||
root_path=settings.root_path,
|
||||
lifespan=lifespan,
|
||||
)
|
||||
|
||||
@app.get("/health")
|
||||
def health():
|
||||
return {"status": "ok"}
|
||||
|
||||
@app.get("/version")
|
||||
def version():
|
||||
return {
|
||||
"app": APP_NAME,
|
||||
"version": APP_VERSION,
|
||||
"language": "python",
|
||||
"root_path": ROOT_PATH
|
||||
}
|
||||
@app.exception_handler(ConversionError)
|
||||
async def conversion_error_handler(request: Request, exc: ConversionError) -> JSONResponse:
|
||||
logger.warning(
|
||||
"Request failed",
|
||||
extra={"error_code": exc.error_code, "path": request.url.path, "status_code": exc.status_code},
|
||||
)
|
||||
return JSONResponse(status_code=exc.status_code, content=exc.to_dict())
|
||||
|
||||
|
||||
@app.exception_handler(Exception)
|
||||
async def unhandled_error_handler(request: Request, exc: Exception) -> JSONResponse:
|
||||
logger.exception("Unhandled error", extra={"path": request.url.path})
|
||||
return JSONResponse(
|
||||
status_code=500,
|
||||
content={
|
||||
"error_code": "internal_error",
|
||||
"message": "Doslo k neocekavane chybe sluzby.",
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
app.include_router(health_router.router)
|
||||
app.include_router(convert_router.router)
|
||||
app.include_router(jobs_router.router)
|
||||
|
||||
+170
@@ -0,0 +1,170 @@
|
||||
"""Request and response models.
|
||||
|
||||
Only `source` is mandatory. Everything else has a working default so that
|
||||
{"source": {"url": "..."}} produces a usable PDF.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime
|
||||
from typing import Literal
|
||||
|
||||
from pydantic import BaseModel, Field, model_validator
|
||||
|
||||
from .config import get_settings
|
||||
|
||||
EngineName = Literal["auto", "weasyprint", "chromium"]
|
||||
PagePosition = Literal[
|
||||
"top-left", "top-center", "top-right",
|
||||
"bottom-left", "bottom-center", "bottom-right",
|
||||
]
|
||||
JobStatus = Literal["queued", "running", "done", "failed", "cancelled", "expired"]
|
||||
|
||||
|
||||
class Source(BaseModel):
|
||||
url: str | None = Field(default=None, description="Adresa HTML dokumentu ke konverzi.")
|
||||
html: str | None = Field(default=None, description="HTML poslane primo v tele requestu.")
|
||||
base_url: str | None = Field(
|
||||
default=None,
|
||||
description="Zaklad pro relativni cesty. Pouziva se hlavne spolu s polem html.",
|
||||
)
|
||||
|
||||
@model_validator(mode="after")
|
||||
def exactly_one_source(self) -> "Source":
|
||||
if bool(self.url) == bool(self.html):
|
||||
raise ValueError("Vyplnte prave jedno z poli source.url a source.html.")
|
||||
return self
|
||||
|
||||
|
||||
class Margin(BaseModel):
|
||||
top: str = "20mm"
|
||||
right: str = "15mm"
|
||||
bottom: str = "20mm"
|
||||
left: str = "15mm"
|
||||
|
||||
|
||||
class PageSettings(BaseModel):
|
||||
format: str = Field(default="A4", description="Nazev formatu (A4, A5, Letter) nebo rozmer 210mm 297mm.")
|
||||
orientation: Literal["portrait", "landscape"] = "portrait"
|
||||
margin: Margin = Field(default_factory=Margin)
|
||||
|
||||
|
||||
class PageNumbers(BaseModel):
|
||||
enabled: bool = False
|
||||
format: str = Field(default="{page} / {pages}", description="Zastupne symboly {page} a {pages}.")
|
||||
position: PagePosition = "bottom-center"
|
||||
mode: Literal["auto", "css", "overlay"] = Field(
|
||||
default="auto",
|
||||
description=(
|
||||
"auto zvoli css u necleneneho dokumentu a overlay u clenene nebo u Chromia. "
|
||||
"css pouziva CSS countery, overlay dopisuje cisla do hotoveho PDF."
|
||||
),
|
||||
)
|
||||
start_at: int = 1
|
||||
|
||||
|
||||
class TocSettings(BaseModel):
|
||||
enabled: bool = False
|
||||
depth: int = Field(default=3, ge=1, le=6)
|
||||
title: str = "Obsah"
|
||||
|
||||
|
||||
class AssetSettings(BaseModel):
|
||||
allow_remote: bool = True
|
||||
timeout_seconds: int = Field(default_factory=lambda: get_settings().asset_timeout_seconds, ge=1)
|
||||
|
||||
|
||||
class ChunkSettings(BaseModel):
|
||||
enabled: bool = True
|
||||
pages_per_chunk: int = Field(default=50, ge=1)
|
||||
|
||||
|
||||
class WaitFor(BaseModel):
|
||||
"""Chromium only. Ignored by the WeasyPrint engine."""
|
||||
|
||||
state: Literal["load", "domcontentloaded", "networkidle"] = "load"
|
||||
selector: str | None = None
|
||||
timeout_seconds: int = Field(default=30, ge=1)
|
||||
|
||||
|
||||
class ConvertRequest(BaseModel):
|
||||
source: Source
|
||||
engine: EngineName = Field(default_factory=lambda: get_settings().default_engine) # type: ignore[arg-type]
|
||||
page: PageSettings = Field(default_factory=PageSettings)
|
||||
page_numbers: PageNumbers = Field(default_factory=PageNumbers)
|
||||
toc: TocSettings = Field(default_factory=TocSettings)
|
||||
outline: bool = Field(default=True, description="Generovat zalozky PDF z nadpisu h1 az h6.")
|
||||
pdf_profile: Literal["pdf/a-1b", "pdf/a-2b", "pdf/a-3b", "pdf/a-4b", "pdf/ua-1"] | None = None
|
||||
assets: AssetSettings = Field(default_factory=AssetSettings)
|
||||
chunking: ChunkSettings = Field(default_factory=ChunkSettings)
|
||||
wait_for: WaitFor = Field(default_factory=WaitFor)
|
||||
filename: str | None = Field(default=None, description="Nazev souboru ve Content-Disposition.")
|
||||
callback_url: str | None = Field(
|
||||
default=None,
|
||||
description="Volitelna adresa, na kterou se po dokonceni jobu posle POST se stavem jobu.",
|
||||
)
|
||||
|
||||
model_config = {
|
||||
"json_schema_extra": {
|
||||
"examples": [
|
||||
{"source": {"url": "https://example.com/dokument.html"}},
|
||||
{
|
||||
"source": {"url": "https://example.com/velky-dokument.html"},
|
||||
"engine": "weasyprint",
|
||||
"page": {"format": "A4", "orientation": "portrait"},
|
||||
"page_numbers": {"enabled": True, "format": "{page} / {pages}"},
|
||||
"toc": {"enabled": True, "depth": 3, "title": "Obsah"},
|
||||
"chunking": {"enabled": True, "pages_per_chunk": 50},
|
||||
},
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
class MissingAsset(BaseModel):
|
||||
url: str
|
||||
reason: str
|
||||
|
||||
|
||||
class JobProgress(BaseModel):
|
||||
pages_rendered: int = 0
|
||||
chunks_done: int = 0
|
||||
chunks_total: int = 0
|
||||
pass_number: int = 0
|
||||
|
||||
|
||||
class ErrorInfo(BaseModel):
|
||||
error_code: str
|
||||
message: str
|
||||
detail: dict | None = None
|
||||
|
||||
|
||||
class JobState(BaseModel):
|
||||
job_id: str
|
||||
status: JobStatus
|
||||
created_at: datetime
|
||||
started_at: datetime | None = None
|
||||
finished_at: datetime | None = None
|
||||
expires_at: datetime | None = None
|
||||
progress: JobProgress = Field(default_factory=JobProgress)
|
||||
engine_used: str | None = None
|
||||
page_count: int | None = None
|
||||
missing_assets: list[MissingAsset] = Field(default_factory=list)
|
||||
warnings: list[str] = Field(default_factory=list)
|
||||
error: ErrorInfo | None = None
|
||||
result_url: str | None = None
|
||||
|
||||
|
||||
class JobAccepted(BaseModel):
|
||||
job_id: str
|
||||
status: JobStatus
|
||||
created_at: datetime
|
||||
result_url: str
|
||||
|
||||
|
||||
class HealthResponse(BaseModel):
|
||||
status: Literal["ok", "degraded"]
|
||||
app: str
|
||||
version: str
|
||||
engines: dict[str, bool]
|
||||
queue: dict[str, int]
|
||||
@@ -0,0 +1,142 @@
|
||||
"""Splitting of a large document into renderable chunks.
|
||||
|
||||
A thousand page document rendered in one pass keeps the whole page tree in
|
||||
memory. Splitting it on structural boundaries keeps memory flat at the cost of
|
||||
having to reassemble the page numbering afterwards.
|
||||
|
||||
Cuts are only ever made between direct children of the container element, so a
|
||||
table or a paragraph is never torn in half.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
from lxml import html as lxml_html
|
||||
|
||||
from .document import SourceDocument
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
SPLIT_TAGS = {"section", "article", "h1"}
|
||||
BREAK_KEYWORDS = ("page-break-before", "break-before")
|
||||
|
||||
# Rough page size heuristic. The real page count is only known after rendering,
|
||||
# so this only has to be good enough to keep chunks roughly even.
|
||||
CHARS_PER_PAGE = 2200
|
||||
IMAGE_CHAR_WEIGHT = 900
|
||||
TABLE_ROW_CHAR_WEIGHT = 120
|
||||
|
||||
|
||||
def is_split_point(element) -> bool:
|
||||
if not isinstance(element.tag, str):
|
||||
return False
|
||||
if element.tag.lower() in SPLIT_TAGS:
|
||||
return True
|
||||
if element.get("data-chunk") is not None:
|
||||
return True
|
||||
style = (element.get("style") or "").lower()
|
||||
return any(keyword in style for keyword in BREAK_KEYWORDS)
|
||||
|
||||
|
||||
def estimate_pages(element) -> float:
|
||||
text_length = len(element.text_content())
|
||||
text_length += IMAGE_CHAR_WEIGHT * len(element.findall(".//img"))
|
||||
text_length += TABLE_ROW_CHAR_WEIGHT * len(element.findall(".//tr"))
|
||||
return max(text_length / CHARS_PER_PAGE, 0.01)
|
||||
|
||||
|
||||
def split_document(document: SourceDocument, pages_per_chunk: int) -> list[str]:
|
||||
"""Return one complete HTML document per chunk.
|
||||
|
||||
Falls back to a single chunk when the document has no usable split points.
|
||||
"""
|
||||
container = document.container
|
||||
children = [child for child in container if isinstance(child.tag, str)]
|
||||
|
||||
split_indexes = [index for index, child in enumerate(children) if is_split_point(child)]
|
||||
if len(split_indexes) < 2:
|
||||
logger.info(
|
||||
"Document has no usable split points, rendering in one pass",
|
||||
extra={"split_points": len(split_indexes), "children": len(children)},
|
||||
)
|
||||
return [document.to_html()]
|
||||
|
||||
groups = _group_children(children, set(split_indexes), pages_per_chunk)
|
||||
if len(groups) < 2:
|
||||
logger.info("Document fits into a single chunk", extra={"children": len(children)})
|
||||
return [document.to_html()]
|
||||
|
||||
prefix, suffix = _skeleton(document, container)
|
||||
child_html = [lxml_html.tostring(child, encoding="unicode") for child in children]
|
||||
|
||||
chunks = ["".join((prefix, *(child_html[index] for index in group), suffix)) for group in groups]
|
||||
logger.info(
|
||||
"Document split into chunks",
|
||||
extra={"chunks": len(chunks), "children": len(children), "pages_per_chunk": pages_per_chunk},
|
||||
)
|
||||
return chunks
|
||||
|
||||
|
||||
def _group_children(children, split_indexes: set[int], pages_per_chunk: int) -> list[list[int]]:
|
||||
groups: list[list[int]] = []
|
||||
current: list[int] = []
|
||||
current_pages = 0.0
|
||||
|
||||
for index, child in enumerate(children):
|
||||
starts_chunk = index in split_indexes and current and current_pages >= pages_per_chunk
|
||||
if starts_chunk:
|
||||
groups.append(current)
|
||||
current = []
|
||||
current_pages = 0.0
|
||||
|
||||
current.append(index)
|
||||
current_pages += estimate_pages(child)
|
||||
|
||||
if current:
|
||||
groups.append(current)
|
||||
|
||||
return groups
|
||||
|
||||
|
||||
def _skeleton(document: SourceDocument, container) -> tuple[str, str]:
|
||||
"""Opening and closing markup shared by every chunk.
|
||||
|
||||
The whole ancestor chain is recreated with its attributes so CSS selectors
|
||||
that depend on it keep matching inside a chunk.
|
||||
"""
|
||||
head_html = lxml_html.tostring(document.head, encoding="unicode")
|
||||
|
||||
chain = []
|
||||
node = container
|
||||
while node is not None and node is not document.tree:
|
||||
chain.append(node)
|
||||
node = node.getparent()
|
||||
chain.reverse()
|
||||
|
||||
opens = [_open_tag(document.tree)]
|
||||
opens.append(head_html)
|
||||
closes = ["</html>"]
|
||||
|
||||
for node in chain:
|
||||
opens.append(_open_tag(node))
|
||||
closes.append(f"</{node.tag}>")
|
||||
|
||||
closes.reverse()
|
||||
return "<!DOCTYPE html>\n" + "".join(opens), "".join(closes)
|
||||
|
||||
|
||||
def _open_tag(element) -> str:
|
||||
attributes = "".join(
|
||||
f' {name}="{_escape(value)}"' for name, value in element.attrib.items()
|
||||
)
|
||||
return f"<{element.tag}{attributes}>"
|
||||
|
||||
|
||||
def _escape(value: str) -> str:
|
||||
return (
|
||||
value.replace("&", "&")
|
||||
.replace('"', """)
|
||||
.replace("<", "<")
|
||||
.replace(">", ">")
|
||||
)
|
||||
@@ -0,0 +1,186 @@
|
||||
"""Parsing of the source document, heading extraction and table of contents."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
|
||||
from lxml import html as lxml_html
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
HEADING_TAGS = ("h1", "h2", "h3", "h4", "h5", "h6")
|
||||
ID_SAFE = re.compile(r"[^a-zA-Z0-9_-]+")
|
||||
|
||||
TOC_PLACEHOLDER = "\u2007\u2007\u2007" # figure spaces, keeps the pass 1 layout stable
|
||||
|
||||
|
||||
@dataclass
|
||||
class Heading:
|
||||
level: int
|
||||
text: str
|
||||
anchor: str
|
||||
page: int | None = None
|
||||
|
||||
|
||||
class SourceDocument:
|
||||
"""Thin wrapper over the parsed document with the operations we need."""
|
||||
|
||||
def __init__(self, html: str) -> None:
|
||||
self.tree = lxml_html.document_fromstring(html)
|
||||
self.head = self.tree.find("head")
|
||||
if self.head is None:
|
||||
self.head = lxml_html.Element("head")
|
||||
self.tree.insert(0, self.head)
|
||||
self.body = self.tree.find("body")
|
||||
if self.body is None:
|
||||
raise ValueError("Dokument neobsahuje element body.")
|
||||
self.container = self._find_container()
|
||||
self._toc_css_added = False
|
||||
|
||||
def _find_container(self):
|
||||
"""Element whose children are the natural split points.
|
||||
|
||||
Many documents wrap everything in a single <main> or <div>. Splitting the
|
||||
body would then produce a single chunk, so we descend through up to two
|
||||
single child wrappers.
|
||||
"""
|
||||
container = self.body
|
||||
for _ in range(2):
|
||||
children = [child for child in container if isinstance(child.tag, str)]
|
||||
if len(children) == 1 and len(list(children[0])) > 1:
|
||||
container = children[0]
|
||||
continue
|
||||
break
|
||||
return container
|
||||
|
||||
# -- styles ---------------------------------------------------------
|
||||
def append_stylesheet(self, css: str) -> None:
|
||||
"""Append a stylesheet as the last element of head so it wins on ties."""
|
||||
style = lxml_html.Element("style")
|
||||
style.set("type", "text/css")
|
||||
style.text = css
|
||||
self.head.append(style)
|
||||
|
||||
def set_base_url(self, base_url: str | None) -> None:
|
||||
if not base_url or self.head.find("base") is not None:
|
||||
return
|
||||
base = lxml_html.Element("base")
|
||||
base.set("href", base_url)
|
||||
self.head.insert(0, base)
|
||||
|
||||
# -- headings -------------------------------------------------------
|
||||
def collect_headings(self, max_depth: int) -> list[Heading]:
|
||||
"""Assign ids to headings that lack one and return them in document order."""
|
||||
headings: list[Heading] = []
|
||||
used: set[str] = {el.get("id") for el in self.tree.iter() if el.get("id")}
|
||||
|
||||
for element in self.body.iter(*HEADING_TAGS):
|
||||
level = int(element.tag[1])
|
||||
if level > max_depth:
|
||||
continue
|
||||
|
||||
text = " ".join(element.text_content().split())
|
||||
if not text:
|
||||
continue
|
||||
|
||||
anchor = element.get("id")
|
||||
if not anchor:
|
||||
anchor = self._unique_anchor(text, used)
|
||||
element.set("id", anchor)
|
||||
used.add(anchor)
|
||||
|
||||
headings.append(Heading(level=level, text=text, anchor=anchor))
|
||||
|
||||
return headings
|
||||
|
||||
@staticmethod
|
||||
def _unique_anchor(text: str, used: set[str]) -> str:
|
||||
base = ID_SAFE.sub("-", text.strip().lower()).strip("-") or "nadpis"
|
||||
base = f"htp-{base[:60]}"
|
||||
candidate = base
|
||||
counter = 2
|
||||
while candidate in used:
|
||||
candidate = f"{base}-{counter}"
|
||||
counter += 1
|
||||
return candidate
|
||||
|
||||
# -- table of contents ----------------------------------------------
|
||||
def insert_toc(self, headings: list[Heading], title: str, pages: dict[str, int] | None) -> None:
|
||||
"""Insert the table of contents as the first block of the body.
|
||||
|
||||
Pass 1 uses a fixed width placeholder instead of the page number so the
|
||||
table keeps exactly the same layout in pass 2.
|
||||
"""
|
||||
container = lxml_html.Element("nav")
|
||||
container.set("id", "htp-toc")
|
||||
container.set("class", "htp-toc")
|
||||
|
||||
heading = lxml_html.Element("h1")
|
||||
heading.set("class", "htp-toc-title")
|
||||
heading.text = title
|
||||
container.append(heading)
|
||||
|
||||
table = lxml_html.Element("table")
|
||||
table.set("class", "htp-toc-table")
|
||||
tbody = lxml_html.Element("tbody")
|
||||
|
||||
for item in headings:
|
||||
if item.anchor == "htp-toc-title":
|
||||
continue
|
||||
row = lxml_html.Element("tr")
|
||||
row.set("class", f"htp-toc-level-{item.level}")
|
||||
|
||||
label_cell = lxml_html.Element("td")
|
||||
label_cell.set("class", "htp-toc-label")
|
||||
link = lxml_html.Element("a")
|
||||
link.set("href", f"#{item.anchor}")
|
||||
link.text = item.text
|
||||
label_cell.append(link)
|
||||
|
||||
page_cell = lxml_html.Element("td")
|
||||
page_cell.set("class", "htp-toc-page")
|
||||
if pages is None:
|
||||
page_cell.text = TOC_PLACEHOLDER
|
||||
else:
|
||||
page_cell.text = str(pages.get(item.anchor, "")) or TOC_PLACEHOLDER
|
||||
|
||||
row.append(label_cell)
|
||||
row.append(page_cell)
|
||||
tbody.append(row)
|
||||
|
||||
table.append(tbody)
|
||||
container.append(table)
|
||||
|
||||
spacer = lxml_html.Element("div")
|
||||
spacer.set("class", "htp-toc-break")
|
||||
container.append(spacer)
|
||||
|
||||
self.container.insert(0, container)
|
||||
if not self._toc_css_added:
|
||||
self.append_stylesheet(TOC_CSS)
|
||||
self._toc_css_added = True
|
||||
|
||||
def remove_toc(self) -> None:
|
||||
existing = self.tree.find(".//nav[@id='htp-toc']")
|
||||
if existing is not None:
|
||||
existing.getparent().remove(existing)
|
||||
|
||||
# -- serialization ---------------------------------------------------
|
||||
def to_html(self) -> str:
|
||||
return "<!DOCTYPE html>\n" + lxml_html.tostring(self.tree, encoding="unicode")
|
||||
|
||||
|
||||
TOC_CSS = """
|
||||
.htp-toc { break-after: page; }
|
||||
.htp-toc-table { width: 100%; border-collapse: collapse; }
|
||||
.htp-toc-table td { padding: 2pt 0; vertical-align: bottom; }
|
||||
.htp-toc-label a { text-decoration: none; color: inherit; }
|
||||
.htp-toc-page { text-align: right; width: 4em; font-variant-numeric: tabular-nums; white-space: nowrap; }
|
||||
.htp-toc-level-2 .htp-toc-label { padding-left: 1.2em; }
|
||||
.htp-toc-level-3 .htp-toc-label { padding-left: 2.4em; }
|
||||
.htp-toc-level-4 .htp-toc-label { padding-left: 3.6em; }
|
||||
.htp-toc-level-5 .htp-toc-label { padding-left: 4.8em; }
|
||||
.htp-toc-level-6 .htp-toc-label { padding-left: 6em; }
|
||||
"""
|
||||
@@ -0,0 +1,53 @@
|
||||
"""Merging of chunk PDFs and reading of basic page geometry."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def merge(paths: list[Path], output_path: Path) -> int:
|
||||
"""Concatenate chunk PDFs into one file.
|
||||
|
||||
PdfWriter.append is used on purpose, it carries over bookmarks and internal
|
||||
links and shifts their page references by the running offset.
|
||||
"""
|
||||
from pypdf import PdfWriter
|
||||
|
||||
if not paths:
|
||||
raise ValueError("Neni co slucovat, seznam PDF je prazdny.")
|
||||
|
||||
if len(paths) == 1:
|
||||
paths[0].replace(output_path)
|
||||
return page_count(output_path)
|
||||
|
||||
writer = PdfWriter()
|
||||
try:
|
||||
for path in paths:
|
||||
writer.append(str(path))
|
||||
with output_path.open("wb") as handle:
|
||||
writer.write(handle)
|
||||
finally:
|
||||
writer.close()
|
||||
|
||||
total = page_count(output_path)
|
||||
logger.info("Chunks merged", extra={"chunks": len(paths), "pages": total})
|
||||
return total
|
||||
|
||||
|
||||
def page_count(path: Path) -> int:
|
||||
from pypdf import PdfReader
|
||||
|
||||
with path.open("rb") as handle:
|
||||
return len(PdfReader(handle).pages)
|
||||
|
||||
|
||||
def first_page_size(path: Path) -> tuple[float, float]:
|
||||
"""Width and height of the first page in points."""
|
||||
from pypdf import PdfReader
|
||||
|
||||
with path.open("rb") as handle:
|
||||
box = PdfReader(handle).pages[0].mediabox
|
||||
return float(box.width), float(box.height)
|
||||
@@ -0,0 +1,80 @@
|
||||
"""Page numbering of the merged document.
|
||||
|
||||
Two ways to number pages:
|
||||
|
||||
css
|
||||
CSS counters do the work during the render. Correct and cheap, but it only
|
||||
works when the whole document is rendered in one pass, because each chunk
|
||||
restarts the page counter.
|
||||
|
||||
overlay
|
||||
A transparent numbering layer with the same page size is rendered once and
|
||||
merged onto the finished PDF. This is the only option for a chunked document
|
||||
and for Chromium, which has no usable page counters.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
from ..errors import EngineUnavailableError
|
||||
from ..models import PageNumbers, PageSettings
|
||||
from .styles import build_overlay_css
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def build_overlay(
|
||||
total_pages: int,
|
||||
width_pt: float,
|
||||
height_pt: float,
|
||||
page: PageSettings,
|
||||
page_numbers: PageNumbers,
|
||||
output_path: Path,
|
||||
) -> Path:
|
||||
"""Render the numbering layer, one empty page per page of the document."""
|
||||
try:
|
||||
from weasyprint import HTML
|
||||
except ImportError as exc:
|
||||
raise EngineUnavailableError(
|
||||
"Cislovani stranek vyzaduje nainstalovany WeasyPrint, ktery kresli cislovaci vrstvu.",
|
||||
) from exc
|
||||
|
||||
css = build_overlay_css(width_pt, height_pt, page, page_numbers, total_pages)
|
||||
slots = '<div class="pdf-page-slot"></div>' * total_pages
|
||||
html = (
|
||||
"<!DOCTYPE html><html><head><meta charset=\"utf-8\">"
|
||||
f"<style>{css}</style></head><body>{slots}</body></html>"
|
||||
)
|
||||
|
||||
HTML(string=html).write_pdf(target=str(output_path))
|
||||
logger.info("Numbering overlay rendered", extra={"pages": total_pages})
|
||||
return output_path
|
||||
|
||||
|
||||
def apply_overlay(document_path: Path, overlay_path: Path, output_path: Path) -> None:
|
||||
"""Stamp the numbering layer onto every page of the document."""
|
||||
from pypdf import PdfReader, PdfWriter
|
||||
|
||||
writer = PdfWriter(clone_from=str(document_path))
|
||||
try:
|
||||
with overlay_path.open("rb") as handle:
|
||||
overlay = PdfReader(handle)
|
||||
available = len(overlay.pages)
|
||||
|
||||
if available < len(writer.pages):
|
||||
logger.warning(
|
||||
"Numbering overlay has fewer pages than the document, tail will stay unnumbered",
|
||||
extra={"overlay_pages": available, "document_pages": len(writer.pages)},
|
||||
)
|
||||
|
||||
for index, page in enumerate(writer.pages):
|
||||
if index >= available:
|
||||
break
|
||||
page.merge_page(overlay.pages[index])
|
||||
|
||||
with output_path.open("wb") as target:
|
||||
writer.write(target)
|
||||
finally:
|
||||
writer.close()
|
||||
@@ -0,0 +1,115 @@
|
||||
"""Generation of the page stylesheet injected into the source document."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from ..models import PageNumbers, PageSettings
|
||||
|
||||
NAMED_SIZE = re.compile(r"^[A-Za-z][A-Za-z0-9]*$")
|
||||
|
||||
MARGIN_BOXES = {
|
||||
"top-left": "@top-left",
|
||||
"top-center": "@top-center",
|
||||
"top-right": "@top-right",
|
||||
"bottom-left": "@bottom-left",
|
||||
"bottom-center": "@bottom-center",
|
||||
"bottom-right": "@bottom-right",
|
||||
}
|
||||
|
||||
PLACEHOLDER = re.compile(r"(\{page\}|\{pages\})")
|
||||
|
||||
|
||||
def page_size_value(page: PageSettings) -> str:
|
||||
fmt = page.format.strip()
|
||||
if NAMED_SIZE.match(fmt):
|
||||
return f"{fmt} {page.orientation}"
|
||||
# Explicit dimensions already carry the orientation.
|
||||
return fmt
|
||||
|
||||
|
||||
def margin_shorthand(page: PageSettings) -> str:
|
||||
m = page.margin
|
||||
return f"{m.top} {m.right} {m.bottom} {m.left}"
|
||||
|
||||
|
||||
def css_content_value(fmt: str, total_pages: int | None) -> str:
|
||||
"""Turn "{page} / {pages}" into a CSS content value.
|
||||
|
||||
When total_pages is known the total is written as a literal, otherwise the
|
||||
CSS counter(pages) is used.
|
||||
"""
|
||||
parts: list[str] = []
|
||||
for token in PLACEHOLDER.split(fmt):
|
||||
if token == "{page}":
|
||||
parts.append("counter(page)")
|
||||
elif token == "{pages}":
|
||||
parts.append(str(total_pages) if total_pages is not None else "counter(pages)")
|
||||
elif token:
|
||||
escaped = token.replace("\\", "\\\\").replace('"', '\\"')
|
||||
parts.append(f'"{escaped}"')
|
||||
return " ".join(parts) if parts else '""'
|
||||
|
||||
|
||||
def build_page_css(
|
||||
page: PageSettings,
|
||||
page_numbers: PageNumbers | None = None,
|
||||
total_pages: int | None = None,
|
||||
outline: bool = True,
|
||||
) -> str:
|
||||
"""Stylesheet applied on top of the document styles."""
|
||||
|
||||
rules = [
|
||||
"@page {",
|
||||
f" size: {page_size_value(page)};",
|
||||
f" margin: {margin_shorthand(page)};",
|
||||
]
|
||||
|
||||
if page_numbers is not None and page_numbers.enabled:
|
||||
box = MARGIN_BOXES[page_numbers.position]
|
||||
rules.append(f" {box} {{")
|
||||
rules.append(f" content: {css_content_value(page_numbers.format, total_pages)};")
|
||||
rules.append(" font-size: 9pt;")
|
||||
rules.append(" color: #444;")
|
||||
rules.append(" }")
|
||||
|
||||
rules.append("}")
|
||||
|
||||
if not outline:
|
||||
rules.append("h1, h2, h3, h4, h5, h6 { bookmark-level: none; }")
|
||||
|
||||
return "\n".join(rules)
|
||||
|
||||
|
||||
def build_overlay_css(
|
||||
width_pt: float,
|
||||
height_pt: float,
|
||||
page: PageSettings,
|
||||
page_numbers: PageNumbers,
|
||||
total_pages: int,
|
||||
) -> str:
|
||||
"""Stylesheet for the transparent numbering layer merged onto the final PDF.
|
||||
|
||||
The size comes from the produced PDF itself, so the overlay always matches
|
||||
even when the source document declares its own @page size.
|
||||
"""
|
||||
box = MARGIN_BOXES[page_numbers.position]
|
||||
reset = ""
|
||||
if page_numbers.start_at != 1:
|
||||
reset = f"body {{ counter-reset: page {page_numbers.start_at - 1}; }}\n"
|
||||
|
||||
return (
|
||||
f"@page {{\n"
|
||||
f" size: {width_pt:.2f}pt {height_pt:.2f}pt;\n"
|
||||
f" margin: {margin_shorthand(page)};\n"
|
||||
f" {box} {{\n"
|
||||
f" content: {css_content_value(page_numbers.format, total_pages)};\n"
|
||||
f" font-size: 9pt;\n"
|
||||
f" color: #444;\n"
|
||||
f" }}\n"
|
||||
f"}}\n"
|
||||
f"{reset}"
|
||||
f"body {{ margin: 0; }}\n"
|
||||
f".pdf-page-slot {{ height: 1px; break-after: page; }}\n"
|
||||
f".pdf-page-slot:last-child {{ break-after: auto; }}\n"
|
||||
)
|
||||
@@ -0,0 +1,78 @@
|
||||
"""Synchronous conversion.
|
||||
|
||||
Suitable for smaller documents. Anything that does not finish within
|
||||
SYNC_TIMEOUT_SECONDS is cancelled and the caller is pointed at /jobs.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
|
||||
from fastapi import APIRouter, Response
|
||||
from fastapi.responses import FileResponse
|
||||
from starlette.background import BackgroundTask
|
||||
|
||||
from ..deps import get_container
|
||||
from ..errors import ConversionError, SyncTooLongError, error_from_code
|
||||
from ..models import ConvertRequest
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(tags=["konverze"])
|
||||
|
||||
|
||||
@router.post(
|
||||
"/convert",
|
||||
summary="Synchronni prevod HTML na PDF",
|
||||
response_class=Response,
|
||||
responses={
|
||||
200: {"content": {"application/pdf": {}}, "description": "Hotove PDF."},
|
||||
413: {"description": "Konverze trvala dele nez limit, pouzijte POST /jobs."},
|
||||
},
|
||||
)
|
||||
async def convert(request: ConvertRequest):
|
||||
container = get_container()
|
||||
settings = container.settings
|
||||
manager = container.jobs
|
||||
|
||||
job = manager.submit(request)
|
||||
|
||||
try:
|
||||
await asyncio.wait_for(job.done.wait(), timeout=settings.sync_timeout_seconds)
|
||||
except asyncio.TimeoutError as exc:
|
||||
manager.cancel(job.id)
|
||||
logger.warning(
|
||||
"Synchronous conversion exceeded its budget",
|
||||
extra={"job_id": job.id, "limit_seconds": settings.sync_timeout_seconds},
|
||||
)
|
||||
raise SyncTooLongError(
|
||||
"Konverze presahla limit pro synchronni pozadavek. Pouzijte asynchronni endpoint POST /jobs.",
|
||||
{"limit_seconds": settings.sync_timeout_seconds},
|
||||
) from exc
|
||||
|
||||
if job.state.status != "done":
|
||||
error = job.state.error
|
||||
if error is not None:
|
||||
raise error_from_code(error.error_code, error.message, error.detail)
|
||||
raise ConversionError("Konverze skoncila ve stavu " + job.state.status + ".")
|
||||
|
||||
path = container.storage.result_path(job.id)
|
||||
filename = request.filename or "dokument.pdf"
|
||||
|
||||
headers = {
|
||||
"X-Page-Count": str(job.state.page_count or 0),
|
||||
"X-Engine-Used": job.state.engine_used or "",
|
||||
}
|
||||
if job.state.missing_assets:
|
||||
headers["X-Missing-Assets"] = str(len(job.state.missing_assets))
|
||||
if job.state.warnings:
|
||||
headers["X-Warnings"] = str(len(job.state.warnings))
|
||||
|
||||
return FileResponse(
|
||||
path,
|
||||
media_type="application/pdf",
|
||||
filename=filename,
|
||||
headers=headers,
|
||||
background=BackgroundTask(container.storage.discard, job.id),
|
||||
)
|
||||
@@ -0,0 +1,47 @@
|
||||
"""Health and version endpoints required by AppFactory."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
from fastapi import APIRouter
|
||||
|
||||
from ..deps import get_container
|
||||
from ..models import HealthResponse
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(tags=["service"])
|
||||
|
||||
|
||||
@router.get("/health", response_model=HealthResponse, summary="Stav sluzby")
|
||||
async def health() -> HealthResponse:
|
||||
container = get_container()
|
||||
|
||||
engines = {}
|
||||
for name, engine in container.engines.items():
|
||||
try:
|
||||
engines[name] = await engine.available()
|
||||
except Exception as exc:
|
||||
logger.error("Engine availability check failed", extra={"engine": name}, exc_info=exc)
|
||||
engines[name] = False
|
||||
|
||||
status = "ok" if any(engines.values()) else "degraded"
|
||||
return HealthResponse(
|
||||
status=status,
|
||||
app=container.settings.app_name,
|
||||
version=container.settings.app_version,
|
||||
engines=engines,
|
||||
queue=container.jobs.stats(),
|
||||
)
|
||||
|
||||
|
||||
@router.get("/version", summary="Verze a zakladni informace o sluzbe")
|
||||
async def version() -> dict:
|
||||
settings = get_container().settings
|
||||
return {
|
||||
"app": settings.app_name,
|
||||
"version": settings.app_version,
|
||||
"language": "python",
|
||||
"root_path": settings.root_path,
|
||||
}
|
||||
@@ -0,0 +1,90 @@
|
||||
"""Asynchronous conversion. This is the path for large documents."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
from fastapi import APIRouter, Response, status
|
||||
from fastapi.responses import FileResponse
|
||||
|
||||
from ..deps import get_container
|
||||
from ..errors import ConversionError, JobNotFoundError
|
||||
from ..models import ConvertRequest, JobAccepted, JobState
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(prefix="/jobs", tags=["joby"])
|
||||
|
||||
|
||||
def _result_url(job_id: str) -> str:
|
||||
root = get_container().settings.root_path.rstrip("/")
|
||||
return f"{root}/jobs/{job_id}/result"
|
||||
|
||||
|
||||
@router.post(
|
||||
"",
|
||||
response_model=JobAccepted,
|
||||
status_code=status.HTTP_202_ACCEPTED,
|
||||
summary="Zaradi konverzi do fronty",
|
||||
)
|
||||
async def create_job(request: ConvertRequest) -> JobAccepted:
|
||||
job = get_container().jobs.submit(request)
|
||||
return JobAccepted(
|
||||
job_id=job.id,
|
||||
status=job.state.status,
|
||||
created_at=job.state.created_at,
|
||||
result_url=_result_url(job.id),
|
||||
)
|
||||
|
||||
|
||||
@router.get("/{job_id}", response_model=JobState, summary="Stav jobu")
|
||||
async def job_state(job_id: str) -> JobState:
|
||||
job = get_container().jobs.get(job_id)
|
||||
state = job.state.model_copy()
|
||||
if state.status == "done":
|
||||
state.result_url = _result_url(job_id)
|
||||
return state
|
||||
|
||||
|
||||
@router.get(
|
||||
"/{job_id}/result",
|
||||
summary="Stahne hotove PDF",
|
||||
response_class=Response,
|
||||
responses={
|
||||
200: {"content": {"application/pdf": {}}, "description": "Hotove PDF."},
|
||||
404: {"description": "Job neexistuje nebo uz expiroval."},
|
||||
409: {"description": "Job jeste nedobehl nebo skoncil chybou."},
|
||||
},
|
||||
)
|
||||
async def job_result(job_id: str):
|
||||
container = get_container()
|
||||
job = container.jobs.get(job_id)
|
||||
|
||||
if job.state.status != "done":
|
||||
error = ConversionError(
|
||||
f"Vysledek neni k dispozici, job je ve stavu {job.state.status}.",
|
||||
{"status": job.state.status},
|
||||
)
|
||||
error.error_code = "result_not_ready"
|
||||
error.status_code = 409
|
||||
raise error
|
||||
|
||||
path = container.storage.result_path(job_id)
|
||||
if not path.exists():
|
||||
raise JobNotFoundError("Soubor s vysledkem uz neexistuje, job pravdepodobne expiroval.")
|
||||
|
||||
filename = job.request.filename or "dokument.pdf"
|
||||
return FileResponse(
|
||||
path,
|
||||
media_type="application/pdf",
|
||||
filename=filename,
|
||||
headers={
|
||||
"X-Page-Count": str(job.state.page_count or 0),
|
||||
"X-Engine-Used": job.state.engine_used or "",
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
@router.delete("/{job_id}", response_model=JobState, summary="Zrusi job nebo smaze jeho vysledek")
|
||||
async def delete_job(job_id: str) -> JobState:
|
||||
return get_container().jobs.cancel(job_id).state
|
||||
@@ -0,0 +1,181 @@
|
||||
"""Fetching of the source document and of its assets.
|
||||
|
||||
Both paths go through UrlGuard. Asset failures never abort the render, they are
|
||||
collected and reported back to the caller so nobody silently receives a PDF with
|
||||
missing images.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import logging
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
import httpx
|
||||
|
||||
from ..config import Settings
|
||||
from ..errors import LimitExceededError, SourceUnavailableError
|
||||
from ..models import MissingAsset
|
||||
from .security import UrlGuard
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
USER_AGENT = "html-to-pdf/1.0 (AppFactory)"
|
||||
|
||||
|
||||
@dataclass
|
||||
class FetchedDocument:
|
||||
html: str
|
||||
base_url: str
|
||||
|
||||
|
||||
@dataclass
|
||||
class AssetReport:
|
||||
"""Collects assets that could not be loaded during a render."""
|
||||
|
||||
missing: list[MissingAsset] = field(default_factory=list)
|
||||
_seen: set[str] = field(default_factory=set)
|
||||
|
||||
def add(self, url: str, reason: str) -> None:
|
||||
if url in self._seen:
|
||||
return
|
||||
self._seen.add(url)
|
||||
self.missing.append(MissingAsset(url=url, reason=reason))
|
||||
logger.warning("Asset could not be loaded", extra={"asset_url": url, "reason": reason})
|
||||
|
||||
|
||||
def fetch_document(url: str, guard: UrlGuard, settings: Settings) -> FetchedDocument:
|
||||
"""Download the source HTML, validating every redirect hop."""
|
||||
|
||||
current = url
|
||||
timeout = httpx.Timeout(settings.fetch_timeout_seconds)
|
||||
|
||||
with httpx.Client(follow_redirects=False, timeout=timeout, headers={"User-Agent": USER_AGENT}) as client:
|
||||
for hop in range(settings.max_redirects + 1):
|
||||
guard.check(current)
|
||||
try:
|
||||
response = client.get(current)
|
||||
except httpx.HTTPError as exc:
|
||||
raise SourceUnavailableError(
|
||||
"Zdrojovy dokument se nepodarilo stahnout.",
|
||||
{"url": current, "reason": str(exc)},
|
||||
) from exc
|
||||
|
||||
if response.is_redirect:
|
||||
location = response.headers.get("location")
|
||||
if not location:
|
||||
raise SourceUnavailableError(
|
||||
"Zdroj vratil presmerovani bez hlavicky Location.", {"url": current}
|
||||
)
|
||||
current = str(response.url.join(location))
|
||||
logger.info("Following redirect", extra={"hop": hop + 1, "target": current})
|
||||
continue
|
||||
|
||||
if response.status_code >= 400:
|
||||
raise SourceUnavailableError(
|
||||
f"Zdrojovy dokument vratil HTTP {response.status_code}.",
|
||||
{"url": current, "status_code": response.status_code},
|
||||
)
|
||||
|
||||
_check_size(len(response.content), settings)
|
||||
return FetchedDocument(html=response.text, base_url=str(response.url))
|
||||
|
||||
raise SourceUnavailableError(
|
||||
"Prekrocen maximalni pocet presmerovani.",
|
||||
{"url": url, "max_redirects": settings.max_redirects},
|
||||
)
|
||||
|
||||
|
||||
def _check_size(size: int, settings: Settings) -> None:
|
||||
if settings.max_html_bytes and size > settings.max_html_bytes:
|
||||
raise LimitExceededError(
|
||||
"Zdrojovy dokument je vetsi nez nakonfigurovany limit MAX_HTML_BYTES.",
|
||||
{"size_bytes": size, "limit_bytes": settings.max_html_bytes},
|
||||
)
|
||||
|
||||
|
||||
def build_url_fetcher(guard: UrlGuard, report: AssetReport, allow_remote: bool, timeout_seconds: int):
|
||||
"""Return a WeasyPrint url_fetcher with SSRF checks and failure reporting."""
|
||||
|
||||
from weasyprint import default_url_fetcher
|
||||
|
||||
def fetcher(url: str, timeout: int = 10, ssl_context=None): # noqa: ARG001
|
||||
if url.startswith("data:"):
|
||||
return default_url_fetcher(url)
|
||||
|
||||
if not allow_remote:
|
||||
report.add(url, "stahovani externich assetu je vypnute")
|
||||
return _empty_asset()
|
||||
|
||||
try:
|
||||
guard.check(url)
|
||||
except Exception as exc: # noqa: BLE001 - reported, never silent
|
||||
report.add(url, f"zablokovano: {exc}")
|
||||
return _empty_asset()
|
||||
|
||||
try:
|
||||
response = httpx.get(
|
||||
url,
|
||||
timeout=timeout_seconds,
|
||||
follow_redirects=False,
|
||||
headers={"User-Agent": USER_AGENT},
|
||||
)
|
||||
while response.is_redirect:
|
||||
location = response.headers.get("location")
|
||||
if not location:
|
||||
raise httpx.HTTPError("presmerovani bez hlavicky Location")
|
||||
target = str(response.url.join(location))
|
||||
guard.check(target)
|
||||
response = httpx.get(
|
||||
target,
|
||||
timeout=timeout_seconds,
|
||||
follow_redirects=False,
|
||||
headers={"User-Agent": USER_AGENT},
|
||||
)
|
||||
|
||||
if response.status_code >= 400:
|
||||
report.add(url, f"HTTP {response.status_code}")
|
||||
return _empty_asset()
|
||||
|
||||
return {
|
||||
"string": response.content,
|
||||
"mime_type": response.headers.get("content-type", "").split(";")[0] or None,
|
||||
"redirected_url": str(response.url),
|
||||
}
|
||||
except Exception as exc: # noqa: BLE001 - reported, never silent
|
||||
report.add(url, str(exc))
|
||||
return _empty_asset()
|
||||
|
||||
return fetcher
|
||||
|
||||
|
||||
def _empty_asset() -> dict:
|
||||
"""Placeholder returned instead of a failed asset so the render continues."""
|
||||
return {"file_obj": io.BytesIO(b""), "mime_type": "application/octet-stream"}
|
||||
|
||||
|
||||
@dataclass
|
||||
class AssetGate:
|
||||
"""Per job gateway for everything the render pulls from the network."""
|
||||
|
||||
guard: UrlGuard
|
||||
report: AssetReport
|
||||
allow_remote: bool = True
|
||||
timeout_seconds: int = 10
|
||||
|
||||
def weasy_fetcher(self):
|
||||
return build_url_fetcher(self.guard, self.report, self.allow_remote, self.timeout_seconds)
|
||||
|
||||
def allowed(self, url: str) -> tuple[bool, str]:
|
||||
"""Decide whether a browser initiated request may proceed."""
|
||||
if url.startswith("data:") or url.startswith("blob:") or url.startswith("about:"):
|
||||
return True, ""
|
||||
if not url.startswith(("http://", "https://")):
|
||||
return False, "nepovolene schema"
|
||||
if not self.allow_remote:
|
||||
return False, "stahovani externich assetu je vypnute"
|
||||
try:
|
||||
self.guard.check(url)
|
||||
except Exception as exc: # noqa: BLE001 - reported, never silent
|
||||
return False, str(exc)
|
||||
return True, ""
|
||||
@@ -0,0 +1,238 @@
|
||||
"""In memory job queue.
|
||||
|
||||
The queue is deliberately in process. There is no broker and no database, which
|
||||
means a restart loses queued work. Such jobs are marked as failed with an
|
||||
explicit reason instead of silently disappearing.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import uuid
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
import httpx
|
||||
|
||||
from ..config import Settings
|
||||
from ..errors import ConversionError, JobNotFoundError, QueueFullError
|
||||
from ..logging_setup import current_job_id
|
||||
from ..models import ConvertRequest, ErrorInfo, JobProgress, JobState
|
||||
from .pipeline import ConversionPipeline
|
||||
from .storage import Storage
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
class Job:
|
||||
def __init__(self, job_id: str, request: ConvertRequest) -> None:
|
||||
self.id = job_id
|
||||
self.request = request
|
||||
self.state = JobState(job_id=job_id, status="queued", created_at=_now())
|
||||
self.task = None
|
||||
self.done = asyncio.Event()
|
||||
|
||||
|
||||
class JobManager:
|
||||
def __init__(self, pipeline: ConversionPipeline, storage: Storage, settings: Settings) -> None:
|
||||
self._pipeline = pipeline
|
||||
self._storage = storage
|
||||
self._settings = settings
|
||||
self._jobs = {}
|
||||
self._queue = asyncio.Queue(maxsize=settings.queue_max_size)
|
||||
self._workers = []
|
||||
self._cleaner = None
|
||||
|
||||
async def start(self) -> None:
|
||||
self._storage.sweep_orphans(set(self._jobs))
|
||||
for index in range(max(1, self._settings.workers)):
|
||||
self._workers.append(asyncio.create_task(self._worker(index), name=f"htp-worker-{index}"))
|
||||
self._cleaner = asyncio.create_task(self._cleanup_loop(), name="htp-cleanup")
|
||||
logger.info("Job manager started", extra={"workers": len(self._workers)})
|
||||
|
||||
async def stop(self) -> None:
|
||||
for task in self._workers:
|
||||
task.cancel()
|
||||
if self._cleaner is not None:
|
||||
self._cleaner.cancel()
|
||||
|
||||
for job in self._jobs.values():
|
||||
if job.state.status in ("queued", "running"):
|
||||
self._fail(
|
||||
job,
|
||||
ErrorInfo(
|
||||
error_code="service_restarted",
|
||||
message="Sluzba byla ukoncena drive, nez job dobehl. Odeslete pozadavek znovu.",
|
||||
),
|
||||
)
|
||||
logger.info("Job manager stopped")
|
||||
|
||||
def submit(self, request: ConvertRequest) -> Job:
|
||||
job = Job(str(uuid.uuid4()), request)
|
||||
self._jobs[job.id] = job
|
||||
try:
|
||||
self._queue.put_nowait(job.id)
|
||||
except asyncio.QueueFull as exc:
|
||||
del self._jobs[job.id]
|
||||
raise QueueFullError(
|
||||
"Fronta je plna, zkuste to prosim za chvili.",
|
||||
{"queue_max_size": self._settings.queue_max_size},
|
||||
) from exc
|
||||
|
||||
logger.info("Job queued", extra={"job_id": job.id, "queue_size": self._queue.qsize()})
|
||||
return job
|
||||
|
||||
def get(self, job_id: str) -> Job:
|
||||
job = self._jobs.get(job_id)
|
||||
if job is None:
|
||||
raise JobNotFoundError("Job s timto identifikatorem neexistuje nebo uz expiroval.")
|
||||
return job
|
||||
|
||||
def cancel(self, job_id: str) -> Job:
|
||||
job = self.get(job_id)
|
||||
if job.task is not None and not job.task.done():
|
||||
job.task.cancel()
|
||||
else:
|
||||
job.state.status = "cancelled"
|
||||
job.state.finished_at = _now()
|
||||
self._storage.discard(job.id)
|
||||
job.done.set()
|
||||
job.state.expires_at = _now() + timedelta(seconds=self._settings.job_result_ttl_seconds)
|
||||
logger.info("Job cancelled", extra={"job_id": job_id})
|
||||
return job
|
||||
|
||||
def stats(self) -> dict:
|
||||
counts = {"queued": 0, "running": 0, "done": 0, "failed": 0, "cancelled": 0, "expired": 0}
|
||||
for job in self._jobs.values():
|
||||
counts[job.state.status] = counts.get(job.state.status, 0) + 1
|
||||
counts["workers"] = len(self._workers)
|
||||
return counts
|
||||
|
||||
async def _worker(self, index: int) -> None:
|
||||
while True:
|
||||
job_id = await self._queue.get()
|
||||
job = self._jobs.get(job_id)
|
||||
if job is None or job.state.status != "queued":
|
||||
self._queue.task_done()
|
||||
continue
|
||||
|
||||
job.task = asyncio.current_task()
|
||||
token = current_job_id.set(job.id)
|
||||
try:
|
||||
await self._execute(job)
|
||||
except asyncio.CancelledError:
|
||||
job.state.status = "cancelled"
|
||||
job.state.finished_at = _now()
|
||||
job.state.expires_at = _now() + timedelta(seconds=self._settings.job_result_ttl_seconds)
|
||||
self._storage.discard(job.id)
|
||||
logger.info("Job execution cancelled", extra={"job_id": job.id})
|
||||
finally:
|
||||
current_job_id.reset(token)
|
||||
job.task = None
|
||||
job.done.set()
|
||||
self._queue.task_done()
|
||||
await self._notify_callback(job)
|
||||
|
||||
async def _execute(self, job: Job) -> None:
|
||||
job.state.status = "running"
|
||||
job.state.started_at = _now()
|
||||
logger.info("Job started")
|
||||
|
||||
def progress(pages, chunks_done, chunks_total, pass_number):
|
||||
job.state.progress = JobProgress(
|
||||
pages_rendered=pages,
|
||||
chunks_done=chunks_done,
|
||||
chunks_total=chunks_total,
|
||||
pass_number=pass_number,
|
||||
)
|
||||
|
||||
workdir = self._storage.job_dir(job.id)
|
||||
try:
|
||||
result = await self._pipeline.run(job.request, workdir, progress)
|
||||
except ConversionError as exc:
|
||||
logger.error("Job failed", extra={"error_code": exc.error_code}, exc_info=exc)
|
||||
self._fail(job, ErrorInfo(error_code=exc.error_code, message=exc.message, detail=exc.detail or None))
|
||||
return
|
||||
except asyncio.CancelledError:
|
||||
raise
|
||||
except Exception as exc:
|
||||
logger.exception("Job failed with an unexpected error")
|
||||
self._fail(
|
||||
job,
|
||||
ErrorInfo(
|
||||
error_code="internal_error",
|
||||
message="Pri generovani PDF doslo k neocekavane chybe.",
|
||||
detail={"reason": str(exc)},
|
||||
),
|
||||
)
|
||||
return
|
||||
|
||||
self._storage.publish(job.id, result.path)
|
||||
job.state.status = "done"
|
||||
job.state.finished_at = _now()
|
||||
job.state.expires_at = _now() + timedelta(seconds=self._settings.job_result_ttl_seconds)
|
||||
job.state.page_count = result.page_count
|
||||
job.state.engine_used = result.engine_used
|
||||
job.state.missing_assets = result.missing_assets
|
||||
job.state.warnings = result.warnings
|
||||
logger.info(
|
||||
"Job finished",
|
||||
extra={
|
||||
"pages": result.page_count,
|
||||
"engine": result.engine_used,
|
||||
"missing_assets": len(result.missing_assets),
|
||||
},
|
||||
)
|
||||
|
||||
def _fail(self, job: Job, error: ErrorInfo) -> None:
|
||||
job.state.status = "failed"
|
||||
job.state.finished_at = _now()
|
||||
job.state.expires_at = _now() + timedelta(seconds=self._settings.job_result_ttl_seconds)
|
||||
job.state.error = error
|
||||
self._storage.discard(job.id)
|
||||
|
||||
async def _notify_callback(self, job: Job) -> None:
|
||||
url = job.request.callback_url
|
||||
if not url or job.state.status not in ("done", "failed"):
|
||||
return
|
||||
|
||||
payload = job.state.model_dump(mode="json")
|
||||
for attempt in range(1, max(1, self._settings.callback_retries) + 1):
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=self._settings.callback_timeout_seconds) as client:
|
||||
response = await client.post(url, json=payload)
|
||||
if response.status_code < 400:
|
||||
logger.info("Callback delivered", extra={"attempt": attempt})
|
||||
return
|
||||
logger.warning(
|
||||
"Callback returned an error status",
|
||||
extra={"attempt": attempt, "status_code": response.status_code},
|
||||
)
|
||||
except Exception as exc:
|
||||
logger.warning("Callback delivery failed", extra={"attempt": attempt}, exc_info=exc)
|
||||
await asyncio.sleep(min(2 ** attempt, 10))
|
||||
|
||||
logger.error("Callback could not be delivered, job result stays available over the API")
|
||||
|
||||
async def _cleanup_loop(self) -> None:
|
||||
while True:
|
||||
await asyncio.sleep(60)
|
||||
try:
|
||||
self._expire_old_jobs()
|
||||
except Exception:
|
||||
logger.exception("Cleanup loop failed")
|
||||
|
||||
def _expire_old_jobs(self) -> None:
|
||||
now = _now()
|
||||
for job_id, job in list(self._jobs.items()):
|
||||
expires_at = job.state.expires_at
|
||||
if expires_at is None or expires_at > now:
|
||||
continue
|
||||
job.state.status = "expired"
|
||||
self._storage.discard(job_id)
|
||||
del self._jobs[job_id]
|
||||
logger.info("Job result expired and was removed", extra={"job_id": job_id})
|
||||
@@ -0,0 +1,350 @@
|
||||
"""The conversion pipeline.
|
||||
|
||||
Order of operations:
|
||||
|
||||
1. load the source HTML, validating the target address
|
||||
2. pick the engine
|
||||
3. decide how page numbers will be produced
|
||||
4. build the document, inject the page stylesheet, optionally insert the table
|
||||
of contents placeholder
|
||||
5. split into chunks
|
||||
6. first render pass, which yields the real page count and anchor positions
|
||||
7. second render pass when a table of contents needs real page numbers
|
||||
8. merge the chunks
|
||||
9. stamp the numbering overlay when CSS counters cannot be used
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Callable
|
||||
|
||||
from ..config import Settings, get_settings
|
||||
from ..errors import (
|
||||
ConversionError,
|
||||
LimitExceededError,
|
||||
RenderTimeoutError,
|
||||
UnsupportedCombinationError,
|
||||
)
|
||||
from ..models import ConvertRequest, MissingAsset
|
||||
from ..pdf import merger, paginator
|
||||
from ..pdf.chunker import split_document
|
||||
from ..pdf.document import SourceDocument
|
||||
from ..pdf.styles import build_page_css
|
||||
from .fetcher import AssetGate, AssetReport, fetch_document
|
||||
from .security import UrlGuard
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
SCRIPT_TAG = re.compile(r"<script\b[^>]*>(.*?)</script>", re.IGNORECASE | re.DOTALL)
|
||||
SCRIPT_SRC = re.compile(r"<script\b[^>]*\bsrc\s*=", re.IGNORECASE)
|
||||
|
||||
ProgressCallback = Callable[[int, int, int, int], None]
|
||||
|
||||
|
||||
@dataclass
|
||||
class ConversionResult:
|
||||
path: Path
|
||||
page_count: int
|
||||
engine_used: str
|
||||
missing_assets: list[MissingAsset] = field(default_factory=list)
|
||||
warnings: list[str] = field(default_factory=list)
|
||||
|
||||
|
||||
class ConversionPipeline:
|
||||
def __init__(self, engines: dict, settings: Settings | None = None) -> None:
|
||||
self._engines = engines
|
||||
self._settings = settings or get_settings()
|
||||
self._guard = UrlGuard(self._settings)
|
||||
|
||||
async def run(
|
||||
self,
|
||||
request: ConvertRequest,
|
||||
workdir: Path,
|
||||
progress: ProgressCallback | None = None,
|
||||
) -> ConversionResult:
|
||||
workdir.mkdir(parents=True, exist_ok=True)
|
||||
timeout = self._settings.max_render_seconds
|
||||
|
||||
coroutine = self._run_with_fallback(request, workdir, progress)
|
||||
if timeout:
|
||||
try:
|
||||
return await asyncio.wait_for(coroutine, timeout=timeout)
|
||||
except asyncio.TimeoutError as exc:
|
||||
raise RenderTimeoutError(
|
||||
"Render prekrocil nakonfigurovany limit MAX_RENDER_SECONDS.",
|
||||
{"limit_seconds": timeout},
|
||||
) from exc
|
||||
return await coroutine
|
||||
|
||||
async def _run_with_fallback(
|
||||
self, request: ConvertRequest, workdir: Path, progress: ProgressCallback | None
|
||||
) -> ConversionResult:
|
||||
raw_html, base_url = await self._load_source(request)
|
||||
requested = request.engine
|
||||
engine_name = self._select_engine(requested, raw_html)
|
||||
|
||||
try:
|
||||
return await self._execute(request, raw_html, base_url, engine_name, workdir, progress)
|
||||
except ConversionError:
|
||||
raise
|
||||
except Exception as exc:
|
||||
if requested != "auto" or engine_name != "weasyprint" or "chromium" not in self._engines:
|
||||
raise
|
||||
logger.warning(
|
||||
"WeasyPrint render failed, falling back to Chromium",
|
||||
exc_info=exc,
|
||||
extra={"engine": engine_name},
|
||||
)
|
||||
result = await self._execute(request, raw_html, base_url, "chromium", workdir, progress)
|
||||
result.warnings.append(
|
||||
"Render pres WeasyPrint selhal, dokument byl vygenerovan pres Chromium."
|
||||
)
|
||||
return result
|
||||
|
||||
def _select_engine(self, requested: str, raw_html: str) -> str:
|
||||
if requested != "auto":
|
||||
if requested not in self._engines:
|
||||
raise UnsupportedCombinationError(
|
||||
f"Engine {requested} neni v teto instanci k dispozici.",
|
||||
{"available": sorted(self._engines)},
|
||||
)
|
||||
return requested
|
||||
|
||||
if self._has_active_scripts(raw_html) and "chromium" in self._engines:
|
||||
logger.info("Auto engine selected chromium because the document contains scripts")
|
||||
return "chromium"
|
||||
|
||||
if "weasyprint" in self._engines:
|
||||
return "weasyprint"
|
||||
|
||||
return next(iter(self._engines))
|
||||
|
||||
@staticmethod
|
||||
def _has_active_scripts(raw_html: str) -> bool:
|
||||
if SCRIPT_SRC.search(raw_html):
|
||||
return True
|
||||
return any(body.strip() for body in SCRIPT_TAG.findall(raw_html))
|
||||
|
||||
async def _load_source(self, request: ConvertRequest) -> tuple:
|
||||
if request.source.html is not None:
|
||||
html = request.source.html
|
||||
limit = self._settings.max_html_bytes
|
||||
if limit and len(html.encode("utf-8")) > limit:
|
||||
raise LimitExceededError(
|
||||
"Zdrojove HTML je vetsi nez nakonfigurovany limit MAX_HTML_BYTES.",
|
||||
{"limit_bytes": limit},
|
||||
)
|
||||
return html, request.source.base_url
|
||||
|
||||
fetched = await asyncio.to_thread(
|
||||
fetch_document, request.source.url, self._guard, self._settings
|
||||
)
|
||||
return fetched.html, request.source.base_url or fetched.base_url
|
||||
|
||||
async def _execute(
|
||||
self,
|
||||
request: ConvertRequest,
|
||||
raw_html: str,
|
||||
base_url: str | None,
|
||||
engine_name: str,
|
||||
workdir: Path,
|
||||
progress: ProgressCallback | None,
|
||||
) -> ConversionResult:
|
||||
engine = self._engines[engine_name]
|
||||
report = AssetReport()
|
||||
gate = AssetGate(
|
||||
guard=self._guard,
|
||||
report=report,
|
||||
allow_remote=request.assets.allow_remote,
|
||||
timeout_seconds=request.assets.timeout_seconds,
|
||||
)
|
||||
warnings: list[str] = []
|
||||
|
||||
numbering_mode = self._numbering_mode(request, engine_name)
|
||||
|
||||
if request.toc.enabled and not getattr(engine, "supports_anchor_pages", False):
|
||||
raise UnsupportedCombinationError(
|
||||
"Generovani obsahu s cisly stranek podporuje pouze engine weasyprint.",
|
||||
{"engine": engine_name},
|
||||
)
|
||||
|
||||
document = SourceDocument(raw_html)
|
||||
document.set_base_url(base_url)
|
||||
document.append_stylesheet(
|
||||
build_page_css(
|
||||
request.page,
|
||||
request.page_numbers if numbering_mode == "css" else None,
|
||||
total_pages=None,
|
||||
outline=request.outline,
|
||||
)
|
||||
)
|
||||
|
||||
headings = document.collect_headings(request.toc.depth) if request.toc.enabled else []
|
||||
if request.toc.enabled:
|
||||
if not headings:
|
||||
warnings.append("Dokument neobsahuje zadne nadpisy, obsah nebyl vygenerovan.")
|
||||
request.toc.enabled = False
|
||||
else:
|
||||
document.insert_toc(headings, request.toc.title, pages=None)
|
||||
|
||||
chunks = self._split(document, request)
|
||||
direct_navigation = self._can_navigate_directly(request, engine_name, chunks)
|
||||
if direct_navigation:
|
||||
logger.info("Chromium will navigate to the source URL directly so its scripts run in context")
|
||||
|
||||
renders = await self._render_pass(
|
||||
engine, chunks, base_url, request, workdir, gate, progress, 1, direct_navigation
|
||||
)
|
||||
|
||||
if request.toc.enabled:
|
||||
anchor_pages = self._absolute_anchor_pages(renders)
|
||||
missing = [item.anchor for item in headings if item.anchor not in anchor_pages]
|
||||
if missing:
|
||||
logger.warning(
|
||||
"Some headings have no anchor position, their page numbers stay empty",
|
||||
extra={"missing_anchors": len(missing)},
|
||||
)
|
||||
document.remove_toc()
|
||||
document.insert_toc(headings, request.toc.title, pages=anchor_pages)
|
||||
chunks = self._split(document, request)
|
||||
renders = await self._render_pass(
|
||||
engine, chunks, base_url, request, workdir, gate, progress, 2
|
||||
)
|
||||
|
||||
if len(chunks) > 1:
|
||||
warnings.append(
|
||||
"Dokument byl rozdelen na casti, odkazy v obsahu proto nejsou klikatelne. "
|
||||
"Cisla stranek jsou spravna."
|
||||
)
|
||||
|
||||
merged_path = workdir / "merged.pdf"
|
||||
total_pages = merger.merge([item.path for item in renders], merged_path)
|
||||
self._check_page_limit(total_pages)
|
||||
|
||||
final_path = merged_path
|
||||
if request.page_numbers.enabled and numbering_mode == "overlay":
|
||||
final_path = await asyncio.to_thread(
|
||||
self._stamp_numbers, merged_path, workdir, request, total_pages
|
||||
)
|
||||
|
||||
on_job_finished = getattr(engine, "on_job_finished", None)
|
||||
if callable(on_job_finished):
|
||||
on_job_finished()
|
||||
|
||||
return ConversionResult(
|
||||
path=final_path,
|
||||
page_count=total_pages,
|
||||
engine_used=engine_name,
|
||||
missing_assets=report.missing,
|
||||
warnings=warnings,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _can_navigate_directly(request: ConvertRequest, engine_name: str, chunks: list) -> bool:
|
||||
"""Chromium renders a foreign page best when it loads the URL itself.
|
||||
|
||||
Only possible when the document is not split and needs no injected
|
||||
markup, otherwise the modified HTML has to be pushed into the page.
|
||||
"""
|
||||
return (
|
||||
engine_name == "chromium"
|
||||
and request.source.url is not None
|
||||
and not request.toc.enabled
|
||||
and len(chunks) == 1
|
||||
)
|
||||
|
||||
def _numbering_mode(self, request: ConvertRequest, engine_name: str) -> str:
|
||||
mode = request.page_numbers.mode
|
||||
if mode == "auto":
|
||||
if engine_name == "weasyprint" and not request.chunking.enabled:
|
||||
return "css"
|
||||
return "overlay"
|
||||
|
||||
if mode == "css" and engine_name != "weasyprint":
|
||||
raise UnsupportedCombinationError(
|
||||
"Rezim cislovani css funguje pouze s enginem weasyprint. Pouzijte overlay nebo auto.",
|
||||
{"engine": engine_name},
|
||||
)
|
||||
if mode == "css" and request.chunking.enabled:
|
||||
raise UnsupportedCombinationError(
|
||||
"Rezim cislovani css nelze kombinovat s chunkovanim, protoze citac stranek se v kazde "
|
||||
"casti restartuje. Vypnete chunking nebo pouzijte overlay.",
|
||||
)
|
||||
return mode
|
||||
|
||||
def _split(self, document: SourceDocument, request: ConvertRequest) -> list:
|
||||
if not request.chunking.enabled:
|
||||
return [document.to_html()]
|
||||
return split_document(document, request.chunking.pages_per_chunk)
|
||||
|
||||
async def _render_pass(
|
||||
self,
|
||||
engine,
|
||||
chunks: list,
|
||||
base_url: str | None,
|
||||
request: ConvertRequest,
|
||||
workdir: Path,
|
||||
gate: AssetGate,
|
||||
progress: ProgressCallback | None,
|
||||
pass_number: int,
|
||||
direct_navigation: bool = False,
|
||||
) -> list:
|
||||
renders = []
|
||||
pages_rendered = 0
|
||||
|
||||
for index, chunk_html in enumerate(chunks):
|
||||
output = workdir / f"pass{pass_number}-chunk{index:04d}.pdf"
|
||||
render = await engine.render_chunk(
|
||||
None if direct_navigation else chunk_html,
|
||||
base_url,
|
||||
request,
|
||||
output,
|
||||
total_pages=None,
|
||||
asset_gate=gate,
|
||||
)
|
||||
renders.append(render)
|
||||
pages_rendered += render.page_count
|
||||
|
||||
if progress is not None:
|
||||
progress(pages_rendered, index + 1, len(chunks), pass_number)
|
||||
|
||||
self._check_page_limit(pages_rendered)
|
||||
|
||||
logger.info(
|
||||
"Render pass finished",
|
||||
extra={"pass_number": pass_number, "chunks": len(chunks), "pages": pages_rendered},
|
||||
)
|
||||
return renders
|
||||
|
||||
@staticmethod
|
||||
def _absolute_anchor_pages(renders: list) -> dict:
|
||||
pages: dict = {}
|
||||
offset = 0
|
||||
for render in renders:
|
||||
for anchor, local_page in render.anchor_pages.items():
|
||||
pages.setdefault(anchor, offset + local_page + 1)
|
||||
offset += render.page_count
|
||||
return pages
|
||||
|
||||
def _check_page_limit(self, pages: int) -> None:
|
||||
if self._settings.max_pages and pages > self._settings.max_pages:
|
||||
raise LimitExceededError(
|
||||
"Dokument ma vice stranek nez nakonfigurovany limit MAX_PAGES.",
|
||||
{"pages": pages, "limit": self._settings.max_pages},
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _stamp_numbers(merged_path: Path, workdir: Path, request: ConvertRequest, total_pages: int) -> Path:
|
||||
width, height = merger.first_page_size(merged_path)
|
||||
overlay_path = workdir / "overlay.pdf"
|
||||
paginator.build_overlay(
|
||||
total_pages, width, height, request.page, request.page_numbers, overlay_path
|
||||
)
|
||||
numbered_path = workdir / "numbered.pdf"
|
||||
paginator.apply_overlay(merged_path, overlay_path, numbered_path)
|
||||
return numbered_path
|
||||
@@ -0,0 +1,132 @@
|
||||
"""SSRF protection.
|
||||
|
||||
The service fetches arbitrary URLs on request, which is exactly the shape of an
|
||||
SSRF vulnerability. Every URL is validated after DNS resolution, not on the
|
||||
string alone, and the check is repeated on every redirect hop.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ipaddress
|
||||
import logging
|
||||
import socket
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from ..config import Settings
|
||||
from ..errors import BlockedTargetError
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
ALLOWED_SCHEMES = ("http", "https")
|
||||
|
||||
# Ranges that must never be reachable from a user supplied URL.
|
||||
PRIVATE_NETWORKS = [
|
||||
ipaddress.ip_network("127.0.0.0/8"),
|
||||
ipaddress.ip_network("10.0.0.0/8"),
|
||||
ipaddress.ip_network("172.16.0.0/12"),
|
||||
ipaddress.ip_network("192.168.0.0/16"),
|
||||
ipaddress.ip_network("169.254.0.0/16"),
|
||||
ipaddress.ip_network("0.0.0.0/8"),
|
||||
ipaddress.ip_network("100.64.0.0/10"),
|
||||
ipaddress.ip_network("192.0.0.0/24"),
|
||||
ipaddress.ip_network("198.18.0.0/15"),
|
||||
ipaddress.ip_network("224.0.0.0/4"),
|
||||
ipaddress.ip_network("240.0.0.0/4"),
|
||||
ipaddress.ip_network("::1/128"),
|
||||
ipaddress.ip_network("fc00::/7"),
|
||||
ipaddress.ip_network("fe80::/10"),
|
||||
ipaddress.ip_network("::/128"),
|
||||
]
|
||||
|
||||
|
||||
class UrlGuard:
|
||||
"""Validates URLs against the configured policy."""
|
||||
|
||||
def __init__(self, settings: Settings) -> None:
|
||||
self._block_private = settings.ssrf_block_private
|
||||
self._allowed_hosts = {host.lower() for host in settings.ssrf_allowed_hosts}
|
||||
self._extra_blocked: list[ipaddress._BaseNetwork] = []
|
||||
|
||||
for cidr in settings.ssrf_extra_blocked_cidrs:
|
||||
try:
|
||||
self._extra_blocked.append(ipaddress.ip_network(cidr, strict=False))
|
||||
except ValueError:
|
||||
logger.warning("Ignoring invalid CIDR in SSRF_EXTRA_BLOCKED_CIDRS", extra={"cidr": cidr})
|
||||
|
||||
def check(self, url: str) -> str:
|
||||
"""Raise BlockedTargetError when the URL must not be fetched.
|
||||
|
||||
Returns the hostname so callers can reuse it without parsing again.
|
||||
"""
|
||||
parsed = urlparse(url)
|
||||
scheme = (parsed.scheme or "").lower()
|
||||
|
||||
if scheme not in ALLOWED_SCHEMES:
|
||||
raise BlockedTargetError(
|
||||
"Povolena jsou pouze schemata http a https.",
|
||||
{"url": url, "scheme": scheme or None},
|
||||
)
|
||||
|
||||
host = parsed.hostname
|
||||
if not host:
|
||||
raise BlockedTargetError("Adresa neobsahuje hostname.", {"url": url})
|
||||
|
||||
if host.lower() in self._allowed_hosts:
|
||||
logger.info("Host explicitly allowlisted", extra={"host": host})
|
||||
return host
|
||||
|
||||
for address in self._resolve(host, url):
|
||||
self._check_address(address, host, url)
|
||||
|
||||
return host
|
||||
|
||||
def _resolve(self, host: str, url: str) -> list[ipaddress.IPv4Address | ipaddress.IPv6Address]:
|
||||
# A literal IP address needs no lookup.
|
||||
try:
|
||||
return [ipaddress.ip_address(host)]
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
try:
|
||||
infos = socket.getaddrinfo(host, None, proto=socket.IPPROTO_TCP)
|
||||
except socket.gaierror as exc:
|
||||
raise BlockedTargetError(
|
||||
f"Hostname {host} se nepodarilo prelozit na IP adresu.",
|
||||
{"url": url, "reason": str(exc)},
|
||||
) from exc
|
||||
|
||||
addresses = []
|
||||
for info in infos:
|
||||
try:
|
||||
addresses.append(ipaddress.ip_address(info[4][0]))
|
||||
except ValueError:
|
||||
continue
|
||||
|
||||
if not addresses:
|
||||
raise BlockedTargetError(f"Hostname {host} nema zadnou pouzitelnou IP adresu.", {"url": url})
|
||||
|
||||
return addresses
|
||||
|
||||
def _check_address(self, address, host: str, url: str) -> None:
|
||||
if self._block_private:
|
||||
for network in PRIVATE_NETWORKS:
|
||||
if address.version == network.version and address in network:
|
||||
logger.warning(
|
||||
"Blocked request to private address",
|
||||
extra={"host": host, "address": str(address), "network": str(network)},
|
||||
)
|
||||
raise BlockedTargetError(
|
||||
"Cilova adresa smeruje do privatniho nebo vyhrazeneho rozsahu a je zablokovana.",
|
||||
{"url": url, "host": host, "address": str(address)},
|
||||
)
|
||||
|
||||
for network in self._extra_blocked:
|
||||
if address.version == network.version and address in network:
|
||||
logger.warning(
|
||||
"Blocked request by configured CIDR",
|
||||
extra={"host": host, "address": str(address), "network": str(network)},
|
||||
)
|
||||
raise BlockedTargetError(
|
||||
"Cilova adresa je v konfigurovanem seznamu blokovanych rozsahu.",
|
||||
{"url": url, "host": host, "address": str(address)},
|
||||
)
|
||||
@@ -0,0 +1,65 @@
|
||||
"""Temporary storage of job working directories and results.
|
||||
|
||||
Nothing is kept longer than needed. A result lives until it is picked up or
|
||||
until its TTL expires, whichever comes first.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import shutil
|
||||
from pathlib import Path
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
RESULT_NAME = "result.pdf"
|
||||
|
||||
|
||||
class Storage:
|
||||
def __init__(self, root: str) -> None:
|
||||
self.root = Path(root)
|
||||
self.root.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
def job_dir(self, job_id: str) -> Path:
|
||||
path = self.root / job_id
|
||||
path.mkdir(parents=True, exist_ok=True)
|
||||
return path
|
||||
|
||||
def result_path(self, job_id: str) -> Path:
|
||||
return self.root / job_id / RESULT_NAME
|
||||
|
||||
def publish(self, job_id: str, produced: Path) -> Path:
|
||||
"""Move the produced file to its final name and drop the intermediates."""
|
||||
target = self.result_path(job_id)
|
||||
if produced != target:
|
||||
produced.replace(target)
|
||||
|
||||
for item in self.job_dir(job_id).iterdir():
|
||||
if item.name == RESULT_NAME:
|
||||
continue
|
||||
self._remove(item)
|
||||
|
||||
return target
|
||||
|
||||
def discard(self, job_id: str) -> None:
|
||||
self._remove(self.root / job_id)
|
||||
|
||||
def _remove(self, path: Path) -> None:
|
||||
try:
|
||||
if path.is_dir():
|
||||
shutil.rmtree(path, ignore_errors=False)
|
||||
elif path.exists():
|
||||
path.unlink()
|
||||
except OSError as exc:
|
||||
logger.warning("Could not remove temporary path", extra={"path": str(path)}, exc_info=exc)
|
||||
|
||||
def sweep_orphans(self, known_job_ids: set[str]) -> int:
|
||||
"""Remove directories that belong to no known job, for example after a restart."""
|
||||
removed = 0
|
||||
for item in self.root.iterdir():
|
||||
if item.is_dir() and item.name not in known_job_ids:
|
||||
self._remove(item)
|
||||
removed += 1
|
||||
if removed:
|
||||
logger.info("Removed orphaned job directories", extra={"count": removed})
|
||||
return removed
|
||||
Reference in New Issue
Block a user