Implementace prevodu HTML na PDF

Sluzba prijme adresu HTML dokumentu nebo HTML v tele requestu a vrati PDF.
Navrzena pro dokumenty o stovkach az tisicich stranek.

Rendering:
- WeasyPrint jako vychozi engine, spravne CSS Paged Media, nizka pametova
  narocnost, bez JavaScriptu
- Chromium pres Playwright pro dokumenty dokreslovane skripty
- rezim auto s detekci skriptu a fallbackem pri selhani WeasyPrintu

Velke dokumenty:
- deleni na casti na strukturalnich hranicich, rez nikdy uvnitr tabulky
  nebo odstavce
- dvoupruchodovy render obsahu se skutecnymi cisly stranek, pozice nadpisu
  se ctou z kotev hlasenych u kazde stranky
- cislovani stranek bud pres CSS countery, nebo pres cislovaci vrstvu
  nastampovanou na hotove PDF, rozmer stranky se cte z vysledneho souboru
- Chromium se restartuje po N jobech, nikdy vsak behem beziciho renderu

API:
- POST /convert synchronne, POST /jobs asynchronne se sledovanim stavu,
  stahovanim vysledku, rusenim a volitelnym callbackem
- GET /health s overenim dostupnosti obou enginu a stavem fronty
- OpenAPI respektuje prefix reverse proxy pres root_path

Bezpecnost a provoz:
- SSRF kontrola po DNS resolvu, na kazdem presmerovani a u vsech pozadavku
  prohlizece
- nedostupne assety render nezastavi, ale hlasi se v odpovedi i v logu
- fronta s omezenym poctem workeru, rozpracovane joby se pri ukonceni
  oznaci jako failed, nezmizi potichu
- strukturovane JSON logovani s job_id
- vsechny limity vypnute ve vychozim stavu

Dockerfile je dvoufazovy, obsahuje zavislosti WeasyPrintu, Chromium
a fonty s ceskou diakritikou.

Autentizace zamerne neni implementovana, zpusob predavani neni domluveny.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
JiriUhlir
2026-08-27 14:50:10 +02:00
co-authored by Claude Opus 5
parent e3cc8f418b
commit 156289fe2d
48 changed files with 4043 additions and 24 deletions
+1
View File
@@ -0,0 +1 @@
"""html-to-pdf service package."""
+79
View File
@@ -0,0 +1,79 @@
"""Runtime configuration read from environment variables.
AppFactory injects variables through the generated runtime .env file, so every
option below has a safe default and the service starts without any variable set.
"""
import os
from dataclasses import dataclass, field
from functools import lru_cache
def _bool(name: str, default: bool) -> bool:
raw = os.getenv(name)
if raw is None or raw.strip() == "":
return default
return raw.strip().lower() in ("1", "true", "yes", "on")
def _int(name: str, default: int) -> int:
raw = os.getenv(name)
if raw is None or raw.strip() == "":
return default
try:
return int(raw)
except ValueError:
return default
def _list(name: str) -> list[str]:
raw = os.getenv(name, "")
return [item.strip() for item in raw.split(",") if item.strip()]
@dataclass(frozen=True)
class Settings:
app_name: str = field(default_factory=lambda: os.getenv("APP_NAME", "html-to-pdf"))
app_version: str = field(default_factory=lambda: os.getenv("APP_VERSION", "1.0.0"))
root_path: str = field(
default_factory=lambda: os.getenv("ROOT_PATH") or os.getenv("BASE_PATH") or ""
)
log_level: str = field(default_factory=lambda: os.getenv("LOG_LEVEL", "INFO").upper())
# Job queue
workers: int = field(default_factory=lambda: _int("WORKERS", 2))
queue_max_size: int = field(default_factory=lambda: _int("QUEUE_MAX_SIZE", 100))
sync_timeout_seconds: int = field(default_factory=lambda: _int("SYNC_TIMEOUT_SECONDS", 60))
job_result_ttl_seconds: int = field(default_factory=lambda: _int("JOB_RESULT_TTL_SECONDS", 3600))
storage_dir: str = field(default_factory=lambda: os.getenv("STORAGE_DIR", "/tmp/html-to-pdf"))
# Engines
default_engine: str = field(default_factory=lambda: os.getenv("DEFAULT_ENGINE", "auto"))
chromium_enabled: bool = field(default_factory=lambda: _bool("CHROMIUM_ENABLED", True))
chromium_restart_after_jobs: int = field(
default_factory=lambda: _int("CHROMIUM_RESTART_AFTER_JOBS", 50)
)
# Network
fetch_timeout_seconds: int = field(default_factory=lambda: _int("FETCH_TIMEOUT_SECONDS", 30))
asset_timeout_seconds: int = field(default_factory=lambda: _int("ASSET_TIMEOUT_SECONDS", 10))
max_redirects: int = field(default_factory=lambda: _int("MAX_REDIRECTS", 5))
# SSRF protection. Blocking is on by default and can be widened or narrowed.
ssrf_block_private: bool = field(default_factory=lambda: _bool("SSRF_BLOCK_PRIVATE", True))
ssrf_extra_blocked_cidrs: list[str] = field(default_factory=lambda: _list("SSRF_EXTRA_BLOCKED_CIDRS"))
ssrf_allowed_hosts: list[str] = field(default_factory=lambda: _list("SSRF_ALLOWED_HOSTS"))
# Optional limits. Zero means no limit, which is the default on purpose.
max_pages: int = field(default_factory=lambda: _int("MAX_PAGES", 0))
max_html_bytes: int = field(default_factory=lambda: _int("MAX_HTML_BYTES", 0))
max_render_seconds: int = field(default_factory=lambda: _int("MAX_RENDER_SECONDS", 0))
# Callback delivery
callback_timeout_seconds: int = field(default_factory=lambda: _int("CALLBACK_TIMEOUT_SECONDS", 15))
callback_retries: int = field(default_factory=lambda: _int("CALLBACK_RETRIES", 3))
@lru_cache(maxsize=1)
def get_settings() -> Settings:
return Settings()
+45
View File
@@ -0,0 +1,45 @@
"""Application container.
Built once during startup and read by the routers. Keeping it here avoids
importing the FastAPI app from the routers.
"""
from __future__ import annotations
from dataclasses import dataclass
from .config import Settings
from .services.jobs import JobManager
from .services.pipeline import ConversionPipeline
from .services.storage import Storage
@dataclass
class Container:
settings: Settings
storage: Storage
pipeline: ConversionPipeline
jobs: JobManager
engines: dict
_container: Container | None = None
def set_container(container: Container) -> None:
global _container
_container = container
def get_container() -> Container:
if _container is None:
raise RuntimeError("Aplikace jeste nebyla inicializovana.")
return _container
def get_jobs() -> JobManager:
return get_container().jobs
def get_settings_dep() -> Settings:
return get_container().settings
View File
+48
View File
@@ -0,0 +1,48 @@
"""Common interface of the render engines."""
from __future__ import annotations
from abc import ABC, abstractmethod
from dataclasses import dataclass, field
from pathlib import Path
from ..models import ConvertRequest
@dataclass
class ChunkRender:
"""Result of rendering a single chunk."""
path: Path
page_count: int
# Anchor name to zero based page index inside this chunk.
anchor_pages: dict[str, int] = field(default_factory=dict)
class RenderEngine(ABC):
name: str
@abstractmethod
async def available(self) -> bool:
"""True when the engine can render right now."""
@abstractmethod
async def render_chunk(
self,
html: str,
base_url: str | None,
request: ConvertRequest,
output_path: Path,
total_pages: int | None = None,
asset_gate=None,
) -> ChunkRender:
"""Render one chunk into output_path."""
async def shutdown(self) -> None:
"""Release engine resources. Default is a no-op."""
return None
@property
def supports_anchor_pages(self) -> bool:
"""True when the engine can report on which page an anchor landed."""
return False
+211
View File
@@ -0,0 +1,211 @@
"""Headless Chromium engine driven by Playwright.
Used for foreign URLs and for documents that are completed by JavaScript.
Chromium has no usable CSS page counters, so page numbering for this engine is
always done by the overlay in the post processing step.
The browser leaks memory over time, therefore the whole instance is restarted
after a configurable number of jobs.
"""
from __future__ import annotations
import asyncio
import logging
from pathlib import Path
from ..config import get_settings
from ..errors import EngineUnavailableError, RenderTimeoutError
from ..models import ConvertRequest
from ..pdf.styles import NAMED_SIZE
from .base import ChunkRender, RenderEngine
logger = logging.getLogger(__name__)
LAUNCH_ARGS = [
"--no-sandbox",
"--disable-dev-shm-usage",
"--disable-gpu",
"--hide-scrollbars",
]
class ChromiumEngine(RenderEngine):
name = "chromium"
def __init__(self) -> None:
self._settings = get_settings()
self._lock = asyncio.Lock()
self._playwright = None
self._browser = None
self._jobs_since_launch = 0
self._active_renders = 0
async def available(self) -> bool:
if not self._settings.chromium_enabled:
return False
try:
await self._ensure_browser(reserve=False)
except Exception as exc: # noqa: BLE001 - reported, never silent
logger.error("Chromium is not available", exc_info=exc)
return False
return True
async def _ensure_browser(self, reserve: bool = True):
"""Return a running browser, restarting it when that is safe.
The restart must never happen while a render is in flight, otherwise the
health check could close the browser under a running job.
"""
async with self._lock:
due_for_restart = self._jobs_since_launch >= self._settings.chromium_restart_after_jobs
if self._browser is not None and self._active_renders == 0 and due_for_restart:
logger.info(
"Restarting Chromium to release memory",
extra={"jobs_since_launch": self._jobs_since_launch},
)
await self._close_browser()
if self._browser is None:
try:
from playwright.async_api import async_playwright
except ImportError as exc:
raise EngineUnavailableError(
"Engine chromium neni k dispozici, chybi balicek playwright.",
) from exc
self._playwright = await async_playwright().start()
self._browser = await self._playwright.chromium.launch(args=LAUNCH_ARGS)
self._jobs_since_launch = 0
logger.info("Chromium launched")
if reserve:
self._active_renders += 1
return self._browser
async def _close_browser(self) -> None:
if self._browser is not None:
try:
await self._browser.close()
except Exception as exc: # noqa: BLE001 - reported, never silent
logger.warning("Closing Chromium failed", exc_info=exc)
self._browser = None
if self._playwright is not None:
try:
await self._playwright.stop()
except Exception as exc: # noqa: BLE001 - reported, never silent
logger.warning("Stopping Playwright failed", exc_info=exc)
self._playwright = None
async def shutdown(self) -> None:
async with self._lock:
await self._close_browser()
def on_job_finished(self) -> None:
self._jobs_since_launch += 1
async def render_chunk(
self,
html: str | None,
base_url: str | None,
request: ConvertRequest,
output_path: Path,
total_pages: int | None = None,
asset_gate=None,
) -> ChunkRender:
browser = await self._ensure_browser()
try:
context = await browser.new_context()
except Exception:
self._active_renders -= 1
raise
try:
if asset_gate is not None:
await context.route("**/*", _make_route_handler(asset_gate))
page = await context.new_page()
timeout_ms = request.wait_for.timeout_seconds * 1000
try:
if html is None:
if not base_url:
raise EngineUnavailableError("Chybi adresa dokumentu pro engine chromium.")
await page.goto(base_url, wait_until=request.wait_for.state, timeout=timeout_ms)
else:
await page.set_content(html, wait_until=request.wait_for.state, timeout=timeout_ms)
if request.wait_for.selector:
await page.wait_for_selector(request.wait_for.selector, timeout=timeout_ms)
except Exception as exc: # noqa: BLE001 - converted to a typed error
if "Timeout" in type(exc).__name__ or "timeout" in str(exc).lower():
raise RenderTimeoutError(
"Chromium nestihl nacist dokument v zadanem casovem limitu.",
{"timeout_seconds": request.wait_for.timeout_seconds},
) from exc
raise
await page.emulate_media(media="print")
pdf_kwargs = _pdf_kwargs(request)
try:
pdf_bytes = await page.pdf(outline=request.outline, **pdf_kwargs)
except TypeError:
logger.warning("Installed Playwright does not support the outline option, continuing without bookmarks")
pdf_bytes = await page.pdf(**pdf_kwargs)
output_path.write_bytes(pdf_bytes)
finally:
self._active_renders -= 1
try:
await context.close()
except Exception as exc:
logger.warning("Closing the browser context failed", exc_info=exc)
return ChunkRender(path=output_path, page_count=_count_pages(output_path))
def _pdf_kwargs(request: ConvertRequest) -> dict:
margin = request.page.margin
kwargs: dict = {
"print_background": True,
"prefer_css_page_size": True,
"margin": {
"top": margin.top,
"right": margin.right,
"bottom": margin.bottom,
"left": margin.left,
},
}
fmt = request.page.format.strip()
if NAMED_SIZE.match(fmt):
kwargs["format"] = fmt
else:
width, _, height = fmt.partition(" ")
if width and height:
kwargs["width"] = width.strip()
kwargs["height"] = height.strip()
else:
kwargs["format"] = fmt
kwargs["landscape"] = request.page.orientation == "landscape"
return kwargs
def _make_route_handler(asset_gate):
async def handler(route, playwright_request):
allowed, reason = asset_gate.allowed(playwright_request.url)
if allowed:
await route.continue_()
return
asset_gate.report.add(playwright_request.url, reason)
await route.abort()
return handler
def _count_pages(path: Path) -> int:
from pypdf import PdfReader
with path.open("rb") as handle:
return len(PdfReader(handle).pages)
+107
View File
@@ -0,0 +1,107 @@
"""WeasyPrint engine.
Default engine for documents we generate ourselves. It implements CSS Paged
Media properly, which is what makes counters, running headers and repeated table
headers work, and it uses far less memory than a headless browser.
It does not execute JavaScript. That is intentional.
"""
from __future__ import annotations
import asyncio
import logging
from pathlib import Path
from ..models import ConvertRequest
from ..pdf.styles import build_page_css
from .base import ChunkRender, RenderEngine
logger = logging.getLogger(__name__)
class WeasyPrintEngine(RenderEngine):
name = "weasyprint"
@property
def supports_anchor_pages(self) -> bool:
return True
async def available(self) -> bool:
try:
import weasyprint # noqa: F401
except Exception as exc: # noqa: BLE001 - reported, never silent
logger.error("WeasyPrint is not importable", exc_info=exc)
return False
return True
async def render_chunk(
self,
html: str,
base_url: str | None,
request: ConvertRequest,
output_path: Path,
total_pages: int | None = None,
asset_gate=None,
) -> ChunkRender:
return await asyncio.to_thread(
self._render_blocking, html, base_url, request, output_path, total_pages, asset_gate
)
def _render_blocking(
self,
html: str,
base_url: str | None,
request: ConvertRequest,
output_path: Path,
total_pages: int | None,
asset_gate,
) -> ChunkRender:
from weasyprint import CSS, HTML
kwargs = {"string": html, "base_url": base_url}
if asset_gate is not None:
kwargs["url_fetcher"] = asset_gate.weasy_fetcher()
stylesheets = []
if total_pages is not None:
# Only used when the page total is known up front, which happens on
# the second pass of an unchunked document.
stylesheets.append(
CSS(
string=build_page_css(
request.page,
request.page_numbers,
total_pages=total_pages,
outline=request.outline,
)
)
)
document = HTML(**kwargs).render(stylesheets=stylesheets or None)
write_kwargs = {}
if request.pdf_profile:
write_kwargs["pdf_variant"] = request.pdf_profile
document.write_pdf(target=str(output_path), **write_kwargs)
return ChunkRender(
path=output_path,
page_count=len(document.pages),
anchor_pages=self._anchor_pages(document),
)
@staticmethod
def _anchor_pages(document) -> dict[str, int]:
"""Map anchor name to the zero based page index it landed on."""
anchors: dict[str, int] = {}
for index, page in enumerate(document.pages):
page_anchors = getattr(page, "anchors", None)
if not page_anchors:
continue
for name in page_anchors:
anchors.setdefault(name, index)
if not anchors:
logger.warning("WeasyPrint returned no anchors, table of contents page numbers may be missing")
return anchors
+103
View File
@@ -0,0 +1,103 @@
"""Application errors.
Every failure surfaces a machine readable error_code and a message in Czech,
because the message is shown to the caller.
"""
from __future__ import annotations
class ConversionError(Exception):
"""Base class for every error the conversion pipeline can raise."""
error_code = "internal_error"
status_code = 500
def __init__(self, message: str, detail: dict | None = None) -> None:
super().__init__(message)
self.message = message
self.detail = detail or {}
def to_dict(self) -> dict:
payload = {"error_code": self.error_code, "message": self.message}
if self.detail:
payload["detail"] = self.detail
return payload
class InvalidRequestError(ConversionError):
error_code = "invalid_request"
status_code = 400
class BlockedTargetError(ConversionError):
"""The requested URL points somewhere the service refuses to reach."""
error_code = "blocked_target"
status_code = 400
class SourceUnavailableError(ConversionError):
error_code = "source_unavailable"
status_code = 502
class RenderTimeoutError(ConversionError):
error_code = "render_timeout"
status_code = 504
class SyncTooLongError(ConversionError):
"""Synchronous conversion exceeded its budget, caller should use /jobs."""
error_code = "sync_too_long"
status_code = 413
class EngineUnavailableError(ConversionError):
error_code = "engine_unavailable"
status_code = 503
class UnsupportedCombinationError(ConversionError):
error_code = "unsupported_combination"
status_code = 400
class LimitExceededError(ConversionError):
error_code = "limit_exceeded"
status_code = 400
class JobNotFoundError(ConversionError):
error_code = "job_not_found"
status_code = 404
class QueueFullError(ConversionError):
error_code = "queue_full"
status_code = 503
ERROR_CLASSES = (
InvalidRequestError,
BlockedTargetError,
SourceUnavailableError,
RenderTimeoutError,
SyncTooLongError,
EngineUnavailableError,
UnsupportedCombinationError,
LimitExceededError,
JobNotFoundError,
QueueFullError,
)
STATUS_BY_CODE = {cls.error_code: cls.status_code for cls in ERROR_CLASSES}
def error_from_code(error_code: str, message: str, detail: dict | None = None) -> ConversionError:
"""Rebuild a typed error from a stored job error."""
error = ConversionError(message, detail)
error.error_code = error_code
error.status_code = STATUS_BY_CODE.get(error_code, 500)
return error
+58
View File
@@ -0,0 +1,58 @@
"""Structured JSON logging.
Every log record is a single JSON line. Anything related to a job carries the
job_id so the whole conversion can be reconstructed from the log.
"""
import json
import logging
import sys
from contextvars import ContextVar
current_job_id: ContextVar[str | None] = ContextVar("current_job_id", default=None)
_RESERVED = {
"args", "asctime", "created", "exc_info", "exc_text", "filename", "funcName",
"levelname", "levelno", "lineno", "module", "msecs", "message", "msg", "name",
"pathname", "process", "processName", "relativeCreated", "stack_info",
"thread", "threadName", "taskName",
}
class JsonFormatter(logging.Formatter):
def format(self, record: logging.LogRecord) -> str:
payload: dict[str, object] = {
"ts": self.formatTime(record, "%Y-%m-%dT%H:%M:%S%z"),
"level": record.levelname,
"logger": record.name,
"message": record.getMessage(),
}
job_id = getattr(record, "job_id", None) or current_job_id.get()
if job_id:
payload["job_id"] = job_id
for key, value in record.__dict__.items():
if key not in _RESERVED and not key.startswith("_") and key != "job_id":
payload[key] = value
if record.exc_info:
payload["exception"] = self.formatException(record.exc_info)
return json.dumps(payload, ensure_ascii=False, default=str)
def setup_logging(level: str) -> None:
handler = logging.StreamHandler(sys.stdout)
handler.setFormatter(JsonFormatter())
root = logging.getLogger()
root.handlers.clear()
root.addHandler(handler)
root.setLevel(getattr(logging, level, logging.INFO))
# uvicorn keeps its own handlers, route them through ours as well
for name in ("uvicorn", "uvicorn.access", "uvicorn.error"):
logger = logging.getLogger(name)
logger.handlers.clear()
logger.propagate = True
+128 -19
View File
@@ -1,25 +1,134 @@
import os
from fastapi import FastAPI
"""Service entry point.
APP_NAME = os.getenv("APP_NAME", "html-to-pdf")
APP_VERSION = os.getenv("APP_VERSION", "1.0.0")
ROOT_PATH = os.getenv("ROOT_PATH", "")
The application runs behind the AppFactory reverse proxy under
/apps/<app-id>. The prefix is stripped before the request reaches the
container, so only the OpenAPI document has to know about it. That is what
root_path does.
"""
from __future__ import annotations
import logging
from contextlib import asynccontextmanager
from fastapi import FastAPI, Request
from fastapi.responses import JSONResponse
from .config import get_settings
from .deps import Container, set_container
from .engines.chromium import ChromiumEngine
from .engines.weasy import WeasyPrintEngine
from .errors import ConversionError
from .logging_setup import setup_logging
from .routers import convert as convert_router
from .routers import health as health_router
from .routers import jobs as jobs_router
from .services.jobs import JobManager
from .services.pipeline import ConversionPipeline
from .services.storage import Storage
logger = logging.getLogger(__name__)
DESCRIPTION = """
Sluzba prevadi HTML dokument na PDF. Prijme adresu dokumentu nebo HTML primo
v tele requestu a vrati soubor PDF.
Pro dokumenty o stovkach az tisicich stranek pouzijte asynchronni endpoint
POST /jobs. Synchronni POST /convert je urceny pro mensi dokumenty.
Dva render enginy:
- weasyprint je vychozi, ma spravne strankovani a nizkou pametovou narocnost,
nespousti JavaScript
- chromium zvladne i dokumenty dokreslovane JavaScriptem, ale nema pouzitelne
CSS countery, takze cisla stranek se dopisuji do hotoveho PDF
"""
@asynccontextmanager
async def lifespan(app: FastAPI):
settings = get_settings()
setup_logging(settings.log_level)
engines: dict = {}
weasy = WeasyPrintEngine()
if await weasy.available():
engines["weasyprint"] = weasy
else:
logger.error("WeasyPrint engine is unavailable, the service will rely on Chromium only")
chromium = ChromiumEngine()
if settings.chromium_enabled:
engines["chromium"] = chromium
else:
logger.info("Chromium engine is disabled by configuration")
if not engines:
logger.error("No render engine is available, conversion requests will fail")
storage = Storage(settings.storage_dir)
pipeline = ConversionPipeline(engines, settings)
manager = JobManager(pipeline, storage, settings)
set_container(
Container(
settings=settings,
storage=storage,
pipeline=pipeline,
jobs=manager,
engines=engines,
)
)
await manager.start()
logger.info(
"Service started",
extra={"engines": sorted(engines), "root_path": settings.root_path},
)
try:
yield
finally:
await manager.stop()
for engine in engines.values():
await engine.shutdown()
logger.info("Service stopped")
settings = get_settings()
setup_logging(settings.log_level)
app = FastAPI(
title=APP_NAME,
version=APP_VERSION,
root_path=ROOT_PATH
title=settings.app_name,
version=settings.app_version,
description=DESCRIPTION,
root_path=settings.root_path,
lifespan=lifespan,
)
@app.get("/health")
def health():
return {"status": "ok"}
@app.get("/version")
def version():
return {
"app": APP_NAME,
"version": APP_VERSION,
"language": "python",
"root_path": ROOT_PATH
}
@app.exception_handler(ConversionError)
async def conversion_error_handler(request: Request, exc: ConversionError) -> JSONResponse:
logger.warning(
"Request failed",
extra={"error_code": exc.error_code, "path": request.url.path, "status_code": exc.status_code},
)
return JSONResponse(status_code=exc.status_code, content=exc.to_dict())
@app.exception_handler(Exception)
async def unhandled_error_handler(request: Request, exc: Exception) -> JSONResponse:
logger.exception("Unhandled error", extra={"path": request.url.path})
return JSONResponse(
status_code=500,
content={
"error_code": "internal_error",
"message": "Doslo k neocekavane chybe sluzby.",
},
)
app.include_router(health_router.router)
app.include_router(convert_router.router)
app.include_router(jobs_router.router)
+170
View File
@@ -0,0 +1,170 @@
"""Request and response models.
Only `source` is mandatory. Everything else has a working default so that
{"source": {"url": "..."}} produces a usable PDF.
"""
from __future__ import annotations
from datetime import datetime
from typing import Literal
from pydantic import BaseModel, Field, model_validator
from .config import get_settings
EngineName = Literal["auto", "weasyprint", "chromium"]
PagePosition = Literal[
"top-left", "top-center", "top-right",
"bottom-left", "bottom-center", "bottom-right",
]
JobStatus = Literal["queued", "running", "done", "failed", "cancelled", "expired"]
class Source(BaseModel):
url: str | None = Field(default=None, description="Adresa HTML dokumentu ke konverzi.")
html: str | None = Field(default=None, description="HTML poslane primo v tele requestu.")
base_url: str | None = Field(
default=None,
description="Zaklad pro relativni cesty. Pouziva se hlavne spolu s polem html.",
)
@model_validator(mode="after")
def exactly_one_source(self) -> "Source":
if bool(self.url) == bool(self.html):
raise ValueError("Vyplnte prave jedno z poli source.url a source.html.")
return self
class Margin(BaseModel):
top: str = "20mm"
right: str = "15mm"
bottom: str = "20mm"
left: str = "15mm"
class PageSettings(BaseModel):
format: str = Field(default="A4", description="Nazev formatu (A4, A5, Letter) nebo rozmer 210mm 297mm.")
orientation: Literal["portrait", "landscape"] = "portrait"
margin: Margin = Field(default_factory=Margin)
class PageNumbers(BaseModel):
enabled: bool = False
format: str = Field(default="{page} / {pages}", description="Zastupne symboly {page} a {pages}.")
position: PagePosition = "bottom-center"
mode: Literal["auto", "css", "overlay"] = Field(
default="auto",
description=(
"auto zvoli css u necleneneho dokumentu a overlay u clenene nebo u Chromia. "
"css pouziva CSS countery, overlay dopisuje cisla do hotoveho PDF."
),
)
start_at: int = 1
class TocSettings(BaseModel):
enabled: bool = False
depth: int = Field(default=3, ge=1, le=6)
title: str = "Obsah"
class AssetSettings(BaseModel):
allow_remote: bool = True
timeout_seconds: int = Field(default_factory=lambda: get_settings().asset_timeout_seconds, ge=1)
class ChunkSettings(BaseModel):
enabled: bool = True
pages_per_chunk: int = Field(default=50, ge=1)
class WaitFor(BaseModel):
"""Chromium only. Ignored by the WeasyPrint engine."""
state: Literal["load", "domcontentloaded", "networkidle"] = "load"
selector: str | None = None
timeout_seconds: int = Field(default=30, ge=1)
class ConvertRequest(BaseModel):
source: Source
engine: EngineName = Field(default_factory=lambda: get_settings().default_engine) # type: ignore[arg-type]
page: PageSettings = Field(default_factory=PageSettings)
page_numbers: PageNumbers = Field(default_factory=PageNumbers)
toc: TocSettings = Field(default_factory=TocSettings)
outline: bool = Field(default=True, description="Generovat zalozky PDF z nadpisu h1 az h6.")
pdf_profile: Literal["pdf/a-1b", "pdf/a-2b", "pdf/a-3b", "pdf/a-4b", "pdf/ua-1"] | None = None
assets: AssetSettings = Field(default_factory=AssetSettings)
chunking: ChunkSettings = Field(default_factory=ChunkSettings)
wait_for: WaitFor = Field(default_factory=WaitFor)
filename: str | None = Field(default=None, description="Nazev souboru ve Content-Disposition.")
callback_url: str | None = Field(
default=None,
description="Volitelna adresa, na kterou se po dokonceni jobu posle POST se stavem jobu.",
)
model_config = {
"json_schema_extra": {
"examples": [
{"source": {"url": "https://example.com/dokument.html"}},
{
"source": {"url": "https://example.com/velky-dokument.html"},
"engine": "weasyprint",
"page": {"format": "A4", "orientation": "portrait"},
"page_numbers": {"enabled": True, "format": "{page} / {pages}"},
"toc": {"enabled": True, "depth": 3, "title": "Obsah"},
"chunking": {"enabled": True, "pages_per_chunk": 50},
},
]
}
}
class MissingAsset(BaseModel):
url: str
reason: str
class JobProgress(BaseModel):
pages_rendered: int = 0
chunks_done: int = 0
chunks_total: int = 0
pass_number: int = 0
class ErrorInfo(BaseModel):
error_code: str
message: str
detail: dict | None = None
class JobState(BaseModel):
job_id: str
status: JobStatus
created_at: datetime
started_at: datetime | None = None
finished_at: datetime | None = None
expires_at: datetime | None = None
progress: JobProgress = Field(default_factory=JobProgress)
engine_used: str | None = None
page_count: int | None = None
missing_assets: list[MissingAsset] = Field(default_factory=list)
warnings: list[str] = Field(default_factory=list)
error: ErrorInfo | None = None
result_url: str | None = None
class JobAccepted(BaseModel):
job_id: str
status: JobStatus
created_at: datetime
result_url: str
class HealthResponse(BaseModel):
status: Literal["ok", "degraded"]
app: str
version: str
engines: dict[str, bool]
queue: dict[str, int]
View File
+142
View File
@@ -0,0 +1,142 @@
"""Splitting of a large document into renderable chunks.
A thousand page document rendered in one pass keeps the whole page tree in
memory. Splitting it on structural boundaries keeps memory flat at the cost of
having to reassemble the page numbering afterwards.
Cuts are only ever made between direct children of the container element, so a
table or a paragraph is never torn in half.
"""
from __future__ import annotations
import logging
from lxml import html as lxml_html
from .document import SourceDocument
logger = logging.getLogger(__name__)
SPLIT_TAGS = {"section", "article", "h1"}
BREAK_KEYWORDS = ("page-break-before", "break-before")
# Rough page size heuristic. The real page count is only known after rendering,
# so this only has to be good enough to keep chunks roughly even.
CHARS_PER_PAGE = 2200
IMAGE_CHAR_WEIGHT = 900
TABLE_ROW_CHAR_WEIGHT = 120
def is_split_point(element) -> bool:
if not isinstance(element.tag, str):
return False
if element.tag.lower() in SPLIT_TAGS:
return True
if element.get("data-chunk") is not None:
return True
style = (element.get("style") or "").lower()
return any(keyword in style for keyword in BREAK_KEYWORDS)
def estimate_pages(element) -> float:
text_length = len(element.text_content())
text_length += IMAGE_CHAR_WEIGHT * len(element.findall(".//img"))
text_length += TABLE_ROW_CHAR_WEIGHT * len(element.findall(".//tr"))
return max(text_length / CHARS_PER_PAGE, 0.01)
def split_document(document: SourceDocument, pages_per_chunk: int) -> list[str]:
"""Return one complete HTML document per chunk.
Falls back to a single chunk when the document has no usable split points.
"""
container = document.container
children = [child for child in container if isinstance(child.tag, str)]
split_indexes = [index for index, child in enumerate(children) if is_split_point(child)]
if len(split_indexes) < 2:
logger.info(
"Document has no usable split points, rendering in one pass",
extra={"split_points": len(split_indexes), "children": len(children)},
)
return [document.to_html()]
groups = _group_children(children, set(split_indexes), pages_per_chunk)
if len(groups) < 2:
logger.info("Document fits into a single chunk", extra={"children": len(children)})
return [document.to_html()]
prefix, suffix = _skeleton(document, container)
child_html = [lxml_html.tostring(child, encoding="unicode") for child in children]
chunks = ["".join((prefix, *(child_html[index] for index in group), suffix)) for group in groups]
logger.info(
"Document split into chunks",
extra={"chunks": len(chunks), "children": len(children), "pages_per_chunk": pages_per_chunk},
)
return chunks
def _group_children(children, split_indexes: set[int], pages_per_chunk: int) -> list[list[int]]:
groups: list[list[int]] = []
current: list[int] = []
current_pages = 0.0
for index, child in enumerate(children):
starts_chunk = index in split_indexes and current and current_pages >= pages_per_chunk
if starts_chunk:
groups.append(current)
current = []
current_pages = 0.0
current.append(index)
current_pages += estimate_pages(child)
if current:
groups.append(current)
return groups
def _skeleton(document: SourceDocument, container) -> tuple[str, str]:
"""Opening and closing markup shared by every chunk.
The whole ancestor chain is recreated with its attributes so CSS selectors
that depend on it keep matching inside a chunk.
"""
head_html = lxml_html.tostring(document.head, encoding="unicode")
chain = []
node = container
while node is not None and node is not document.tree:
chain.append(node)
node = node.getparent()
chain.reverse()
opens = [_open_tag(document.tree)]
opens.append(head_html)
closes = ["</html>"]
for node in chain:
opens.append(_open_tag(node))
closes.append(f"</{node.tag}>")
closes.reverse()
return "<!DOCTYPE html>\n" + "".join(opens), "".join(closes)
def _open_tag(element) -> str:
attributes = "".join(
f' {name}="{_escape(value)}"' for name, value in element.attrib.items()
)
return f"<{element.tag}{attributes}>"
def _escape(value: str) -> str:
return (
value.replace("&", "&amp;")
.replace('"', "&quot;")
.replace("<", "&lt;")
.replace(">", "&gt;")
)
+186
View File
@@ -0,0 +1,186 @@
"""Parsing of the source document, heading extraction and table of contents."""
from __future__ import annotations
import logging
import re
from dataclasses import dataclass
from lxml import html as lxml_html
logger = logging.getLogger(__name__)
HEADING_TAGS = ("h1", "h2", "h3", "h4", "h5", "h6")
ID_SAFE = re.compile(r"[^a-zA-Z0-9_-]+")
TOC_PLACEHOLDER = "\u2007\u2007\u2007" # figure spaces, keeps the pass 1 layout stable
@dataclass
class Heading:
level: int
text: str
anchor: str
page: int | None = None
class SourceDocument:
"""Thin wrapper over the parsed document with the operations we need."""
def __init__(self, html: str) -> None:
self.tree = lxml_html.document_fromstring(html)
self.head = self.tree.find("head")
if self.head is None:
self.head = lxml_html.Element("head")
self.tree.insert(0, self.head)
self.body = self.tree.find("body")
if self.body is None:
raise ValueError("Dokument neobsahuje element body.")
self.container = self._find_container()
self._toc_css_added = False
def _find_container(self):
"""Element whose children are the natural split points.
Many documents wrap everything in a single <main> or <div>. Splitting the
body would then produce a single chunk, so we descend through up to two
single child wrappers.
"""
container = self.body
for _ in range(2):
children = [child for child in container if isinstance(child.tag, str)]
if len(children) == 1 and len(list(children[0])) > 1:
container = children[0]
continue
break
return container
# -- styles ---------------------------------------------------------
def append_stylesheet(self, css: str) -> None:
"""Append a stylesheet as the last element of head so it wins on ties."""
style = lxml_html.Element("style")
style.set("type", "text/css")
style.text = css
self.head.append(style)
def set_base_url(self, base_url: str | None) -> None:
if not base_url or self.head.find("base") is not None:
return
base = lxml_html.Element("base")
base.set("href", base_url)
self.head.insert(0, base)
# -- headings -------------------------------------------------------
def collect_headings(self, max_depth: int) -> list[Heading]:
"""Assign ids to headings that lack one and return them in document order."""
headings: list[Heading] = []
used: set[str] = {el.get("id") for el in self.tree.iter() if el.get("id")}
for element in self.body.iter(*HEADING_TAGS):
level = int(element.tag[1])
if level > max_depth:
continue
text = " ".join(element.text_content().split())
if not text:
continue
anchor = element.get("id")
if not anchor:
anchor = self._unique_anchor(text, used)
element.set("id", anchor)
used.add(anchor)
headings.append(Heading(level=level, text=text, anchor=anchor))
return headings
@staticmethod
def _unique_anchor(text: str, used: set[str]) -> str:
base = ID_SAFE.sub("-", text.strip().lower()).strip("-") or "nadpis"
base = f"htp-{base[:60]}"
candidate = base
counter = 2
while candidate in used:
candidate = f"{base}-{counter}"
counter += 1
return candidate
# -- table of contents ----------------------------------------------
def insert_toc(self, headings: list[Heading], title: str, pages: dict[str, int] | None) -> None:
"""Insert the table of contents as the first block of the body.
Pass 1 uses a fixed width placeholder instead of the page number so the
table keeps exactly the same layout in pass 2.
"""
container = lxml_html.Element("nav")
container.set("id", "htp-toc")
container.set("class", "htp-toc")
heading = lxml_html.Element("h1")
heading.set("class", "htp-toc-title")
heading.text = title
container.append(heading)
table = lxml_html.Element("table")
table.set("class", "htp-toc-table")
tbody = lxml_html.Element("tbody")
for item in headings:
if item.anchor == "htp-toc-title":
continue
row = lxml_html.Element("tr")
row.set("class", f"htp-toc-level-{item.level}")
label_cell = lxml_html.Element("td")
label_cell.set("class", "htp-toc-label")
link = lxml_html.Element("a")
link.set("href", f"#{item.anchor}")
link.text = item.text
label_cell.append(link)
page_cell = lxml_html.Element("td")
page_cell.set("class", "htp-toc-page")
if pages is None:
page_cell.text = TOC_PLACEHOLDER
else:
page_cell.text = str(pages.get(item.anchor, "")) or TOC_PLACEHOLDER
row.append(label_cell)
row.append(page_cell)
tbody.append(row)
table.append(tbody)
container.append(table)
spacer = lxml_html.Element("div")
spacer.set("class", "htp-toc-break")
container.append(spacer)
self.container.insert(0, container)
if not self._toc_css_added:
self.append_stylesheet(TOC_CSS)
self._toc_css_added = True
def remove_toc(self) -> None:
existing = self.tree.find(".//nav[@id='htp-toc']")
if existing is not None:
existing.getparent().remove(existing)
# -- serialization ---------------------------------------------------
def to_html(self) -> str:
return "<!DOCTYPE html>\n" + lxml_html.tostring(self.tree, encoding="unicode")
TOC_CSS = """
.htp-toc { break-after: page; }
.htp-toc-table { width: 100%; border-collapse: collapse; }
.htp-toc-table td { padding: 2pt 0; vertical-align: bottom; }
.htp-toc-label a { text-decoration: none; color: inherit; }
.htp-toc-page { text-align: right; width: 4em; font-variant-numeric: tabular-nums; white-space: nowrap; }
.htp-toc-level-2 .htp-toc-label { padding-left: 1.2em; }
.htp-toc-level-3 .htp-toc-label { padding-left: 2.4em; }
.htp-toc-level-4 .htp-toc-label { padding-left: 3.6em; }
.htp-toc-level-5 .htp-toc-label { padding-left: 4.8em; }
.htp-toc-level-6 .htp-toc-label { padding-left: 6em; }
"""
+53
View File
@@ -0,0 +1,53 @@
"""Merging of chunk PDFs and reading of basic page geometry."""
from __future__ import annotations
import logging
from pathlib import Path
logger = logging.getLogger(__name__)
def merge(paths: list[Path], output_path: Path) -> int:
"""Concatenate chunk PDFs into one file.
PdfWriter.append is used on purpose, it carries over bookmarks and internal
links and shifts their page references by the running offset.
"""
from pypdf import PdfWriter
if not paths:
raise ValueError("Neni co slucovat, seznam PDF je prazdny.")
if len(paths) == 1:
paths[0].replace(output_path)
return page_count(output_path)
writer = PdfWriter()
try:
for path in paths:
writer.append(str(path))
with output_path.open("wb") as handle:
writer.write(handle)
finally:
writer.close()
total = page_count(output_path)
logger.info("Chunks merged", extra={"chunks": len(paths), "pages": total})
return total
def page_count(path: Path) -> int:
from pypdf import PdfReader
with path.open("rb") as handle:
return len(PdfReader(handle).pages)
def first_page_size(path: Path) -> tuple[float, float]:
"""Width and height of the first page in points."""
from pypdf import PdfReader
with path.open("rb") as handle:
box = PdfReader(handle).pages[0].mediabox
return float(box.width), float(box.height)
+80
View File
@@ -0,0 +1,80 @@
"""Page numbering of the merged document.
Two ways to number pages:
css
CSS counters do the work during the render. Correct and cheap, but it only
works when the whole document is rendered in one pass, because each chunk
restarts the page counter.
overlay
A transparent numbering layer with the same page size is rendered once and
merged onto the finished PDF. This is the only option for a chunked document
and for Chromium, which has no usable page counters.
"""
from __future__ import annotations
import logging
from pathlib import Path
from ..errors import EngineUnavailableError
from ..models import PageNumbers, PageSettings
from .styles import build_overlay_css
logger = logging.getLogger(__name__)
def build_overlay(
total_pages: int,
width_pt: float,
height_pt: float,
page: PageSettings,
page_numbers: PageNumbers,
output_path: Path,
) -> Path:
"""Render the numbering layer, one empty page per page of the document."""
try:
from weasyprint import HTML
except ImportError as exc:
raise EngineUnavailableError(
"Cislovani stranek vyzaduje nainstalovany WeasyPrint, ktery kresli cislovaci vrstvu.",
) from exc
css = build_overlay_css(width_pt, height_pt, page, page_numbers, total_pages)
slots = '<div class="pdf-page-slot"></div>' * total_pages
html = (
"<!DOCTYPE html><html><head><meta charset=\"utf-8\">"
f"<style>{css}</style></head><body>{slots}</body></html>"
)
HTML(string=html).write_pdf(target=str(output_path))
logger.info("Numbering overlay rendered", extra={"pages": total_pages})
return output_path
def apply_overlay(document_path: Path, overlay_path: Path, output_path: Path) -> None:
"""Stamp the numbering layer onto every page of the document."""
from pypdf import PdfReader, PdfWriter
writer = PdfWriter(clone_from=str(document_path))
try:
with overlay_path.open("rb") as handle:
overlay = PdfReader(handle)
available = len(overlay.pages)
if available < len(writer.pages):
logger.warning(
"Numbering overlay has fewer pages than the document, tail will stay unnumbered",
extra={"overlay_pages": available, "document_pages": len(writer.pages)},
)
for index, page in enumerate(writer.pages):
if index >= available:
break
page.merge_page(overlay.pages[index])
with output_path.open("wb") as target:
writer.write(target)
finally:
writer.close()
+115
View File
@@ -0,0 +1,115 @@
"""Generation of the page stylesheet injected into the source document."""
from __future__ import annotations
import re
from ..models import PageNumbers, PageSettings
NAMED_SIZE = re.compile(r"^[A-Za-z][A-Za-z0-9]*$")
MARGIN_BOXES = {
"top-left": "@top-left",
"top-center": "@top-center",
"top-right": "@top-right",
"bottom-left": "@bottom-left",
"bottom-center": "@bottom-center",
"bottom-right": "@bottom-right",
}
PLACEHOLDER = re.compile(r"(\{page\}|\{pages\})")
def page_size_value(page: PageSettings) -> str:
fmt = page.format.strip()
if NAMED_SIZE.match(fmt):
return f"{fmt} {page.orientation}"
# Explicit dimensions already carry the orientation.
return fmt
def margin_shorthand(page: PageSettings) -> str:
m = page.margin
return f"{m.top} {m.right} {m.bottom} {m.left}"
def css_content_value(fmt: str, total_pages: int | None) -> str:
"""Turn "{page} / {pages}" into a CSS content value.
When total_pages is known the total is written as a literal, otherwise the
CSS counter(pages) is used.
"""
parts: list[str] = []
for token in PLACEHOLDER.split(fmt):
if token == "{page}":
parts.append("counter(page)")
elif token == "{pages}":
parts.append(str(total_pages) if total_pages is not None else "counter(pages)")
elif token:
escaped = token.replace("\\", "\\\\").replace('"', '\\"')
parts.append(f'"{escaped}"')
return " ".join(parts) if parts else '""'
def build_page_css(
page: PageSettings,
page_numbers: PageNumbers | None = None,
total_pages: int | None = None,
outline: bool = True,
) -> str:
"""Stylesheet applied on top of the document styles."""
rules = [
"@page {",
f" size: {page_size_value(page)};",
f" margin: {margin_shorthand(page)};",
]
if page_numbers is not None and page_numbers.enabled:
box = MARGIN_BOXES[page_numbers.position]
rules.append(f" {box} {{")
rules.append(f" content: {css_content_value(page_numbers.format, total_pages)};")
rules.append(" font-size: 9pt;")
rules.append(" color: #444;")
rules.append(" }")
rules.append("}")
if not outline:
rules.append("h1, h2, h3, h4, h5, h6 { bookmark-level: none; }")
return "\n".join(rules)
def build_overlay_css(
width_pt: float,
height_pt: float,
page: PageSettings,
page_numbers: PageNumbers,
total_pages: int,
) -> str:
"""Stylesheet for the transparent numbering layer merged onto the final PDF.
The size comes from the produced PDF itself, so the overlay always matches
even when the source document declares its own @page size.
"""
box = MARGIN_BOXES[page_numbers.position]
reset = ""
if page_numbers.start_at != 1:
reset = f"body {{ counter-reset: page {page_numbers.start_at - 1}; }}\n"
return (
f"@page {{\n"
f" size: {width_pt:.2f}pt {height_pt:.2f}pt;\n"
f" margin: {margin_shorthand(page)};\n"
f" {box} {{\n"
f" content: {css_content_value(page_numbers.format, total_pages)};\n"
f" font-size: 9pt;\n"
f" color: #444;\n"
f" }}\n"
f"}}\n"
f"{reset}"
f"body {{ margin: 0; }}\n"
f".pdf-page-slot {{ height: 1px; break-after: page; }}\n"
f".pdf-page-slot:last-child {{ break-after: auto; }}\n"
)
View File
+78
View File
@@ -0,0 +1,78 @@
"""Synchronous conversion.
Suitable for smaller documents. Anything that does not finish within
SYNC_TIMEOUT_SECONDS is cancelled and the caller is pointed at /jobs.
"""
from __future__ import annotations
import asyncio
import logging
from fastapi import APIRouter, Response
from fastapi.responses import FileResponse
from starlette.background import BackgroundTask
from ..deps import get_container
from ..errors import ConversionError, SyncTooLongError, error_from_code
from ..models import ConvertRequest
logger = logging.getLogger(__name__)
router = APIRouter(tags=["konverze"])
@router.post(
"/convert",
summary="Synchronni prevod HTML na PDF",
response_class=Response,
responses={
200: {"content": {"application/pdf": {}}, "description": "Hotove PDF."},
413: {"description": "Konverze trvala dele nez limit, pouzijte POST /jobs."},
},
)
async def convert(request: ConvertRequest):
container = get_container()
settings = container.settings
manager = container.jobs
job = manager.submit(request)
try:
await asyncio.wait_for(job.done.wait(), timeout=settings.sync_timeout_seconds)
except asyncio.TimeoutError as exc:
manager.cancel(job.id)
logger.warning(
"Synchronous conversion exceeded its budget",
extra={"job_id": job.id, "limit_seconds": settings.sync_timeout_seconds},
)
raise SyncTooLongError(
"Konverze presahla limit pro synchronni pozadavek. Pouzijte asynchronni endpoint POST /jobs.",
{"limit_seconds": settings.sync_timeout_seconds},
) from exc
if job.state.status != "done":
error = job.state.error
if error is not None:
raise error_from_code(error.error_code, error.message, error.detail)
raise ConversionError("Konverze skoncila ve stavu " + job.state.status + ".")
path = container.storage.result_path(job.id)
filename = request.filename or "dokument.pdf"
headers = {
"X-Page-Count": str(job.state.page_count or 0),
"X-Engine-Used": job.state.engine_used or "",
}
if job.state.missing_assets:
headers["X-Missing-Assets"] = str(len(job.state.missing_assets))
if job.state.warnings:
headers["X-Warnings"] = str(len(job.state.warnings))
return FileResponse(
path,
media_type="application/pdf",
filename=filename,
headers=headers,
background=BackgroundTask(container.storage.discard, job.id),
)
+47
View File
@@ -0,0 +1,47 @@
"""Health and version endpoints required by AppFactory."""
from __future__ import annotations
import logging
from fastapi import APIRouter
from ..deps import get_container
from ..models import HealthResponse
logger = logging.getLogger(__name__)
router = APIRouter(tags=["service"])
@router.get("/health", response_model=HealthResponse, summary="Stav sluzby")
async def health() -> HealthResponse:
container = get_container()
engines = {}
for name, engine in container.engines.items():
try:
engines[name] = await engine.available()
except Exception as exc:
logger.error("Engine availability check failed", extra={"engine": name}, exc_info=exc)
engines[name] = False
status = "ok" if any(engines.values()) else "degraded"
return HealthResponse(
status=status,
app=container.settings.app_name,
version=container.settings.app_version,
engines=engines,
queue=container.jobs.stats(),
)
@router.get("/version", summary="Verze a zakladni informace o sluzbe")
async def version() -> dict:
settings = get_container().settings
return {
"app": settings.app_name,
"version": settings.app_version,
"language": "python",
"root_path": settings.root_path,
}
+90
View File
@@ -0,0 +1,90 @@
"""Asynchronous conversion. This is the path for large documents."""
from __future__ import annotations
import logging
from fastapi import APIRouter, Response, status
from fastapi.responses import FileResponse
from ..deps import get_container
from ..errors import ConversionError, JobNotFoundError
from ..models import ConvertRequest, JobAccepted, JobState
logger = logging.getLogger(__name__)
router = APIRouter(prefix="/jobs", tags=["joby"])
def _result_url(job_id: str) -> str:
root = get_container().settings.root_path.rstrip("/")
return f"{root}/jobs/{job_id}/result"
@router.post(
"",
response_model=JobAccepted,
status_code=status.HTTP_202_ACCEPTED,
summary="Zaradi konverzi do fronty",
)
async def create_job(request: ConvertRequest) -> JobAccepted:
job = get_container().jobs.submit(request)
return JobAccepted(
job_id=job.id,
status=job.state.status,
created_at=job.state.created_at,
result_url=_result_url(job.id),
)
@router.get("/{job_id}", response_model=JobState, summary="Stav jobu")
async def job_state(job_id: str) -> JobState:
job = get_container().jobs.get(job_id)
state = job.state.model_copy()
if state.status == "done":
state.result_url = _result_url(job_id)
return state
@router.get(
"/{job_id}/result",
summary="Stahne hotove PDF",
response_class=Response,
responses={
200: {"content": {"application/pdf": {}}, "description": "Hotove PDF."},
404: {"description": "Job neexistuje nebo uz expiroval."},
409: {"description": "Job jeste nedobehl nebo skoncil chybou."},
},
)
async def job_result(job_id: str):
container = get_container()
job = container.jobs.get(job_id)
if job.state.status != "done":
error = ConversionError(
f"Vysledek neni k dispozici, job je ve stavu {job.state.status}.",
{"status": job.state.status},
)
error.error_code = "result_not_ready"
error.status_code = 409
raise error
path = container.storage.result_path(job_id)
if not path.exists():
raise JobNotFoundError("Soubor s vysledkem uz neexistuje, job pravdepodobne expiroval.")
filename = job.request.filename or "dokument.pdf"
return FileResponse(
path,
media_type="application/pdf",
filename=filename,
headers={
"X-Page-Count": str(job.state.page_count or 0),
"X-Engine-Used": job.state.engine_used or "",
},
)
@router.delete("/{job_id}", response_model=JobState, summary="Zrusi job nebo smaze jeho vysledek")
async def delete_job(job_id: str) -> JobState:
return get_container().jobs.cancel(job_id).state
View File
+181
View File
@@ -0,0 +1,181 @@
"""Fetching of the source document and of its assets.
Both paths go through UrlGuard. Asset failures never abort the render, they are
collected and reported back to the caller so nobody silently receives a PDF with
missing images.
"""
from __future__ import annotations
import io
import logging
from dataclasses import dataclass, field
import httpx
from ..config import Settings
from ..errors import LimitExceededError, SourceUnavailableError
from ..models import MissingAsset
from .security import UrlGuard
logger = logging.getLogger(__name__)
USER_AGENT = "html-to-pdf/1.0 (AppFactory)"
@dataclass
class FetchedDocument:
html: str
base_url: str
@dataclass
class AssetReport:
"""Collects assets that could not be loaded during a render."""
missing: list[MissingAsset] = field(default_factory=list)
_seen: set[str] = field(default_factory=set)
def add(self, url: str, reason: str) -> None:
if url in self._seen:
return
self._seen.add(url)
self.missing.append(MissingAsset(url=url, reason=reason))
logger.warning("Asset could not be loaded", extra={"asset_url": url, "reason": reason})
def fetch_document(url: str, guard: UrlGuard, settings: Settings) -> FetchedDocument:
"""Download the source HTML, validating every redirect hop."""
current = url
timeout = httpx.Timeout(settings.fetch_timeout_seconds)
with httpx.Client(follow_redirects=False, timeout=timeout, headers={"User-Agent": USER_AGENT}) as client:
for hop in range(settings.max_redirects + 1):
guard.check(current)
try:
response = client.get(current)
except httpx.HTTPError as exc:
raise SourceUnavailableError(
"Zdrojovy dokument se nepodarilo stahnout.",
{"url": current, "reason": str(exc)},
) from exc
if response.is_redirect:
location = response.headers.get("location")
if not location:
raise SourceUnavailableError(
"Zdroj vratil presmerovani bez hlavicky Location.", {"url": current}
)
current = str(response.url.join(location))
logger.info("Following redirect", extra={"hop": hop + 1, "target": current})
continue
if response.status_code >= 400:
raise SourceUnavailableError(
f"Zdrojovy dokument vratil HTTP {response.status_code}.",
{"url": current, "status_code": response.status_code},
)
_check_size(len(response.content), settings)
return FetchedDocument(html=response.text, base_url=str(response.url))
raise SourceUnavailableError(
"Prekrocen maximalni pocet presmerovani.",
{"url": url, "max_redirects": settings.max_redirects},
)
def _check_size(size: int, settings: Settings) -> None:
if settings.max_html_bytes and size > settings.max_html_bytes:
raise LimitExceededError(
"Zdrojovy dokument je vetsi nez nakonfigurovany limit MAX_HTML_BYTES.",
{"size_bytes": size, "limit_bytes": settings.max_html_bytes},
)
def build_url_fetcher(guard: UrlGuard, report: AssetReport, allow_remote: bool, timeout_seconds: int):
"""Return a WeasyPrint url_fetcher with SSRF checks and failure reporting."""
from weasyprint import default_url_fetcher
def fetcher(url: str, timeout: int = 10, ssl_context=None): # noqa: ARG001
if url.startswith("data:"):
return default_url_fetcher(url)
if not allow_remote:
report.add(url, "stahovani externich assetu je vypnute")
return _empty_asset()
try:
guard.check(url)
except Exception as exc: # noqa: BLE001 - reported, never silent
report.add(url, f"zablokovano: {exc}")
return _empty_asset()
try:
response = httpx.get(
url,
timeout=timeout_seconds,
follow_redirects=False,
headers={"User-Agent": USER_AGENT},
)
while response.is_redirect:
location = response.headers.get("location")
if not location:
raise httpx.HTTPError("presmerovani bez hlavicky Location")
target = str(response.url.join(location))
guard.check(target)
response = httpx.get(
target,
timeout=timeout_seconds,
follow_redirects=False,
headers={"User-Agent": USER_AGENT},
)
if response.status_code >= 400:
report.add(url, f"HTTP {response.status_code}")
return _empty_asset()
return {
"string": response.content,
"mime_type": response.headers.get("content-type", "").split(";")[0] or None,
"redirected_url": str(response.url),
}
except Exception as exc: # noqa: BLE001 - reported, never silent
report.add(url, str(exc))
return _empty_asset()
return fetcher
def _empty_asset() -> dict:
"""Placeholder returned instead of a failed asset so the render continues."""
return {"file_obj": io.BytesIO(b""), "mime_type": "application/octet-stream"}
@dataclass
class AssetGate:
"""Per job gateway for everything the render pulls from the network."""
guard: UrlGuard
report: AssetReport
allow_remote: bool = True
timeout_seconds: int = 10
def weasy_fetcher(self):
return build_url_fetcher(self.guard, self.report, self.allow_remote, self.timeout_seconds)
def allowed(self, url: str) -> tuple[bool, str]:
"""Decide whether a browser initiated request may proceed."""
if url.startswith("data:") or url.startswith("blob:") or url.startswith("about:"):
return True, ""
if not url.startswith(("http://", "https://")):
return False, "nepovolene schema"
if not self.allow_remote:
return False, "stahovani externich assetu je vypnute"
try:
self.guard.check(url)
except Exception as exc: # noqa: BLE001 - reported, never silent
return False, str(exc)
return True, ""
+238
View File
@@ -0,0 +1,238 @@
"""In memory job queue.
The queue is deliberately in process. There is no broker and no database, which
means a restart loses queued work. Such jobs are marked as failed with an
explicit reason instead of silently disappearing.
"""
from __future__ import annotations
import asyncio
import logging
import uuid
from datetime import datetime, timedelta, timezone
import httpx
from ..config import Settings
from ..errors import ConversionError, JobNotFoundError, QueueFullError
from ..logging_setup import current_job_id
from ..models import ConvertRequest, ErrorInfo, JobProgress, JobState
from .pipeline import ConversionPipeline
from .storage import Storage
logger = logging.getLogger(__name__)
def _now() -> datetime:
return datetime.now(timezone.utc)
class Job:
def __init__(self, job_id: str, request: ConvertRequest) -> None:
self.id = job_id
self.request = request
self.state = JobState(job_id=job_id, status="queued", created_at=_now())
self.task = None
self.done = asyncio.Event()
class JobManager:
def __init__(self, pipeline: ConversionPipeline, storage: Storage, settings: Settings) -> None:
self._pipeline = pipeline
self._storage = storage
self._settings = settings
self._jobs = {}
self._queue = asyncio.Queue(maxsize=settings.queue_max_size)
self._workers = []
self._cleaner = None
async def start(self) -> None:
self._storage.sweep_orphans(set(self._jobs))
for index in range(max(1, self._settings.workers)):
self._workers.append(asyncio.create_task(self._worker(index), name=f"htp-worker-{index}"))
self._cleaner = asyncio.create_task(self._cleanup_loop(), name="htp-cleanup")
logger.info("Job manager started", extra={"workers": len(self._workers)})
async def stop(self) -> None:
for task in self._workers:
task.cancel()
if self._cleaner is not None:
self._cleaner.cancel()
for job in self._jobs.values():
if job.state.status in ("queued", "running"):
self._fail(
job,
ErrorInfo(
error_code="service_restarted",
message="Sluzba byla ukoncena drive, nez job dobehl. Odeslete pozadavek znovu.",
),
)
logger.info("Job manager stopped")
def submit(self, request: ConvertRequest) -> Job:
job = Job(str(uuid.uuid4()), request)
self._jobs[job.id] = job
try:
self._queue.put_nowait(job.id)
except asyncio.QueueFull as exc:
del self._jobs[job.id]
raise QueueFullError(
"Fronta je plna, zkuste to prosim za chvili.",
{"queue_max_size": self._settings.queue_max_size},
) from exc
logger.info("Job queued", extra={"job_id": job.id, "queue_size": self._queue.qsize()})
return job
def get(self, job_id: str) -> Job:
job = self._jobs.get(job_id)
if job is None:
raise JobNotFoundError("Job s timto identifikatorem neexistuje nebo uz expiroval.")
return job
def cancel(self, job_id: str) -> Job:
job = self.get(job_id)
if job.task is not None and not job.task.done():
job.task.cancel()
else:
job.state.status = "cancelled"
job.state.finished_at = _now()
self._storage.discard(job.id)
job.done.set()
job.state.expires_at = _now() + timedelta(seconds=self._settings.job_result_ttl_seconds)
logger.info("Job cancelled", extra={"job_id": job_id})
return job
def stats(self) -> dict:
counts = {"queued": 0, "running": 0, "done": 0, "failed": 0, "cancelled": 0, "expired": 0}
for job in self._jobs.values():
counts[job.state.status] = counts.get(job.state.status, 0) + 1
counts["workers"] = len(self._workers)
return counts
async def _worker(self, index: int) -> None:
while True:
job_id = await self._queue.get()
job = self._jobs.get(job_id)
if job is None or job.state.status != "queued":
self._queue.task_done()
continue
job.task = asyncio.current_task()
token = current_job_id.set(job.id)
try:
await self._execute(job)
except asyncio.CancelledError:
job.state.status = "cancelled"
job.state.finished_at = _now()
job.state.expires_at = _now() + timedelta(seconds=self._settings.job_result_ttl_seconds)
self._storage.discard(job.id)
logger.info("Job execution cancelled", extra={"job_id": job.id})
finally:
current_job_id.reset(token)
job.task = None
job.done.set()
self._queue.task_done()
await self._notify_callback(job)
async def _execute(self, job: Job) -> None:
job.state.status = "running"
job.state.started_at = _now()
logger.info("Job started")
def progress(pages, chunks_done, chunks_total, pass_number):
job.state.progress = JobProgress(
pages_rendered=pages,
chunks_done=chunks_done,
chunks_total=chunks_total,
pass_number=pass_number,
)
workdir = self._storage.job_dir(job.id)
try:
result = await self._pipeline.run(job.request, workdir, progress)
except ConversionError as exc:
logger.error("Job failed", extra={"error_code": exc.error_code}, exc_info=exc)
self._fail(job, ErrorInfo(error_code=exc.error_code, message=exc.message, detail=exc.detail or None))
return
except asyncio.CancelledError:
raise
except Exception as exc:
logger.exception("Job failed with an unexpected error")
self._fail(
job,
ErrorInfo(
error_code="internal_error",
message="Pri generovani PDF doslo k neocekavane chybe.",
detail={"reason": str(exc)},
),
)
return
self._storage.publish(job.id, result.path)
job.state.status = "done"
job.state.finished_at = _now()
job.state.expires_at = _now() + timedelta(seconds=self._settings.job_result_ttl_seconds)
job.state.page_count = result.page_count
job.state.engine_used = result.engine_used
job.state.missing_assets = result.missing_assets
job.state.warnings = result.warnings
logger.info(
"Job finished",
extra={
"pages": result.page_count,
"engine": result.engine_used,
"missing_assets": len(result.missing_assets),
},
)
def _fail(self, job: Job, error: ErrorInfo) -> None:
job.state.status = "failed"
job.state.finished_at = _now()
job.state.expires_at = _now() + timedelta(seconds=self._settings.job_result_ttl_seconds)
job.state.error = error
self._storage.discard(job.id)
async def _notify_callback(self, job: Job) -> None:
url = job.request.callback_url
if not url or job.state.status not in ("done", "failed"):
return
payload = job.state.model_dump(mode="json")
for attempt in range(1, max(1, self._settings.callback_retries) + 1):
try:
async with httpx.AsyncClient(timeout=self._settings.callback_timeout_seconds) as client:
response = await client.post(url, json=payload)
if response.status_code < 400:
logger.info("Callback delivered", extra={"attempt": attempt})
return
logger.warning(
"Callback returned an error status",
extra={"attempt": attempt, "status_code": response.status_code},
)
except Exception as exc:
logger.warning("Callback delivery failed", extra={"attempt": attempt}, exc_info=exc)
await asyncio.sleep(min(2 ** attempt, 10))
logger.error("Callback could not be delivered, job result stays available over the API")
async def _cleanup_loop(self) -> None:
while True:
await asyncio.sleep(60)
try:
self._expire_old_jobs()
except Exception:
logger.exception("Cleanup loop failed")
def _expire_old_jobs(self) -> None:
now = _now()
for job_id, job in list(self._jobs.items()):
expires_at = job.state.expires_at
if expires_at is None or expires_at > now:
continue
job.state.status = "expired"
self._storage.discard(job_id)
del self._jobs[job_id]
logger.info("Job result expired and was removed", extra={"job_id": job_id})
+350
View File
@@ -0,0 +1,350 @@
"""The conversion pipeline.
Order of operations:
1. load the source HTML, validating the target address
2. pick the engine
3. decide how page numbers will be produced
4. build the document, inject the page stylesheet, optionally insert the table
of contents placeholder
5. split into chunks
6. first render pass, which yields the real page count and anchor positions
7. second render pass when a table of contents needs real page numbers
8. merge the chunks
9. stamp the numbering overlay when CSS counters cannot be used
"""
from __future__ import annotations
import asyncio
import logging
import re
from dataclasses import dataclass, field
from pathlib import Path
from typing import Callable
from ..config import Settings, get_settings
from ..errors import (
ConversionError,
LimitExceededError,
RenderTimeoutError,
UnsupportedCombinationError,
)
from ..models import ConvertRequest, MissingAsset
from ..pdf import merger, paginator
from ..pdf.chunker import split_document
from ..pdf.document import SourceDocument
from ..pdf.styles import build_page_css
from .fetcher import AssetGate, AssetReport, fetch_document
from .security import UrlGuard
logger = logging.getLogger(__name__)
SCRIPT_TAG = re.compile(r"<script\b[^>]*>(.*?)</script>", re.IGNORECASE | re.DOTALL)
SCRIPT_SRC = re.compile(r"<script\b[^>]*\bsrc\s*=", re.IGNORECASE)
ProgressCallback = Callable[[int, int, int, int], None]
@dataclass
class ConversionResult:
path: Path
page_count: int
engine_used: str
missing_assets: list[MissingAsset] = field(default_factory=list)
warnings: list[str] = field(default_factory=list)
class ConversionPipeline:
def __init__(self, engines: dict, settings: Settings | None = None) -> None:
self._engines = engines
self._settings = settings or get_settings()
self._guard = UrlGuard(self._settings)
async def run(
self,
request: ConvertRequest,
workdir: Path,
progress: ProgressCallback | None = None,
) -> ConversionResult:
workdir.mkdir(parents=True, exist_ok=True)
timeout = self._settings.max_render_seconds
coroutine = self._run_with_fallback(request, workdir, progress)
if timeout:
try:
return await asyncio.wait_for(coroutine, timeout=timeout)
except asyncio.TimeoutError as exc:
raise RenderTimeoutError(
"Render prekrocil nakonfigurovany limit MAX_RENDER_SECONDS.",
{"limit_seconds": timeout},
) from exc
return await coroutine
async def _run_with_fallback(
self, request: ConvertRequest, workdir: Path, progress: ProgressCallback | None
) -> ConversionResult:
raw_html, base_url = await self._load_source(request)
requested = request.engine
engine_name = self._select_engine(requested, raw_html)
try:
return await self._execute(request, raw_html, base_url, engine_name, workdir, progress)
except ConversionError:
raise
except Exception as exc:
if requested != "auto" or engine_name != "weasyprint" or "chromium" not in self._engines:
raise
logger.warning(
"WeasyPrint render failed, falling back to Chromium",
exc_info=exc,
extra={"engine": engine_name},
)
result = await self._execute(request, raw_html, base_url, "chromium", workdir, progress)
result.warnings.append(
"Render pres WeasyPrint selhal, dokument byl vygenerovan pres Chromium."
)
return result
def _select_engine(self, requested: str, raw_html: str) -> str:
if requested != "auto":
if requested not in self._engines:
raise UnsupportedCombinationError(
f"Engine {requested} neni v teto instanci k dispozici.",
{"available": sorted(self._engines)},
)
return requested
if self._has_active_scripts(raw_html) and "chromium" in self._engines:
logger.info("Auto engine selected chromium because the document contains scripts")
return "chromium"
if "weasyprint" in self._engines:
return "weasyprint"
return next(iter(self._engines))
@staticmethod
def _has_active_scripts(raw_html: str) -> bool:
if SCRIPT_SRC.search(raw_html):
return True
return any(body.strip() for body in SCRIPT_TAG.findall(raw_html))
async def _load_source(self, request: ConvertRequest) -> tuple:
if request.source.html is not None:
html = request.source.html
limit = self._settings.max_html_bytes
if limit and len(html.encode("utf-8")) > limit:
raise LimitExceededError(
"Zdrojove HTML je vetsi nez nakonfigurovany limit MAX_HTML_BYTES.",
{"limit_bytes": limit},
)
return html, request.source.base_url
fetched = await asyncio.to_thread(
fetch_document, request.source.url, self._guard, self._settings
)
return fetched.html, request.source.base_url or fetched.base_url
async def _execute(
self,
request: ConvertRequest,
raw_html: str,
base_url: str | None,
engine_name: str,
workdir: Path,
progress: ProgressCallback | None,
) -> ConversionResult:
engine = self._engines[engine_name]
report = AssetReport()
gate = AssetGate(
guard=self._guard,
report=report,
allow_remote=request.assets.allow_remote,
timeout_seconds=request.assets.timeout_seconds,
)
warnings: list[str] = []
numbering_mode = self._numbering_mode(request, engine_name)
if request.toc.enabled and not getattr(engine, "supports_anchor_pages", False):
raise UnsupportedCombinationError(
"Generovani obsahu s cisly stranek podporuje pouze engine weasyprint.",
{"engine": engine_name},
)
document = SourceDocument(raw_html)
document.set_base_url(base_url)
document.append_stylesheet(
build_page_css(
request.page,
request.page_numbers if numbering_mode == "css" else None,
total_pages=None,
outline=request.outline,
)
)
headings = document.collect_headings(request.toc.depth) if request.toc.enabled else []
if request.toc.enabled:
if not headings:
warnings.append("Dokument neobsahuje zadne nadpisy, obsah nebyl vygenerovan.")
request.toc.enabled = False
else:
document.insert_toc(headings, request.toc.title, pages=None)
chunks = self._split(document, request)
direct_navigation = self._can_navigate_directly(request, engine_name, chunks)
if direct_navigation:
logger.info("Chromium will navigate to the source URL directly so its scripts run in context")
renders = await self._render_pass(
engine, chunks, base_url, request, workdir, gate, progress, 1, direct_navigation
)
if request.toc.enabled:
anchor_pages = self._absolute_anchor_pages(renders)
missing = [item.anchor for item in headings if item.anchor not in anchor_pages]
if missing:
logger.warning(
"Some headings have no anchor position, their page numbers stay empty",
extra={"missing_anchors": len(missing)},
)
document.remove_toc()
document.insert_toc(headings, request.toc.title, pages=anchor_pages)
chunks = self._split(document, request)
renders = await self._render_pass(
engine, chunks, base_url, request, workdir, gate, progress, 2
)
if len(chunks) > 1:
warnings.append(
"Dokument byl rozdelen na casti, odkazy v obsahu proto nejsou klikatelne. "
"Cisla stranek jsou spravna."
)
merged_path = workdir / "merged.pdf"
total_pages = merger.merge([item.path for item in renders], merged_path)
self._check_page_limit(total_pages)
final_path = merged_path
if request.page_numbers.enabled and numbering_mode == "overlay":
final_path = await asyncio.to_thread(
self._stamp_numbers, merged_path, workdir, request, total_pages
)
on_job_finished = getattr(engine, "on_job_finished", None)
if callable(on_job_finished):
on_job_finished()
return ConversionResult(
path=final_path,
page_count=total_pages,
engine_used=engine_name,
missing_assets=report.missing,
warnings=warnings,
)
@staticmethod
def _can_navigate_directly(request: ConvertRequest, engine_name: str, chunks: list) -> bool:
"""Chromium renders a foreign page best when it loads the URL itself.
Only possible when the document is not split and needs no injected
markup, otherwise the modified HTML has to be pushed into the page.
"""
return (
engine_name == "chromium"
and request.source.url is not None
and not request.toc.enabled
and len(chunks) == 1
)
def _numbering_mode(self, request: ConvertRequest, engine_name: str) -> str:
mode = request.page_numbers.mode
if mode == "auto":
if engine_name == "weasyprint" and not request.chunking.enabled:
return "css"
return "overlay"
if mode == "css" and engine_name != "weasyprint":
raise UnsupportedCombinationError(
"Rezim cislovani css funguje pouze s enginem weasyprint. Pouzijte overlay nebo auto.",
{"engine": engine_name},
)
if mode == "css" and request.chunking.enabled:
raise UnsupportedCombinationError(
"Rezim cislovani css nelze kombinovat s chunkovanim, protoze citac stranek se v kazde "
"casti restartuje. Vypnete chunking nebo pouzijte overlay.",
)
return mode
def _split(self, document: SourceDocument, request: ConvertRequest) -> list:
if not request.chunking.enabled:
return [document.to_html()]
return split_document(document, request.chunking.pages_per_chunk)
async def _render_pass(
self,
engine,
chunks: list,
base_url: str | None,
request: ConvertRequest,
workdir: Path,
gate: AssetGate,
progress: ProgressCallback | None,
pass_number: int,
direct_navigation: bool = False,
) -> list:
renders = []
pages_rendered = 0
for index, chunk_html in enumerate(chunks):
output = workdir / f"pass{pass_number}-chunk{index:04d}.pdf"
render = await engine.render_chunk(
None if direct_navigation else chunk_html,
base_url,
request,
output,
total_pages=None,
asset_gate=gate,
)
renders.append(render)
pages_rendered += render.page_count
if progress is not None:
progress(pages_rendered, index + 1, len(chunks), pass_number)
self._check_page_limit(pages_rendered)
logger.info(
"Render pass finished",
extra={"pass_number": pass_number, "chunks": len(chunks), "pages": pages_rendered},
)
return renders
@staticmethod
def _absolute_anchor_pages(renders: list) -> dict:
pages: dict = {}
offset = 0
for render in renders:
for anchor, local_page in render.anchor_pages.items():
pages.setdefault(anchor, offset + local_page + 1)
offset += render.page_count
return pages
def _check_page_limit(self, pages: int) -> None:
if self._settings.max_pages and pages > self._settings.max_pages:
raise LimitExceededError(
"Dokument ma vice stranek nez nakonfigurovany limit MAX_PAGES.",
{"pages": pages, "limit": self._settings.max_pages},
)
@staticmethod
def _stamp_numbers(merged_path: Path, workdir: Path, request: ConvertRequest, total_pages: int) -> Path:
width, height = merger.first_page_size(merged_path)
overlay_path = workdir / "overlay.pdf"
paginator.build_overlay(
total_pages, width, height, request.page, request.page_numbers, overlay_path
)
numbered_path = workdir / "numbered.pdf"
paginator.apply_overlay(merged_path, overlay_path, numbered_path)
return numbered_path
+132
View File
@@ -0,0 +1,132 @@
"""SSRF protection.
The service fetches arbitrary URLs on request, which is exactly the shape of an
SSRF vulnerability. Every URL is validated after DNS resolution, not on the
string alone, and the check is repeated on every redirect hop.
"""
from __future__ import annotations
import ipaddress
import logging
import socket
from urllib.parse import urlparse
from ..config import Settings
from ..errors import BlockedTargetError
logger = logging.getLogger(__name__)
ALLOWED_SCHEMES = ("http", "https")
# Ranges that must never be reachable from a user supplied URL.
PRIVATE_NETWORKS = [
ipaddress.ip_network("127.0.0.0/8"),
ipaddress.ip_network("10.0.0.0/8"),
ipaddress.ip_network("172.16.0.0/12"),
ipaddress.ip_network("192.168.0.0/16"),
ipaddress.ip_network("169.254.0.0/16"),
ipaddress.ip_network("0.0.0.0/8"),
ipaddress.ip_network("100.64.0.0/10"),
ipaddress.ip_network("192.0.0.0/24"),
ipaddress.ip_network("198.18.0.0/15"),
ipaddress.ip_network("224.0.0.0/4"),
ipaddress.ip_network("240.0.0.0/4"),
ipaddress.ip_network("::1/128"),
ipaddress.ip_network("fc00::/7"),
ipaddress.ip_network("fe80::/10"),
ipaddress.ip_network("::/128"),
]
class UrlGuard:
"""Validates URLs against the configured policy."""
def __init__(self, settings: Settings) -> None:
self._block_private = settings.ssrf_block_private
self._allowed_hosts = {host.lower() for host in settings.ssrf_allowed_hosts}
self._extra_blocked: list[ipaddress._BaseNetwork] = []
for cidr in settings.ssrf_extra_blocked_cidrs:
try:
self._extra_blocked.append(ipaddress.ip_network(cidr, strict=False))
except ValueError:
logger.warning("Ignoring invalid CIDR in SSRF_EXTRA_BLOCKED_CIDRS", extra={"cidr": cidr})
def check(self, url: str) -> str:
"""Raise BlockedTargetError when the URL must not be fetched.
Returns the hostname so callers can reuse it without parsing again.
"""
parsed = urlparse(url)
scheme = (parsed.scheme or "").lower()
if scheme not in ALLOWED_SCHEMES:
raise BlockedTargetError(
"Povolena jsou pouze schemata http a https.",
{"url": url, "scheme": scheme or None},
)
host = parsed.hostname
if not host:
raise BlockedTargetError("Adresa neobsahuje hostname.", {"url": url})
if host.lower() in self._allowed_hosts:
logger.info("Host explicitly allowlisted", extra={"host": host})
return host
for address in self._resolve(host, url):
self._check_address(address, host, url)
return host
def _resolve(self, host: str, url: str) -> list[ipaddress.IPv4Address | ipaddress.IPv6Address]:
# A literal IP address needs no lookup.
try:
return [ipaddress.ip_address(host)]
except ValueError:
pass
try:
infos = socket.getaddrinfo(host, None, proto=socket.IPPROTO_TCP)
except socket.gaierror as exc:
raise BlockedTargetError(
f"Hostname {host} se nepodarilo prelozit na IP adresu.",
{"url": url, "reason": str(exc)},
) from exc
addresses = []
for info in infos:
try:
addresses.append(ipaddress.ip_address(info[4][0]))
except ValueError:
continue
if not addresses:
raise BlockedTargetError(f"Hostname {host} nema zadnou pouzitelnou IP adresu.", {"url": url})
return addresses
def _check_address(self, address, host: str, url: str) -> None:
if self._block_private:
for network in PRIVATE_NETWORKS:
if address.version == network.version and address in network:
logger.warning(
"Blocked request to private address",
extra={"host": host, "address": str(address), "network": str(network)},
)
raise BlockedTargetError(
"Cilova adresa smeruje do privatniho nebo vyhrazeneho rozsahu a je zablokovana.",
{"url": url, "host": host, "address": str(address)},
)
for network in self._extra_blocked:
if address.version == network.version and address in network:
logger.warning(
"Blocked request by configured CIDR",
extra={"host": host, "address": str(address), "network": str(network)},
)
raise BlockedTargetError(
"Cilova adresa je v konfigurovanem seznamu blokovanych rozsahu.",
{"url": url, "host": host, "address": str(address)},
)
+65
View File
@@ -0,0 +1,65 @@
"""Temporary storage of job working directories and results.
Nothing is kept longer than needed. A result lives until it is picked up or
until its TTL expires, whichever comes first.
"""
from __future__ import annotations
import logging
import shutil
from pathlib import Path
logger = logging.getLogger(__name__)
RESULT_NAME = "result.pdf"
class Storage:
def __init__(self, root: str) -> None:
self.root = Path(root)
self.root.mkdir(parents=True, exist_ok=True)
def job_dir(self, job_id: str) -> Path:
path = self.root / job_id
path.mkdir(parents=True, exist_ok=True)
return path
def result_path(self, job_id: str) -> Path:
return self.root / job_id / RESULT_NAME
def publish(self, job_id: str, produced: Path) -> Path:
"""Move the produced file to its final name and drop the intermediates."""
target = self.result_path(job_id)
if produced != target:
produced.replace(target)
for item in self.job_dir(job_id).iterdir():
if item.name == RESULT_NAME:
continue
self._remove(item)
return target
def discard(self, job_id: str) -> None:
self._remove(self.root / job_id)
def _remove(self, path: Path) -> None:
try:
if path.is_dir():
shutil.rmtree(path, ignore_errors=False)
elif path.exists():
path.unlink()
except OSError as exc:
logger.warning("Could not remove temporary path", extra={"path": str(path)}, exc_info=exc)
def sweep_orphans(self, known_job_ids: set[str]) -> int:
"""Remove directories that belong to no known job, for example after a restart."""
removed = 0
for item in self.root.iterdir():
if item.is_dir() and item.name not in known_job_ids:
self._remove(item)
removed += 1
if removed:
logger.info("Removed orphaned job directories", extra={"count": removed})
return removed