diff --git a/Dockerfile b/Dockerfile index f1dc70a..b2b9ec4 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,12 +1,63 @@ -FROM python:3.12-slim +# Build stage: only wheels, no compilers in the final image. +FROM python:3.12-slim AS builder -WORKDIR /app +ENV PIP_NO_CACHE_DIR=1 \ + PIP_DISABLE_PIP_VERSION_CHECK=1 + +RUN apt-get update && apt-get install -y --no-install-recommends \ + build-essential \ + libxml2-dev \ + libxslt1-dev \ + zlib1g-dev \ + && rm -rf /var/lib/apt/lists/* + +RUN python -m venv /opt/venv +ENV PATH="/opt/venv/bin:$PATH" COPY requirements.txt . RUN pip install --no-cache-dir -r requirements.txt + +# Runtime stage. +FROM python:3.12-slim + +ENV PATH="/opt/venv/bin:$PATH" \ + PYTHONUNBUFFERED=1 \ + PLAYWRIGHT_BROWSERS_PATH=/ms-playwright \ + STORAGE_DIR=/tmp/html-to-pdf + +# WeasyPrint needs cairo, pango and harfbuzz. The font packages cover Czech +# diacritics, without them the output silently falls back to boxes. +RUN apt-get update && apt-get install -y --no-install-recommends \ + libcairo2 \ + libpango-1.0-0 \ + libpangocairo-1.0-0 \ + libpangoft2-1.0-0 \ + libgdk-pixbuf-2.0-0 \ + libharfbuzz0b \ + libffi8 \ + shared-mime-info \ + fonts-dejavu \ + fonts-liberation2 \ + fonts-noto-core \ + fonts-noto-color-emoji \ + qpdf \ + ca-certificates \ + curl \ + && rm -rf /var/lib/apt/lists/* + +COPY --from=builder /opt/venv /opt/venv + +# Chromium plus its own system dependencies for the second render engine. +RUN playwright install --with-deps chromium \ + && rm -rf /var/lib/apt/lists/* + +WORKDIR /app COPY app ./app EXPOSE 8000 +HEALTHCHECK --interval=30s --timeout=5s --start-period=20s --retries=3 \ + CMD curl -fsS http://127.0.0.1:8000/health || exit 1 + CMD ["uvicorn", "app.main:app", "--host", "0.0.0.0", "--port", "8000"] diff --git a/README.md b/README.md index 3278008..3b57f21 100644 --- a/README.md +++ b/README.md @@ -1,3 +1,69 @@ # html-to-pdf -Generated by AppFactory. +Sluzba prevadi HTML dokument na PDF. Prijme adresu dokumentu nebo HTML primo +v tele requestu a vrati soubor PDF. Je navrzena pro dokumenty o stovkach az +tisicich stranek. + +Bezi v AppFactory na `https://services.csbot.cz/apps/html-to-pdf`. + +## Jak to funguje + +Dva render enginy: + +- **WeasyPrint** je vychozi. Ma spravne CSS Paged Media, tedy countery stranek, + opakovane hlavicky tabulek a zalozky. Nizka pametova narocnost. Nespousti + JavaScript. +- **Chromium** pres Playwright zvladne i dokumenty dokreslovane JavaScriptem. + Nema pouzitelne countery stranek, cisla se proto dopisuji do hotoveho PDF. + +Velke dokumenty se renderuji po castech a vysledky se slucuji. Cisla stranek +a obsah se dopocitavaji dvoupruchodovym renderem, aby po slouceni sedely. + +## Spusteni + +```bash +docker build -t html-to-pdf . +docker run --rm -p 8000:8000 html-to-pdf +``` + +## Priklad volani + +Maly dokument synchronne: + +```bash +curl -X POST http://localhost:8000/convert \ + -H "Content-Type: application/json" \ + -d '{"source": {"url": "https://example.com/dokument.html"}}' \ + --output dokument.pdf +``` + +Velky dokument asynchronne: + +```bash +# zarazeni do fronty +curl -X POST http://localhost:8000/jobs \ + -H "Content-Type: application/json" \ + -d '{ + "source": {"url": "https://example.com/velky.html"}, + "page_numbers": {"enabled": true}, + "toc": {"enabled": true} + }' + +# stav +curl http://localhost:8000/jobs/ + +# stazeni +curl http://localhost:8000/jobs//result --output velky.pdf +``` + +## Dokumentace + +Podrobnosti jsou ve slozce [documentation](documentation/README.md): + +- [api.md](documentation/api.md) endpointy a telo requestu +- [architektura.md](documentation/architektura.md) jak sluzba funguje uvnitr +- [konfigurace.md](documentation/konfigurace.md) environment variables +- [provoz.md](documentation/provoz.md) nasazeni a znama omezeni +- [zmeny.md](documentation/zmeny.md) zaznam zmen + +Interaktivni dokumentace API je na `/docs`. diff --git a/app/__init__.py b/app/__init__.py new file mode 100644 index 0000000..7f77114 --- /dev/null +++ b/app/__init__.py @@ -0,0 +1 @@ +"""html-to-pdf service package.""" diff --git a/app/config.py b/app/config.py new file mode 100644 index 0000000..ae954d1 --- /dev/null +++ b/app/config.py @@ -0,0 +1,79 @@ +"""Runtime configuration read from environment variables. + +AppFactory injects variables through the generated runtime .env file, so every +option below has a safe default and the service starts without any variable set. +""" + +import os +from dataclasses import dataclass, field +from functools import lru_cache + + +def _bool(name: str, default: bool) -> bool: + raw = os.getenv(name) + if raw is None or raw.strip() == "": + return default + return raw.strip().lower() in ("1", "true", "yes", "on") + + +def _int(name: str, default: int) -> int: + raw = os.getenv(name) + if raw is None or raw.strip() == "": + return default + try: + return int(raw) + except ValueError: + return default + + +def _list(name: str) -> list[str]: + raw = os.getenv(name, "") + return [item.strip() for item in raw.split(",") if item.strip()] + + +@dataclass(frozen=True) +class Settings: + app_name: str = field(default_factory=lambda: os.getenv("APP_NAME", "html-to-pdf")) + app_version: str = field(default_factory=lambda: os.getenv("APP_VERSION", "1.0.0")) + root_path: str = field( + default_factory=lambda: os.getenv("ROOT_PATH") or os.getenv("BASE_PATH") or "" + ) + log_level: str = field(default_factory=lambda: os.getenv("LOG_LEVEL", "INFO").upper()) + + # Job queue + workers: int = field(default_factory=lambda: _int("WORKERS", 2)) + queue_max_size: int = field(default_factory=lambda: _int("QUEUE_MAX_SIZE", 100)) + sync_timeout_seconds: int = field(default_factory=lambda: _int("SYNC_TIMEOUT_SECONDS", 60)) + job_result_ttl_seconds: int = field(default_factory=lambda: _int("JOB_RESULT_TTL_SECONDS", 3600)) + storage_dir: str = field(default_factory=lambda: os.getenv("STORAGE_DIR", "/tmp/html-to-pdf")) + + # Engines + default_engine: str = field(default_factory=lambda: os.getenv("DEFAULT_ENGINE", "auto")) + chromium_enabled: bool = field(default_factory=lambda: _bool("CHROMIUM_ENABLED", True)) + chromium_restart_after_jobs: int = field( + default_factory=lambda: _int("CHROMIUM_RESTART_AFTER_JOBS", 50) + ) + + # Network + fetch_timeout_seconds: int = field(default_factory=lambda: _int("FETCH_TIMEOUT_SECONDS", 30)) + asset_timeout_seconds: int = field(default_factory=lambda: _int("ASSET_TIMEOUT_SECONDS", 10)) + max_redirects: int = field(default_factory=lambda: _int("MAX_REDIRECTS", 5)) + + # SSRF protection. Blocking is on by default and can be widened or narrowed. + ssrf_block_private: bool = field(default_factory=lambda: _bool("SSRF_BLOCK_PRIVATE", True)) + ssrf_extra_blocked_cidrs: list[str] = field(default_factory=lambda: _list("SSRF_EXTRA_BLOCKED_CIDRS")) + ssrf_allowed_hosts: list[str] = field(default_factory=lambda: _list("SSRF_ALLOWED_HOSTS")) + + # Optional limits. Zero means no limit, which is the default on purpose. + max_pages: int = field(default_factory=lambda: _int("MAX_PAGES", 0)) + max_html_bytes: int = field(default_factory=lambda: _int("MAX_HTML_BYTES", 0)) + max_render_seconds: int = field(default_factory=lambda: _int("MAX_RENDER_SECONDS", 0)) + + # Callback delivery + callback_timeout_seconds: int = field(default_factory=lambda: _int("CALLBACK_TIMEOUT_SECONDS", 15)) + callback_retries: int = field(default_factory=lambda: _int("CALLBACK_RETRIES", 3)) + + +@lru_cache(maxsize=1) +def get_settings() -> Settings: + return Settings() diff --git a/app/deps.py b/app/deps.py new file mode 100644 index 0000000..79e3eca --- /dev/null +++ b/app/deps.py @@ -0,0 +1,45 @@ +"""Application container. + +Built once during startup and read by the routers. Keeping it here avoids +importing the FastAPI app from the routers. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +from .config import Settings +from .services.jobs import JobManager +from .services.pipeline import ConversionPipeline +from .services.storage import Storage + + +@dataclass +class Container: + settings: Settings + storage: Storage + pipeline: ConversionPipeline + jobs: JobManager + engines: dict + + +_container: Container | None = None + + +def set_container(container: Container) -> None: + global _container + _container = container + + +def get_container() -> Container: + if _container is None: + raise RuntimeError("Aplikace jeste nebyla inicializovana.") + return _container + + +def get_jobs() -> JobManager: + return get_container().jobs + + +def get_settings_dep() -> Settings: + return get_container().settings diff --git a/app/engines/__init__.py b/app/engines/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/app/engines/base.py b/app/engines/base.py new file mode 100644 index 0000000..3f58f43 --- /dev/null +++ b/app/engines/base.py @@ -0,0 +1,48 @@ +"""Common interface of the render engines.""" + +from __future__ import annotations + +from abc import ABC, abstractmethod +from dataclasses import dataclass, field +from pathlib import Path + +from ..models import ConvertRequest + + +@dataclass +class ChunkRender: + """Result of rendering a single chunk.""" + + path: Path + page_count: int + # Anchor name to zero based page index inside this chunk. + anchor_pages: dict[str, int] = field(default_factory=dict) + + +class RenderEngine(ABC): + name: str + + @abstractmethod + async def available(self) -> bool: + """True when the engine can render right now.""" + + @abstractmethod + async def render_chunk( + self, + html: str, + base_url: str | None, + request: ConvertRequest, + output_path: Path, + total_pages: int | None = None, + asset_gate=None, + ) -> ChunkRender: + """Render one chunk into output_path.""" + + async def shutdown(self) -> None: + """Release engine resources. Default is a no-op.""" + return None + + @property + def supports_anchor_pages(self) -> bool: + """True when the engine can report on which page an anchor landed.""" + return False diff --git a/app/engines/chromium.py b/app/engines/chromium.py new file mode 100644 index 0000000..dbf6a33 --- /dev/null +++ b/app/engines/chromium.py @@ -0,0 +1,211 @@ +"""Headless Chromium engine driven by Playwright. + +Used for foreign URLs and for documents that are completed by JavaScript. +Chromium has no usable CSS page counters, so page numbering for this engine is +always done by the overlay in the post processing step. + +The browser leaks memory over time, therefore the whole instance is restarted +after a configurable number of jobs. +""" + +from __future__ import annotations + +import asyncio +import logging +from pathlib import Path + +from ..config import get_settings +from ..errors import EngineUnavailableError, RenderTimeoutError +from ..models import ConvertRequest +from ..pdf.styles import NAMED_SIZE +from .base import ChunkRender, RenderEngine + +logger = logging.getLogger(__name__) + +LAUNCH_ARGS = [ + "--no-sandbox", + "--disable-dev-shm-usage", + "--disable-gpu", + "--hide-scrollbars", +] + + +class ChromiumEngine(RenderEngine): + name = "chromium" + + def __init__(self) -> None: + self._settings = get_settings() + self._lock = asyncio.Lock() + self._playwright = None + self._browser = None + self._jobs_since_launch = 0 + self._active_renders = 0 + + async def available(self) -> bool: + if not self._settings.chromium_enabled: + return False + try: + await self._ensure_browser(reserve=False) + except Exception as exc: # noqa: BLE001 - reported, never silent + logger.error("Chromium is not available", exc_info=exc) + return False + return True + + async def _ensure_browser(self, reserve: bool = True): + """Return a running browser, restarting it when that is safe. + + The restart must never happen while a render is in flight, otherwise the + health check could close the browser under a running job. + """ + async with self._lock: + due_for_restart = self._jobs_since_launch >= self._settings.chromium_restart_after_jobs + if self._browser is not None and self._active_renders == 0 and due_for_restart: + logger.info( + "Restarting Chromium to release memory", + extra={"jobs_since_launch": self._jobs_since_launch}, + ) + await self._close_browser() + + if self._browser is None: + try: + from playwright.async_api import async_playwright + except ImportError as exc: + raise EngineUnavailableError( + "Engine chromium neni k dispozici, chybi balicek playwright.", + ) from exc + + self._playwright = await async_playwright().start() + self._browser = await self._playwright.chromium.launch(args=LAUNCH_ARGS) + self._jobs_since_launch = 0 + logger.info("Chromium launched") + + if reserve: + self._active_renders += 1 + + return self._browser + + async def _close_browser(self) -> None: + if self._browser is not None: + try: + await self._browser.close() + except Exception as exc: # noqa: BLE001 - reported, never silent + logger.warning("Closing Chromium failed", exc_info=exc) + self._browser = None + if self._playwright is not None: + try: + await self._playwright.stop() + except Exception as exc: # noqa: BLE001 - reported, never silent + logger.warning("Stopping Playwright failed", exc_info=exc) + self._playwright = None + + async def shutdown(self) -> None: + async with self._lock: + await self._close_browser() + + def on_job_finished(self) -> None: + self._jobs_since_launch += 1 + + async def render_chunk( + self, + html: str | None, + base_url: str | None, + request: ConvertRequest, + output_path: Path, + total_pages: int | None = None, + asset_gate=None, + ) -> ChunkRender: + browser = await self._ensure_browser() + try: + context = await browser.new_context() + except Exception: + self._active_renders -= 1 + raise + + try: + if asset_gate is not None: + await context.route("**/*", _make_route_handler(asset_gate)) + + page = await context.new_page() + timeout_ms = request.wait_for.timeout_seconds * 1000 + + try: + if html is None: + if not base_url: + raise EngineUnavailableError("Chybi adresa dokumentu pro engine chromium.") + await page.goto(base_url, wait_until=request.wait_for.state, timeout=timeout_ms) + else: + await page.set_content(html, wait_until=request.wait_for.state, timeout=timeout_ms) + + if request.wait_for.selector: + await page.wait_for_selector(request.wait_for.selector, timeout=timeout_ms) + except Exception as exc: # noqa: BLE001 - converted to a typed error + if "Timeout" in type(exc).__name__ or "timeout" in str(exc).lower(): + raise RenderTimeoutError( + "Chromium nestihl nacist dokument v zadanem casovem limitu.", + {"timeout_seconds": request.wait_for.timeout_seconds}, + ) from exc + raise + + await page.emulate_media(media="print") + pdf_kwargs = _pdf_kwargs(request) + + try: + pdf_bytes = await page.pdf(outline=request.outline, **pdf_kwargs) + except TypeError: + logger.warning("Installed Playwright does not support the outline option, continuing without bookmarks") + pdf_bytes = await page.pdf(**pdf_kwargs) + + output_path.write_bytes(pdf_bytes) + finally: + self._active_renders -= 1 + try: + await context.close() + except Exception as exc: + logger.warning("Closing the browser context failed", exc_info=exc) + + return ChunkRender(path=output_path, page_count=_count_pages(output_path)) + + +def _pdf_kwargs(request: ConvertRequest) -> dict: + margin = request.page.margin + kwargs: dict = { + "print_background": True, + "prefer_css_page_size": True, + "margin": { + "top": margin.top, + "right": margin.right, + "bottom": margin.bottom, + "left": margin.left, + }, + } + fmt = request.page.format.strip() + if NAMED_SIZE.match(fmt): + kwargs["format"] = fmt + else: + width, _, height = fmt.partition(" ") + if width and height: + kwargs["width"] = width.strip() + kwargs["height"] = height.strip() + else: + kwargs["format"] = fmt + kwargs["landscape"] = request.page.orientation == "landscape" + return kwargs + + +def _make_route_handler(asset_gate): + async def handler(route, playwright_request): + allowed, reason = asset_gate.allowed(playwright_request.url) + if allowed: + await route.continue_() + return + asset_gate.report.add(playwright_request.url, reason) + await route.abort() + + return handler + + +def _count_pages(path: Path) -> int: + from pypdf import PdfReader + + with path.open("rb") as handle: + return len(PdfReader(handle).pages) diff --git a/app/engines/weasy.py b/app/engines/weasy.py new file mode 100644 index 0000000..9c3399d --- /dev/null +++ b/app/engines/weasy.py @@ -0,0 +1,107 @@ +"""WeasyPrint engine. + +Default engine for documents we generate ourselves. It implements CSS Paged +Media properly, which is what makes counters, running headers and repeated table +headers work, and it uses far less memory than a headless browser. + +It does not execute JavaScript. That is intentional. +""" + +from __future__ import annotations + +import asyncio +import logging +from pathlib import Path + +from ..models import ConvertRequest +from ..pdf.styles import build_page_css +from .base import ChunkRender, RenderEngine + +logger = logging.getLogger(__name__) + + +class WeasyPrintEngine(RenderEngine): + name = "weasyprint" + + @property + def supports_anchor_pages(self) -> bool: + return True + + async def available(self) -> bool: + try: + import weasyprint # noqa: F401 + except Exception as exc: # noqa: BLE001 - reported, never silent + logger.error("WeasyPrint is not importable", exc_info=exc) + return False + return True + + async def render_chunk( + self, + html: str, + base_url: str | None, + request: ConvertRequest, + output_path: Path, + total_pages: int | None = None, + asset_gate=None, + ) -> ChunkRender: + return await asyncio.to_thread( + self._render_blocking, html, base_url, request, output_path, total_pages, asset_gate + ) + + def _render_blocking( + self, + html: str, + base_url: str | None, + request: ConvertRequest, + output_path: Path, + total_pages: int | None, + asset_gate, + ) -> ChunkRender: + from weasyprint import CSS, HTML + + kwargs = {"string": html, "base_url": base_url} + if asset_gate is not None: + kwargs["url_fetcher"] = asset_gate.weasy_fetcher() + + stylesheets = [] + if total_pages is not None: + # Only used when the page total is known up front, which happens on + # the second pass of an unchunked document. + stylesheets.append( + CSS( + string=build_page_css( + request.page, + request.page_numbers, + total_pages=total_pages, + outline=request.outline, + ) + ) + ) + + document = HTML(**kwargs).render(stylesheets=stylesheets or None) + + write_kwargs = {} + if request.pdf_profile: + write_kwargs["pdf_variant"] = request.pdf_profile + + document.write_pdf(target=str(output_path), **write_kwargs) + + return ChunkRender( + path=output_path, + page_count=len(document.pages), + anchor_pages=self._anchor_pages(document), + ) + + @staticmethod + def _anchor_pages(document) -> dict[str, int]: + """Map anchor name to the zero based page index it landed on.""" + anchors: dict[str, int] = {} + for index, page in enumerate(document.pages): + page_anchors = getattr(page, "anchors", None) + if not page_anchors: + continue + for name in page_anchors: + anchors.setdefault(name, index) + if not anchors: + logger.warning("WeasyPrint returned no anchors, table of contents page numbers may be missing") + return anchors diff --git a/app/errors.py b/app/errors.py new file mode 100644 index 0000000..13c793e --- /dev/null +++ b/app/errors.py @@ -0,0 +1,103 @@ +"""Application errors. + +Every failure surfaces a machine readable error_code and a message in Czech, +because the message is shown to the caller. +""" + +from __future__ import annotations + + +class ConversionError(Exception): + """Base class for every error the conversion pipeline can raise.""" + + error_code = "internal_error" + status_code = 500 + + def __init__(self, message: str, detail: dict | None = None) -> None: + super().__init__(message) + self.message = message + self.detail = detail or {} + + def to_dict(self) -> dict: + payload = {"error_code": self.error_code, "message": self.message} + if self.detail: + payload["detail"] = self.detail + return payload + + +class InvalidRequestError(ConversionError): + error_code = "invalid_request" + status_code = 400 + + +class BlockedTargetError(ConversionError): + """The requested URL points somewhere the service refuses to reach.""" + + error_code = "blocked_target" + status_code = 400 + + +class SourceUnavailableError(ConversionError): + error_code = "source_unavailable" + status_code = 502 + + +class RenderTimeoutError(ConversionError): + error_code = "render_timeout" + status_code = 504 + + +class SyncTooLongError(ConversionError): + """Synchronous conversion exceeded its budget, caller should use /jobs.""" + + error_code = "sync_too_long" + status_code = 413 + + +class EngineUnavailableError(ConversionError): + error_code = "engine_unavailable" + status_code = 503 + + +class UnsupportedCombinationError(ConversionError): + error_code = "unsupported_combination" + status_code = 400 + + +class LimitExceededError(ConversionError): + error_code = "limit_exceeded" + status_code = 400 + + +class JobNotFoundError(ConversionError): + error_code = "job_not_found" + status_code = 404 + + +class QueueFullError(ConversionError): + error_code = "queue_full" + status_code = 503 + + +ERROR_CLASSES = ( + InvalidRequestError, + BlockedTargetError, + SourceUnavailableError, + RenderTimeoutError, + SyncTooLongError, + EngineUnavailableError, + UnsupportedCombinationError, + LimitExceededError, + JobNotFoundError, + QueueFullError, +) + +STATUS_BY_CODE = {cls.error_code: cls.status_code for cls in ERROR_CLASSES} + + +def error_from_code(error_code: str, message: str, detail: dict | None = None) -> ConversionError: + """Rebuild a typed error from a stored job error.""" + error = ConversionError(message, detail) + error.error_code = error_code + error.status_code = STATUS_BY_CODE.get(error_code, 500) + return error diff --git a/app/logging_setup.py b/app/logging_setup.py new file mode 100644 index 0000000..78930b3 --- /dev/null +++ b/app/logging_setup.py @@ -0,0 +1,58 @@ +"""Structured JSON logging. + +Every log record is a single JSON line. Anything related to a job carries the +job_id so the whole conversion can be reconstructed from the log. +""" + +import json +import logging +import sys +from contextvars import ContextVar + +current_job_id: ContextVar[str | None] = ContextVar("current_job_id", default=None) + +_RESERVED = { + "args", "asctime", "created", "exc_info", "exc_text", "filename", "funcName", + "levelname", "levelno", "lineno", "module", "msecs", "message", "msg", "name", + "pathname", "process", "processName", "relativeCreated", "stack_info", + "thread", "threadName", "taskName", +} + + +class JsonFormatter(logging.Formatter): + def format(self, record: logging.LogRecord) -> str: + payload: dict[str, object] = { + "ts": self.formatTime(record, "%Y-%m-%dT%H:%M:%S%z"), + "level": record.levelname, + "logger": record.name, + "message": record.getMessage(), + } + + job_id = getattr(record, "job_id", None) or current_job_id.get() + if job_id: + payload["job_id"] = job_id + + for key, value in record.__dict__.items(): + if key not in _RESERVED and not key.startswith("_") and key != "job_id": + payload[key] = value + + if record.exc_info: + payload["exception"] = self.formatException(record.exc_info) + + return json.dumps(payload, ensure_ascii=False, default=str) + + +def setup_logging(level: str) -> None: + handler = logging.StreamHandler(sys.stdout) + handler.setFormatter(JsonFormatter()) + + root = logging.getLogger() + root.handlers.clear() + root.addHandler(handler) + root.setLevel(getattr(logging, level, logging.INFO)) + + # uvicorn keeps its own handlers, route them through ours as well + for name in ("uvicorn", "uvicorn.access", "uvicorn.error"): + logger = logging.getLogger(name) + logger.handlers.clear() + logger.propagate = True diff --git a/app/main.py b/app/main.py index f543c27..b70a94a 100644 --- a/app/main.py +++ b/app/main.py @@ -1,25 +1,134 @@ -import os -from fastapi import FastAPI +"""Service entry point. -APP_NAME = os.getenv("APP_NAME", "html-to-pdf") -APP_VERSION = os.getenv("APP_VERSION", "1.0.0") -ROOT_PATH = os.getenv("ROOT_PATH", "") +The application runs behind the AppFactory reverse proxy under +/apps/. The prefix is stripped before the request reaches the +container, so only the OpenAPI document has to know about it. That is what +root_path does. +""" + +from __future__ import annotations + +import logging +from contextlib import asynccontextmanager + +from fastapi import FastAPI, Request +from fastapi.responses import JSONResponse + +from .config import get_settings +from .deps import Container, set_container +from .engines.chromium import ChromiumEngine +from .engines.weasy import WeasyPrintEngine +from .errors import ConversionError +from .logging_setup import setup_logging +from .routers import convert as convert_router +from .routers import health as health_router +from .routers import jobs as jobs_router +from .services.jobs import JobManager +from .services.pipeline import ConversionPipeline +from .services.storage import Storage + +logger = logging.getLogger(__name__) + +DESCRIPTION = """ +Sluzba prevadi HTML dokument na PDF. Prijme adresu dokumentu nebo HTML primo +v tele requestu a vrati soubor PDF. + +Pro dokumenty o stovkach az tisicich stranek pouzijte asynchronni endpoint +POST /jobs. Synchronni POST /convert je urceny pro mensi dokumenty. + +Dva render enginy: + +- weasyprint je vychozi, ma spravne strankovani a nizkou pametovou narocnost, + nespousti JavaScript +- chromium zvladne i dokumenty dokreslovane JavaScriptem, ale nema pouzitelne + CSS countery, takze cisla stranek se dopisuji do hotoveho PDF +""" + + +@asynccontextmanager +async def lifespan(app: FastAPI): + settings = get_settings() + setup_logging(settings.log_level) + + engines: dict = {} + + weasy = WeasyPrintEngine() + if await weasy.available(): + engines["weasyprint"] = weasy + else: + logger.error("WeasyPrint engine is unavailable, the service will rely on Chromium only") + + chromium = ChromiumEngine() + if settings.chromium_enabled: + engines["chromium"] = chromium + else: + logger.info("Chromium engine is disabled by configuration") + + if not engines: + logger.error("No render engine is available, conversion requests will fail") + + storage = Storage(settings.storage_dir) + pipeline = ConversionPipeline(engines, settings) + manager = JobManager(pipeline, storage, settings) + + set_container( + Container( + settings=settings, + storage=storage, + pipeline=pipeline, + jobs=manager, + engines=engines, + ) + ) + + await manager.start() + logger.info( + "Service started", + extra={"engines": sorted(engines), "root_path": settings.root_path}, + ) + + try: + yield + finally: + await manager.stop() + for engine in engines.values(): + await engine.shutdown() + logger.info("Service stopped") + + +settings = get_settings() +setup_logging(settings.log_level) app = FastAPI( - title=APP_NAME, - version=APP_VERSION, - root_path=ROOT_PATH + title=settings.app_name, + version=settings.app_version, + description=DESCRIPTION, + root_path=settings.root_path, + lifespan=lifespan, ) -@app.get("/health") -def health(): - return {"status": "ok"} -@app.get("/version") -def version(): - return { - "app": APP_NAME, - "version": APP_VERSION, - "language": "python", - "root_path": ROOT_PATH - } +@app.exception_handler(ConversionError) +async def conversion_error_handler(request: Request, exc: ConversionError) -> JSONResponse: + logger.warning( + "Request failed", + extra={"error_code": exc.error_code, "path": request.url.path, "status_code": exc.status_code}, + ) + return JSONResponse(status_code=exc.status_code, content=exc.to_dict()) + + +@app.exception_handler(Exception) +async def unhandled_error_handler(request: Request, exc: Exception) -> JSONResponse: + logger.exception("Unhandled error", extra={"path": request.url.path}) + return JSONResponse( + status_code=500, + content={ + "error_code": "internal_error", + "message": "Doslo k neocekavane chybe sluzby.", + }, + ) + + +app.include_router(health_router.router) +app.include_router(convert_router.router) +app.include_router(jobs_router.router) diff --git a/app/models.py b/app/models.py new file mode 100644 index 0000000..0e2deb7 --- /dev/null +++ b/app/models.py @@ -0,0 +1,170 @@ +"""Request and response models. + +Only `source` is mandatory. Everything else has a working default so that +{"source": {"url": "..."}} produces a usable PDF. +""" + +from __future__ import annotations + +from datetime import datetime +from typing import Literal + +from pydantic import BaseModel, Field, model_validator + +from .config import get_settings + +EngineName = Literal["auto", "weasyprint", "chromium"] +PagePosition = Literal[ + "top-left", "top-center", "top-right", + "bottom-left", "bottom-center", "bottom-right", +] +JobStatus = Literal["queued", "running", "done", "failed", "cancelled", "expired"] + + +class Source(BaseModel): + url: str | None = Field(default=None, description="Adresa HTML dokumentu ke konverzi.") + html: str | None = Field(default=None, description="HTML poslane primo v tele requestu.") + base_url: str | None = Field( + default=None, + description="Zaklad pro relativni cesty. Pouziva se hlavne spolu s polem html.", + ) + + @model_validator(mode="after") + def exactly_one_source(self) -> "Source": + if bool(self.url) == bool(self.html): + raise ValueError("Vyplnte prave jedno z poli source.url a source.html.") + return self + + +class Margin(BaseModel): + top: str = "20mm" + right: str = "15mm" + bottom: str = "20mm" + left: str = "15mm" + + +class PageSettings(BaseModel): + format: str = Field(default="A4", description="Nazev formatu (A4, A5, Letter) nebo rozmer 210mm 297mm.") + orientation: Literal["portrait", "landscape"] = "portrait" + margin: Margin = Field(default_factory=Margin) + + +class PageNumbers(BaseModel): + enabled: bool = False + format: str = Field(default="{page} / {pages}", description="Zastupne symboly {page} a {pages}.") + position: PagePosition = "bottom-center" + mode: Literal["auto", "css", "overlay"] = Field( + default="auto", + description=( + "auto zvoli css u necleneneho dokumentu a overlay u clenene nebo u Chromia. " + "css pouziva CSS countery, overlay dopisuje cisla do hotoveho PDF." + ), + ) + start_at: int = 1 + + +class TocSettings(BaseModel): + enabled: bool = False + depth: int = Field(default=3, ge=1, le=6) + title: str = "Obsah" + + +class AssetSettings(BaseModel): + allow_remote: bool = True + timeout_seconds: int = Field(default_factory=lambda: get_settings().asset_timeout_seconds, ge=1) + + +class ChunkSettings(BaseModel): + enabled: bool = True + pages_per_chunk: int = Field(default=50, ge=1) + + +class WaitFor(BaseModel): + """Chromium only. Ignored by the WeasyPrint engine.""" + + state: Literal["load", "domcontentloaded", "networkidle"] = "load" + selector: str | None = None + timeout_seconds: int = Field(default=30, ge=1) + + +class ConvertRequest(BaseModel): + source: Source + engine: EngineName = Field(default_factory=lambda: get_settings().default_engine) # type: ignore[arg-type] + page: PageSettings = Field(default_factory=PageSettings) + page_numbers: PageNumbers = Field(default_factory=PageNumbers) + toc: TocSettings = Field(default_factory=TocSettings) + outline: bool = Field(default=True, description="Generovat zalozky PDF z nadpisu h1 az h6.") + pdf_profile: Literal["pdf/a-1b", "pdf/a-2b", "pdf/a-3b", "pdf/a-4b", "pdf/ua-1"] | None = None + assets: AssetSettings = Field(default_factory=AssetSettings) + chunking: ChunkSettings = Field(default_factory=ChunkSettings) + wait_for: WaitFor = Field(default_factory=WaitFor) + filename: str | None = Field(default=None, description="Nazev souboru ve Content-Disposition.") + callback_url: str | None = Field( + default=None, + description="Volitelna adresa, na kterou se po dokonceni jobu posle POST se stavem jobu.", + ) + + model_config = { + "json_schema_extra": { + "examples": [ + {"source": {"url": "https://example.com/dokument.html"}}, + { + "source": {"url": "https://example.com/velky-dokument.html"}, + "engine": "weasyprint", + "page": {"format": "A4", "orientation": "portrait"}, + "page_numbers": {"enabled": True, "format": "{page} / {pages}"}, + "toc": {"enabled": True, "depth": 3, "title": "Obsah"}, + "chunking": {"enabled": True, "pages_per_chunk": 50}, + }, + ] + } + } + + +class MissingAsset(BaseModel): + url: str + reason: str + + +class JobProgress(BaseModel): + pages_rendered: int = 0 + chunks_done: int = 0 + chunks_total: int = 0 + pass_number: int = 0 + + +class ErrorInfo(BaseModel): + error_code: str + message: str + detail: dict | None = None + + +class JobState(BaseModel): + job_id: str + status: JobStatus + created_at: datetime + started_at: datetime | None = None + finished_at: datetime | None = None + expires_at: datetime | None = None + progress: JobProgress = Field(default_factory=JobProgress) + engine_used: str | None = None + page_count: int | None = None + missing_assets: list[MissingAsset] = Field(default_factory=list) + warnings: list[str] = Field(default_factory=list) + error: ErrorInfo | None = None + result_url: str | None = None + + +class JobAccepted(BaseModel): + job_id: str + status: JobStatus + created_at: datetime + result_url: str + + +class HealthResponse(BaseModel): + status: Literal["ok", "degraded"] + app: str + version: str + engines: dict[str, bool] + queue: dict[str, int] diff --git a/app/pdf/__init__.py b/app/pdf/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/app/pdf/chunker.py b/app/pdf/chunker.py new file mode 100644 index 0000000..1d1296b --- /dev/null +++ b/app/pdf/chunker.py @@ -0,0 +1,142 @@ +"""Splitting of a large document into renderable chunks. + +A thousand page document rendered in one pass keeps the whole page tree in +memory. Splitting it on structural boundaries keeps memory flat at the cost of +having to reassemble the page numbering afterwards. + +Cuts are only ever made between direct children of the container element, so a +table or a paragraph is never torn in half. +""" + +from __future__ import annotations + +import logging + +from lxml import html as lxml_html + +from .document import SourceDocument + +logger = logging.getLogger(__name__) + +SPLIT_TAGS = {"section", "article", "h1"} +BREAK_KEYWORDS = ("page-break-before", "break-before") + +# Rough page size heuristic. The real page count is only known after rendering, +# so this only has to be good enough to keep chunks roughly even. +CHARS_PER_PAGE = 2200 +IMAGE_CHAR_WEIGHT = 900 +TABLE_ROW_CHAR_WEIGHT = 120 + + +def is_split_point(element) -> bool: + if not isinstance(element.tag, str): + return False + if element.tag.lower() in SPLIT_TAGS: + return True + if element.get("data-chunk") is not None: + return True + style = (element.get("style") or "").lower() + return any(keyword in style for keyword in BREAK_KEYWORDS) + + +def estimate_pages(element) -> float: + text_length = len(element.text_content()) + text_length += IMAGE_CHAR_WEIGHT * len(element.findall(".//img")) + text_length += TABLE_ROW_CHAR_WEIGHT * len(element.findall(".//tr")) + return max(text_length / CHARS_PER_PAGE, 0.01) + + +def split_document(document: SourceDocument, pages_per_chunk: int) -> list[str]: + """Return one complete HTML document per chunk. + + Falls back to a single chunk when the document has no usable split points. + """ + container = document.container + children = [child for child in container if isinstance(child.tag, str)] + + split_indexes = [index for index, child in enumerate(children) if is_split_point(child)] + if len(split_indexes) < 2: + logger.info( + "Document has no usable split points, rendering in one pass", + extra={"split_points": len(split_indexes), "children": len(children)}, + ) + return [document.to_html()] + + groups = _group_children(children, set(split_indexes), pages_per_chunk) + if len(groups) < 2: + logger.info("Document fits into a single chunk", extra={"children": len(children)}) + return [document.to_html()] + + prefix, suffix = _skeleton(document, container) + child_html = [lxml_html.tostring(child, encoding="unicode") for child in children] + + chunks = ["".join((prefix, *(child_html[index] for index in group), suffix)) for group in groups] + logger.info( + "Document split into chunks", + extra={"chunks": len(chunks), "children": len(children), "pages_per_chunk": pages_per_chunk}, + ) + return chunks + + +def _group_children(children, split_indexes: set[int], pages_per_chunk: int) -> list[list[int]]: + groups: list[list[int]] = [] + current: list[int] = [] + current_pages = 0.0 + + for index, child in enumerate(children): + starts_chunk = index in split_indexes and current and current_pages >= pages_per_chunk + if starts_chunk: + groups.append(current) + current = [] + current_pages = 0.0 + + current.append(index) + current_pages += estimate_pages(child) + + if current: + groups.append(current) + + return groups + + +def _skeleton(document: SourceDocument, container) -> tuple[str, str]: + """Opening and closing markup shared by every chunk. + + The whole ancestor chain is recreated with its attributes so CSS selectors + that depend on it keep matching inside a chunk. + """ + head_html = lxml_html.tostring(document.head, encoding="unicode") + + chain = [] + node = container + while node is not None and node is not document.tree: + chain.append(node) + node = node.getparent() + chain.reverse() + + opens = [_open_tag(document.tree)] + opens.append(head_html) + closes = [""] + + for node in chain: + opens.append(_open_tag(node)) + closes.append(f"") + + closes.reverse() + return "\n" + "".join(opens), "".join(closes) + + +def _open_tag(element) -> str: + attributes = "".join( + f' {name}="{_escape(value)}"' for name, value in element.attrib.items() + ) + return f"<{element.tag}{attributes}>" + + +def _escape(value: str) -> str: + return ( + value.replace("&", "&") + .replace('"', """) + .replace("<", "<") + .replace(">", ">") + ) diff --git a/app/pdf/document.py b/app/pdf/document.py new file mode 100644 index 0000000..3237c33 --- /dev/null +++ b/app/pdf/document.py @@ -0,0 +1,186 @@ +"""Parsing of the source document, heading extraction and table of contents.""" + +from __future__ import annotations + +import logging +import re +from dataclasses import dataclass + +from lxml import html as lxml_html + +logger = logging.getLogger(__name__) + +HEADING_TAGS = ("h1", "h2", "h3", "h4", "h5", "h6") +ID_SAFE = re.compile(r"[^a-zA-Z0-9_-]+") + +TOC_PLACEHOLDER = "\u2007\u2007\u2007" # figure spaces, keeps the pass 1 layout stable + + +@dataclass +class Heading: + level: int + text: str + anchor: str + page: int | None = None + + +class SourceDocument: + """Thin wrapper over the parsed document with the operations we need.""" + + def __init__(self, html: str) -> None: + self.tree = lxml_html.document_fromstring(html) + self.head = self.tree.find("head") + if self.head is None: + self.head = lxml_html.Element("head") + self.tree.insert(0, self.head) + self.body = self.tree.find("body") + if self.body is None: + raise ValueError("Dokument neobsahuje element body.") + self.container = self._find_container() + self._toc_css_added = False + + def _find_container(self): + """Element whose children are the natural split points. + + Many documents wrap everything in a single
or
. Splitting the + body would then produce a single chunk, so we descend through up to two + single child wrappers. + """ + container = self.body + for _ in range(2): + children = [child for child in container if isinstance(child.tag, str)] + if len(children) == 1 and len(list(children[0])) > 1: + container = children[0] + continue + break + return container + + # -- styles --------------------------------------------------------- + def append_stylesheet(self, css: str) -> None: + """Append a stylesheet as the last element of head so it wins on ties.""" + style = lxml_html.Element("style") + style.set("type", "text/css") + style.text = css + self.head.append(style) + + def set_base_url(self, base_url: str | None) -> None: + if not base_url or self.head.find("base") is not None: + return + base = lxml_html.Element("base") + base.set("href", base_url) + self.head.insert(0, base) + + # -- headings ------------------------------------------------------- + def collect_headings(self, max_depth: int) -> list[Heading]: + """Assign ids to headings that lack one and return them in document order.""" + headings: list[Heading] = [] + used: set[str] = {el.get("id") for el in self.tree.iter() if el.get("id")} + + for element in self.body.iter(*HEADING_TAGS): + level = int(element.tag[1]) + if level > max_depth: + continue + + text = " ".join(element.text_content().split()) + if not text: + continue + + anchor = element.get("id") + if not anchor: + anchor = self._unique_anchor(text, used) + element.set("id", anchor) + used.add(anchor) + + headings.append(Heading(level=level, text=text, anchor=anchor)) + + return headings + + @staticmethod + def _unique_anchor(text: str, used: set[str]) -> str: + base = ID_SAFE.sub("-", text.strip().lower()).strip("-") or "nadpis" + base = f"htp-{base[:60]}" + candidate = base + counter = 2 + while candidate in used: + candidate = f"{base}-{counter}" + counter += 1 + return candidate + + # -- table of contents ---------------------------------------------- + def insert_toc(self, headings: list[Heading], title: str, pages: dict[str, int] | None) -> None: + """Insert the table of contents as the first block of the body. + + Pass 1 uses a fixed width placeholder instead of the page number so the + table keeps exactly the same layout in pass 2. + """ + container = lxml_html.Element("nav") + container.set("id", "htp-toc") + container.set("class", "htp-toc") + + heading = lxml_html.Element("h1") + heading.set("class", "htp-toc-title") + heading.text = title + container.append(heading) + + table = lxml_html.Element("table") + table.set("class", "htp-toc-table") + tbody = lxml_html.Element("tbody") + + for item in headings: + if item.anchor == "htp-toc-title": + continue + row = lxml_html.Element("tr") + row.set("class", f"htp-toc-level-{item.level}") + + label_cell = lxml_html.Element("td") + label_cell.set("class", "htp-toc-label") + link = lxml_html.Element("a") + link.set("href", f"#{item.anchor}") + link.text = item.text + label_cell.append(link) + + page_cell = lxml_html.Element("td") + page_cell.set("class", "htp-toc-page") + if pages is None: + page_cell.text = TOC_PLACEHOLDER + else: + page_cell.text = str(pages.get(item.anchor, "")) or TOC_PLACEHOLDER + + row.append(label_cell) + row.append(page_cell) + tbody.append(row) + + table.append(tbody) + container.append(table) + + spacer = lxml_html.Element("div") + spacer.set("class", "htp-toc-break") + container.append(spacer) + + self.container.insert(0, container) + if not self._toc_css_added: + self.append_stylesheet(TOC_CSS) + self._toc_css_added = True + + def remove_toc(self) -> None: + existing = self.tree.find(".//nav[@id='htp-toc']") + if existing is not None: + existing.getparent().remove(existing) + + # -- serialization --------------------------------------------------- + def to_html(self) -> str: + return "\n" + lxml_html.tostring(self.tree, encoding="unicode") + + +TOC_CSS = """ +.htp-toc { break-after: page; } +.htp-toc-table { width: 100%; border-collapse: collapse; } +.htp-toc-table td { padding: 2pt 0; vertical-align: bottom; } +.htp-toc-label a { text-decoration: none; color: inherit; } +.htp-toc-page { text-align: right; width: 4em; font-variant-numeric: tabular-nums; white-space: nowrap; } +.htp-toc-level-2 .htp-toc-label { padding-left: 1.2em; } +.htp-toc-level-3 .htp-toc-label { padding-left: 2.4em; } +.htp-toc-level-4 .htp-toc-label { padding-left: 3.6em; } +.htp-toc-level-5 .htp-toc-label { padding-left: 4.8em; } +.htp-toc-level-6 .htp-toc-label { padding-left: 6em; } +""" diff --git a/app/pdf/merger.py b/app/pdf/merger.py new file mode 100644 index 0000000..e7edcd3 --- /dev/null +++ b/app/pdf/merger.py @@ -0,0 +1,53 @@ +"""Merging of chunk PDFs and reading of basic page geometry.""" + +from __future__ import annotations + +import logging +from pathlib import Path + +logger = logging.getLogger(__name__) + + +def merge(paths: list[Path], output_path: Path) -> int: + """Concatenate chunk PDFs into one file. + + PdfWriter.append is used on purpose, it carries over bookmarks and internal + links and shifts their page references by the running offset. + """ + from pypdf import PdfWriter + + if not paths: + raise ValueError("Neni co slucovat, seznam PDF je prazdny.") + + if len(paths) == 1: + paths[0].replace(output_path) + return page_count(output_path) + + writer = PdfWriter() + try: + for path in paths: + writer.append(str(path)) + with output_path.open("wb") as handle: + writer.write(handle) + finally: + writer.close() + + total = page_count(output_path) + logger.info("Chunks merged", extra={"chunks": len(paths), "pages": total}) + return total + + +def page_count(path: Path) -> int: + from pypdf import PdfReader + + with path.open("rb") as handle: + return len(PdfReader(handle).pages) + + +def first_page_size(path: Path) -> tuple[float, float]: + """Width and height of the first page in points.""" + from pypdf import PdfReader + + with path.open("rb") as handle: + box = PdfReader(handle).pages[0].mediabox + return float(box.width), float(box.height) diff --git a/app/pdf/paginator.py b/app/pdf/paginator.py new file mode 100644 index 0000000..d53ea83 --- /dev/null +++ b/app/pdf/paginator.py @@ -0,0 +1,80 @@ +"""Page numbering of the merged document. + +Two ways to number pages: + +css + CSS counters do the work during the render. Correct and cheap, but it only + works when the whole document is rendered in one pass, because each chunk + restarts the page counter. + +overlay + A transparent numbering layer with the same page size is rendered once and + merged onto the finished PDF. This is the only option for a chunked document + and for Chromium, which has no usable page counters. +""" + +from __future__ import annotations + +import logging +from pathlib import Path + +from ..errors import EngineUnavailableError +from ..models import PageNumbers, PageSettings +from .styles import build_overlay_css + +logger = logging.getLogger(__name__) + + +def build_overlay( + total_pages: int, + width_pt: float, + height_pt: float, + page: PageSettings, + page_numbers: PageNumbers, + output_path: Path, +) -> Path: + """Render the numbering layer, one empty page per page of the document.""" + try: + from weasyprint import HTML + except ImportError as exc: + raise EngineUnavailableError( + "Cislovani stranek vyzaduje nainstalovany WeasyPrint, ktery kresli cislovaci vrstvu.", + ) from exc + + css = build_overlay_css(width_pt, height_pt, page, page_numbers, total_pages) + slots = '
' * total_pages + html = ( + "" + f"{slots}" + ) + + HTML(string=html).write_pdf(target=str(output_path)) + logger.info("Numbering overlay rendered", extra={"pages": total_pages}) + return output_path + + +def apply_overlay(document_path: Path, overlay_path: Path, output_path: Path) -> None: + """Stamp the numbering layer onto every page of the document.""" + from pypdf import PdfReader, PdfWriter + + writer = PdfWriter(clone_from=str(document_path)) + try: + with overlay_path.open("rb") as handle: + overlay = PdfReader(handle) + available = len(overlay.pages) + + if available < len(writer.pages): + logger.warning( + "Numbering overlay has fewer pages than the document, tail will stay unnumbered", + extra={"overlay_pages": available, "document_pages": len(writer.pages)}, + ) + + for index, page in enumerate(writer.pages): + if index >= available: + break + page.merge_page(overlay.pages[index]) + + with output_path.open("wb") as target: + writer.write(target) + finally: + writer.close() diff --git a/app/pdf/styles.py b/app/pdf/styles.py new file mode 100644 index 0000000..02275f8 --- /dev/null +++ b/app/pdf/styles.py @@ -0,0 +1,115 @@ +"""Generation of the page stylesheet injected into the source document.""" + +from __future__ import annotations + +import re + +from ..models import PageNumbers, PageSettings + +NAMED_SIZE = re.compile(r"^[A-Za-z][A-Za-z0-9]*$") + +MARGIN_BOXES = { + "top-left": "@top-left", + "top-center": "@top-center", + "top-right": "@top-right", + "bottom-left": "@bottom-left", + "bottom-center": "@bottom-center", + "bottom-right": "@bottom-right", +} + +PLACEHOLDER = re.compile(r"(\{page\}|\{pages\})") + + +def page_size_value(page: PageSettings) -> str: + fmt = page.format.strip() + if NAMED_SIZE.match(fmt): + return f"{fmt} {page.orientation}" + # Explicit dimensions already carry the orientation. + return fmt + + +def margin_shorthand(page: PageSettings) -> str: + m = page.margin + return f"{m.top} {m.right} {m.bottom} {m.left}" + + +def css_content_value(fmt: str, total_pages: int | None) -> str: + """Turn "{page} / {pages}" into a CSS content value. + + When total_pages is known the total is written as a literal, otherwise the + CSS counter(pages) is used. + """ + parts: list[str] = [] + for token in PLACEHOLDER.split(fmt): + if token == "{page}": + parts.append("counter(page)") + elif token == "{pages}": + parts.append(str(total_pages) if total_pages is not None else "counter(pages)") + elif token: + escaped = token.replace("\\", "\\\\").replace('"', '\\"') + parts.append(f'"{escaped}"') + return " ".join(parts) if parts else '""' + + +def build_page_css( + page: PageSettings, + page_numbers: PageNumbers | None = None, + total_pages: int | None = None, + outline: bool = True, +) -> str: + """Stylesheet applied on top of the document styles.""" + + rules = [ + "@page {", + f" size: {page_size_value(page)};", + f" margin: {margin_shorthand(page)};", + ] + + if page_numbers is not None and page_numbers.enabled: + box = MARGIN_BOXES[page_numbers.position] + rules.append(f" {box} {{") + rules.append(f" content: {css_content_value(page_numbers.format, total_pages)};") + rules.append(" font-size: 9pt;") + rules.append(" color: #444;") + rules.append(" }") + + rules.append("}") + + if not outline: + rules.append("h1, h2, h3, h4, h5, h6 { bookmark-level: none; }") + + return "\n".join(rules) + + +def build_overlay_css( + width_pt: float, + height_pt: float, + page: PageSettings, + page_numbers: PageNumbers, + total_pages: int, +) -> str: + """Stylesheet for the transparent numbering layer merged onto the final PDF. + + The size comes from the produced PDF itself, so the overlay always matches + even when the source document declares its own @page size. + """ + box = MARGIN_BOXES[page_numbers.position] + reset = "" + if page_numbers.start_at != 1: + reset = f"body {{ counter-reset: page {page_numbers.start_at - 1}; }}\n" + + return ( + f"@page {{\n" + f" size: {width_pt:.2f}pt {height_pt:.2f}pt;\n" + f" margin: {margin_shorthand(page)};\n" + f" {box} {{\n" + f" content: {css_content_value(page_numbers.format, total_pages)};\n" + f" font-size: 9pt;\n" + f" color: #444;\n" + f" }}\n" + f"}}\n" + f"{reset}" + f"body {{ margin: 0; }}\n" + f".pdf-page-slot {{ height: 1px; break-after: page; }}\n" + f".pdf-page-slot:last-child {{ break-after: auto; }}\n" + ) diff --git a/app/routers/__init__.py b/app/routers/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/app/routers/convert.py b/app/routers/convert.py new file mode 100644 index 0000000..57eb51a --- /dev/null +++ b/app/routers/convert.py @@ -0,0 +1,78 @@ +"""Synchronous conversion. + +Suitable for smaller documents. Anything that does not finish within +SYNC_TIMEOUT_SECONDS is cancelled and the caller is pointed at /jobs. +""" + +from __future__ import annotations + +import asyncio +import logging + +from fastapi import APIRouter, Response +from fastapi.responses import FileResponse +from starlette.background import BackgroundTask + +from ..deps import get_container +from ..errors import ConversionError, SyncTooLongError, error_from_code +from ..models import ConvertRequest + +logger = logging.getLogger(__name__) + +router = APIRouter(tags=["konverze"]) + + +@router.post( + "/convert", + summary="Synchronni prevod HTML na PDF", + response_class=Response, + responses={ + 200: {"content": {"application/pdf": {}}, "description": "Hotove PDF."}, + 413: {"description": "Konverze trvala dele nez limit, pouzijte POST /jobs."}, + }, +) +async def convert(request: ConvertRequest): + container = get_container() + settings = container.settings + manager = container.jobs + + job = manager.submit(request) + + try: + await asyncio.wait_for(job.done.wait(), timeout=settings.sync_timeout_seconds) + except asyncio.TimeoutError as exc: + manager.cancel(job.id) + logger.warning( + "Synchronous conversion exceeded its budget", + extra={"job_id": job.id, "limit_seconds": settings.sync_timeout_seconds}, + ) + raise SyncTooLongError( + "Konverze presahla limit pro synchronni pozadavek. Pouzijte asynchronni endpoint POST /jobs.", + {"limit_seconds": settings.sync_timeout_seconds}, + ) from exc + + if job.state.status != "done": + error = job.state.error + if error is not None: + raise error_from_code(error.error_code, error.message, error.detail) + raise ConversionError("Konverze skoncila ve stavu " + job.state.status + ".") + + path = container.storage.result_path(job.id) + filename = request.filename or "dokument.pdf" + + headers = { + "X-Page-Count": str(job.state.page_count or 0), + "X-Engine-Used": job.state.engine_used or "", + } + if job.state.missing_assets: + headers["X-Missing-Assets"] = str(len(job.state.missing_assets)) + if job.state.warnings: + headers["X-Warnings"] = str(len(job.state.warnings)) + + return FileResponse( + path, + media_type="application/pdf", + filename=filename, + headers=headers, + background=BackgroundTask(container.storage.discard, job.id), + ) diff --git a/app/routers/health.py b/app/routers/health.py new file mode 100644 index 0000000..79500e0 --- /dev/null +++ b/app/routers/health.py @@ -0,0 +1,47 @@ +"""Health and version endpoints required by AppFactory.""" + +from __future__ import annotations + +import logging + +from fastapi import APIRouter + +from ..deps import get_container +from ..models import HealthResponse + +logger = logging.getLogger(__name__) + +router = APIRouter(tags=["service"]) + + +@router.get("/health", response_model=HealthResponse, summary="Stav sluzby") +async def health() -> HealthResponse: + container = get_container() + + engines = {} + for name, engine in container.engines.items(): + try: + engines[name] = await engine.available() + except Exception as exc: + logger.error("Engine availability check failed", extra={"engine": name}, exc_info=exc) + engines[name] = False + + status = "ok" if any(engines.values()) else "degraded" + return HealthResponse( + status=status, + app=container.settings.app_name, + version=container.settings.app_version, + engines=engines, + queue=container.jobs.stats(), + ) + + +@router.get("/version", summary="Verze a zakladni informace o sluzbe") +async def version() -> dict: + settings = get_container().settings + return { + "app": settings.app_name, + "version": settings.app_version, + "language": "python", + "root_path": settings.root_path, + } diff --git a/app/routers/jobs.py b/app/routers/jobs.py new file mode 100644 index 0000000..e45c133 --- /dev/null +++ b/app/routers/jobs.py @@ -0,0 +1,90 @@ +"""Asynchronous conversion. This is the path for large documents.""" + +from __future__ import annotations + +import logging + +from fastapi import APIRouter, Response, status +from fastapi.responses import FileResponse + +from ..deps import get_container +from ..errors import ConversionError, JobNotFoundError +from ..models import ConvertRequest, JobAccepted, JobState + +logger = logging.getLogger(__name__) + +router = APIRouter(prefix="/jobs", tags=["joby"]) + + +def _result_url(job_id: str) -> str: + root = get_container().settings.root_path.rstrip("/") + return f"{root}/jobs/{job_id}/result" + + +@router.post( + "", + response_model=JobAccepted, + status_code=status.HTTP_202_ACCEPTED, + summary="Zaradi konverzi do fronty", +) +async def create_job(request: ConvertRequest) -> JobAccepted: + job = get_container().jobs.submit(request) + return JobAccepted( + job_id=job.id, + status=job.state.status, + created_at=job.state.created_at, + result_url=_result_url(job.id), + ) + + +@router.get("/{job_id}", response_model=JobState, summary="Stav jobu") +async def job_state(job_id: str) -> JobState: + job = get_container().jobs.get(job_id) + state = job.state.model_copy() + if state.status == "done": + state.result_url = _result_url(job_id) + return state + + +@router.get( + "/{job_id}/result", + summary="Stahne hotove PDF", + response_class=Response, + responses={ + 200: {"content": {"application/pdf": {}}, "description": "Hotove PDF."}, + 404: {"description": "Job neexistuje nebo uz expiroval."}, + 409: {"description": "Job jeste nedobehl nebo skoncil chybou."}, + }, +) +async def job_result(job_id: str): + container = get_container() + job = container.jobs.get(job_id) + + if job.state.status != "done": + error = ConversionError( + f"Vysledek neni k dispozici, job je ve stavu {job.state.status}.", + {"status": job.state.status}, + ) + error.error_code = "result_not_ready" + error.status_code = 409 + raise error + + path = container.storage.result_path(job_id) + if not path.exists(): + raise JobNotFoundError("Soubor s vysledkem uz neexistuje, job pravdepodobne expiroval.") + + filename = job.request.filename or "dokument.pdf" + return FileResponse( + path, + media_type="application/pdf", + filename=filename, + headers={ + "X-Page-Count": str(job.state.page_count or 0), + "X-Engine-Used": job.state.engine_used or "", + }, + ) + + +@router.delete("/{job_id}", response_model=JobState, summary="Zrusi job nebo smaze jeho vysledek") +async def delete_job(job_id: str) -> JobState: + return get_container().jobs.cancel(job_id).state diff --git a/app/services/__init__.py b/app/services/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/app/services/fetcher.py b/app/services/fetcher.py new file mode 100644 index 0000000..a172629 --- /dev/null +++ b/app/services/fetcher.py @@ -0,0 +1,181 @@ +"""Fetching of the source document and of its assets. + +Both paths go through UrlGuard. Asset failures never abort the render, they are +collected and reported back to the caller so nobody silently receives a PDF with +missing images. +""" + +from __future__ import annotations + +import io +import logging +from dataclasses import dataclass, field + +import httpx + +from ..config import Settings +from ..errors import LimitExceededError, SourceUnavailableError +from ..models import MissingAsset +from .security import UrlGuard + +logger = logging.getLogger(__name__) + +USER_AGENT = "html-to-pdf/1.0 (AppFactory)" + + +@dataclass +class FetchedDocument: + html: str + base_url: str + + +@dataclass +class AssetReport: + """Collects assets that could not be loaded during a render.""" + + missing: list[MissingAsset] = field(default_factory=list) + _seen: set[str] = field(default_factory=set) + + def add(self, url: str, reason: str) -> None: + if url in self._seen: + return + self._seen.add(url) + self.missing.append(MissingAsset(url=url, reason=reason)) + logger.warning("Asset could not be loaded", extra={"asset_url": url, "reason": reason}) + + +def fetch_document(url: str, guard: UrlGuard, settings: Settings) -> FetchedDocument: + """Download the source HTML, validating every redirect hop.""" + + current = url + timeout = httpx.Timeout(settings.fetch_timeout_seconds) + + with httpx.Client(follow_redirects=False, timeout=timeout, headers={"User-Agent": USER_AGENT}) as client: + for hop in range(settings.max_redirects + 1): + guard.check(current) + try: + response = client.get(current) + except httpx.HTTPError as exc: + raise SourceUnavailableError( + "Zdrojovy dokument se nepodarilo stahnout.", + {"url": current, "reason": str(exc)}, + ) from exc + + if response.is_redirect: + location = response.headers.get("location") + if not location: + raise SourceUnavailableError( + "Zdroj vratil presmerovani bez hlavicky Location.", {"url": current} + ) + current = str(response.url.join(location)) + logger.info("Following redirect", extra={"hop": hop + 1, "target": current}) + continue + + if response.status_code >= 400: + raise SourceUnavailableError( + f"Zdrojovy dokument vratil HTTP {response.status_code}.", + {"url": current, "status_code": response.status_code}, + ) + + _check_size(len(response.content), settings) + return FetchedDocument(html=response.text, base_url=str(response.url)) + + raise SourceUnavailableError( + "Prekrocen maximalni pocet presmerovani.", + {"url": url, "max_redirects": settings.max_redirects}, + ) + + +def _check_size(size: int, settings: Settings) -> None: + if settings.max_html_bytes and size > settings.max_html_bytes: + raise LimitExceededError( + "Zdrojovy dokument je vetsi nez nakonfigurovany limit MAX_HTML_BYTES.", + {"size_bytes": size, "limit_bytes": settings.max_html_bytes}, + ) + + +def build_url_fetcher(guard: UrlGuard, report: AssetReport, allow_remote: bool, timeout_seconds: int): + """Return a WeasyPrint url_fetcher with SSRF checks and failure reporting.""" + + from weasyprint import default_url_fetcher + + def fetcher(url: str, timeout: int = 10, ssl_context=None): # noqa: ARG001 + if url.startswith("data:"): + return default_url_fetcher(url) + + if not allow_remote: + report.add(url, "stahovani externich assetu je vypnute") + return _empty_asset() + + try: + guard.check(url) + except Exception as exc: # noqa: BLE001 - reported, never silent + report.add(url, f"zablokovano: {exc}") + return _empty_asset() + + try: + response = httpx.get( + url, + timeout=timeout_seconds, + follow_redirects=False, + headers={"User-Agent": USER_AGENT}, + ) + while response.is_redirect: + location = response.headers.get("location") + if not location: + raise httpx.HTTPError("presmerovani bez hlavicky Location") + target = str(response.url.join(location)) + guard.check(target) + response = httpx.get( + target, + timeout=timeout_seconds, + follow_redirects=False, + headers={"User-Agent": USER_AGENT}, + ) + + if response.status_code >= 400: + report.add(url, f"HTTP {response.status_code}") + return _empty_asset() + + return { + "string": response.content, + "mime_type": response.headers.get("content-type", "").split(";")[0] or None, + "redirected_url": str(response.url), + } + except Exception as exc: # noqa: BLE001 - reported, never silent + report.add(url, str(exc)) + return _empty_asset() + + return fetcher + + +def _empty_asset() -> dict: + """Placeholder returned instead of a failed asset so the render continues.""" + return {"file_obj": io.BytesIO(b""), "mime_type": "application/octet-stream"} + + +@dataclass +class AssetGate: + """Per job gateway for everything the render pulls from the network.""" + + guard: UrlGuard + report: AssetReport + allow_remote: bool = True + timeout_seconds: int = 10 + + def weasy_fetcher(self): + return build_url_fetcher(self.guard, self.report, self.allow_remote, self.timeout_seconds) + + def allowed(self, url: str) -> tuple[bool, str]: + """Decide whether a browser initiated request may proceed.""" + if url.startswith("data:") or url.startswith("blob:") or url.startswith("about:"): + return True, "" + if not url.startswith(("http://", "https://")): + return False, "nepovolene schema" + if not self.allow_remote: + return False, "stahovani externich assetu je vypnute" + try: + self.guard.check(url) + except Exception as exc: # noqa: BLE001 - reported, never silent + return False, str(exc) + return True, "" diff --git a/app/services/jobs.py b/app/services/jobs.py new file mode 100644 index 0000000..bfcac67 --- /dev/null +++ b/app/services/jobs.py @@ -0,0 +1,238 @@ +"""In memory job queue. + +The queue is deliberately in process. There is no broker and no database, which +means a restart loses queued work. Such jobs are marked as failed with an +explicit reason instead of silently disappearing. +""" + +from __future__ import annotations + +import asyncio +import logging +import uuid +from datetime import datetime, timedelta, timezone + +import httpx + +from ..config import Settings +from ..errors import ConversionError, JobNotFoundError, QueueFullError +from ..logging_setup import current_job_id +from ..models import ConvertRequest, ErrorInfo, JobProgress, JobState +from .pipeline import ConversionPipeline +from .storage import Storage + +logger = logging.getLogger(__name__) + + +def _now() -> datetime: + return datetime.now(timezone.utc) + + +class Job: + def __init__(self, job_id: str, request: ConvertRequest) -> None: + self.id = job_id + self.request = request + self.state = JobState(job_id=job_id, status="queued", created_at=_now()) + self.task = None + self.done = asyncio.Event() + + +class JobManager: + def __init__(self, pipeline: ConversionPipeline, storage: Storage, settings: Settings) -> None: + self._pipeline = pipeline + self._storage = storage + self._settings = settings + self._jobs = {} + self._queue = asyncio.Queue(maxsize=settings.queue_max_size) + self._workers = [] + self._cleaner = None + + async def start(self) -> None: + self._storage.sweep_orphans(set(self._jobs)) + for index in range(max(1, self._settings.workers)): + self._workers.append(asyncio.create_task(self._worker(index), name=f"htp-worker-{index}")) + self._cleaner = asyncio.create_task(self._cleanup_loop(), name="htp-cleanup") + logger.info("Job manager started", extra={"workers": len(self._workers)}) + + async def stop(self) -> None: + for task in self._workers: + task.cancel() + if self._cleaner is not None: + self._cleaner.cancel() + + for job in self._jobs.values(): + if job.state.status in ("queued", "running"): + self._fail( + job, + ErrorInfo( + error_code="service_restarted", + message="Sluzba byla ukoncena drive, nez job dobehl. Odeslete pozadavek znovu.", + ), + ) + logger.info("Job manager stopped") + + def submit(self, request: ConvertRequest) -> Job: + job = Job(str(uuid.uuid4()), request) + self._jobs[job.id] = job + try: + self._queue.put_nowait(job.id) + except asyncio.QueueFull as exc: + del self._jobs[job.id] + raise QueueFullError( + "Fronta je plna, zkuste to prosim za chvili.", + {"queue_max_size": self._settings.queue_max_size}, + ) from exc + + logger.info("Job queued", extra={"job_id": job.id, "queue_size": self._queue.qsize()}) + return job + + def get(self, job_id: str) -> Job: + job = self._jobs.get(job_id) + if job is None: + raise JobNotFoundError("Job s timto identifikatorem neexistuje nebo uz expiroval.") + return job + + def cancel(self, job_id: str) -> Job: + job = self.get(job_id) + if job.task is not None and not job.task.done(): + job.task.cancel() + else: + job.state.status = "cancelled" + job.state.finished_at = _now() + self._storage.discard(job.id) + job.done.set() + job.state.expires_at = _now() + timedelta(seconds=self._settings.job_result_ttl_seconds) + logger.info("Job cancelled", extra={"job_id": job_id}) + return job + + def stats(self) -> dict: + counts = {"queued": 0, "running": 0, "done": 0, "failed": 0, "cancelled": 0, "expired": 0} + for job in self._jobs.values(): + counts[job.state.status] = counts.get(job.state.status, 0) + 1 + counts["workers"] = len(self._workers) + return counts + + async def _worker(self, index: int) -> None: + while True: + job_id = await self._queue.get() + job = self._jobs.get(job_id) + if job is None or job.state.status != "queued": + self._queue.task_done() + continue + + job.task = asyncio.current_task() + token = current_job_id.set(job.id) + try: + await self._execute(job) + except asyncio.CancelledError: + job.state.status = "cancelled" + job.state.finished_at = _now() + job.state.expires_at = _now() + timedelta(seconds=self._settings.job_result_ttl_seconds) + self._storage.discard(job.id) + logger.info("Job execution cancelled", extra={"job_id": job.id}) + finally: + current_job_id.reset(token) + job.task = None + job.done.set() + self._queue.task_done() + await self._notify_callback(job) + + async def _execute(self, job: Job) -> None: + job.state.status = "running" + job.state.started_at = _now() + logger.info("Job started") + + def progress(pages, chunks_done, chunks_total, pass_number): + job.state.progress = JobProgress( + pages_rendered=pages, + chunks_done=chunks_done, + chunks_total=chunks_total, + pass_number=pass_number, + ) + + workdir = self._storage.job_dir(job.id) + try: + result = await self._pipeline.run(job.request, workdir, progress) + except ConversionError as exc: + logger.error("Job failed", extra={"error_code": exc.error_code}, exc_info=exc) + self._fail(job, ErrorInfo(error_code=exc.error_code, message=exc.message, detail=exc.detail or None)) + return + except asyncio.CancelledError: + raise + except Exception as exc: + logger.exception("Job failed with an unexpected error") + self._fail( + job, + ErrorInfo( + error_code="internal_error", + message="Pri generovani PDF doslo k neocekavane chybe.", + detail={"reason": str(exc)}, + ), + ) + return + + self._storage.publish(job.id, result.path) + job.state.status = "done" + job.state.finished_at = _now() + job.state.expires_at = _now() + timedelta(seconds=self._settings.job_result_ttl_seconds) + job.state.page_count = result.page_count + job.state.engine_used = result.engine_used + job.state.missing_assets = result.missing_assets + job.state.warnings = result.warnings + logger.info( + "Job finished", + extra={ + "pages": result.page_count, + "engine": result.engine_used, + "missing_assets": len(result.missing_assets), + }, + ) + + def _fail(self, job: Job, error: ErrorInfo) -> None: + job.state.status = "failed" + job.state.finished_at = _now() + job.state.expires_at = _now() + timedelta(seconds=self._settings.job_result_ttl_seconds) + job.state.error = error + self._storage.discard(job.id) + + async def _notify_callback(self, job: Job) -> None: + url = job.request.callback_url + if not url or job.state.status not in ("done", "failed"): + return + + payload = job.state.model_dump(mode="json") + for attempt in range(1, max(1, self._settings.callback_retries) + 1): + try: + async with httpx.AsyncClient(timeout=self._settings.callback_timeout_seconds) as client: + response = await client.post(url, json=payload) + if response.status_code < 400: + logger.info("Callback delivered", extra={"attempt": attempt}) + return + logger.warning( + "Callback returned an error status", + extra={"attempt": attempt, "status_code": response.status_code}, + ) + except Exception as exc: + logger.warning("Callback delivery failed", extra={"attempt": attempt}, exc_info=exc) + await asyncio.sleep(min(2 ** attempt, 10)) + + logger.error("Callback could not be delivered, job result stays available over the API") + + async def _cleanup_loop(self) -> None: + while True: + await asyncio.sleep(60) + try: + self._expire_old_jobs() + except Exception: + logger.exception("Cleanup loop failed") + + def _expire_old_jobs(self) -> None: + now = _now() + for job_id, job in list(self._jobs.items()): + expires_at = job.state.expires_at + if expires_at is None or expires_at > now: + continue + job.state.status = "expired" + self._storage.discard(job_id) + del self._jobs[job_id] + logger.info("Job result expired and was removed", extra={"job_id": job_id}) diff --git a/app/services/pipeline.py b/app/services/pipeline.py new file mode 100644 index 0000000..404fd18 --- /dev/null +++ b/app/services/pipeline.py @@ -0,0 +1,350 @@ +"""The conversion pipeline. + +Order of operations: + +1. load the source HTML, validating the target address +2. pick the engine +3. decide how page numbers will be produced +4. build the document, inject the page stylesheet, optionally insert the table + of contents placeholder +5. split into chunks +6. first render pass, which yields the real page count and anchor positions +7. second render pass when a table of contents needs real page numbers +8. merge the chunks +9. stamp the numbering overlay when CSS counters cannot be used +""" + +from __future__ import annotations + +import asyncio +import logging +import re +from dataclasses import dataclass, field +from pathlib import Path +from typing import Callable + +from ..config import Settings, get_settings +from ..errors import ( + ConversionError, + LimitExceededError, + RenderTimeoutError, + UnsupportedCombinationError, +) +from ..models import ConvertRequest, MissingAsset +from ..pdf import merger, paginator +from ..pdf.chunker import split_document +from ..pdf.document import SourceDocument +from ..pdf.styles import build_page_css +from .fetcher import AssetGate, AssetReport, fetch_document +from .security import UrlGuard + +logger = logging.getLogger(__name__) + +SCRIPT_TAG = re.compile(r"]*>(.*?)", re.IGNORECASE | re.DOTALL) +SCRIPT_SRC = re.compile(r"]*\bsrc\s*=", re.IGNORECASE) + +ProgressCallback = Callable[[int, int, int, int], None] + + +@dataclass +class ConversionResult: + path: Path + page_count: int + engine_used: str + missing_assets: list[MissingAsset] = field(default_factory=list) + warnings: list[str] = field(default_factory=list) + + +class ConversionPipeline: + def __init__(self, engines: dict, settings: Settings | None = None) -> None: + self._engines = engines + self._settings = settings or get_settings() + self._guard = UrlGuard(self._settings) + + async def run( + self, + request: ConvertRequest, + workdir: Path, + progress: ProgressCallback | None = None, + ) -> ConversionResult: + workdir.mkdir(parents=True, exist_ok=True) + timeout = self._settings.max_render_seconds + + coroutine = self._run_with_fallback(request, workdir, progress) + if timeout: + try: + return await asyncio.wait_for(coroutine, timeout=timeout) + except asyncio.TimeoutError as exc: + raise RenderTimeoutError( + "Render prekrocil nakonfigurovany limit MAX_RENDER_SECONDS.", + {"limit_seconds": timeout}, + ) from exc + return await coroutine + + async def _run_with_fallback( + self, request: ConvertRequest, workdir: Path, progress: ProgressCallback | None + ) -> ConversionResult: + raw_html, base_url = await self._load_source(request) + requested = request.engine + engine_name = self._select_engine(requested, raw_html) + + try: + return await self._execute(request, raw_html, base_url, engine_name, workdir, progress) + except ConversionError: + raise + except Exception as exc: + if requested != "auto" or engine_name != "weasyprint" or "chromium" not in self._engines: + raise + logger.warning( + "WeasyPrint render failed, falling back to Chromium", + exc_info=exc, + extra={"engine": engine_name}, + ) + result = await self._execute(request, raw_html, base_url, "chromium", workdir, progress) + result.warnings.append( + "Render pres WeasyPrint selhal, dokument byl vygenerovan pres Chromium." + ) + return result + + def _select_engine(self, requested: str, raw_html: str) -> str: + if requested != "auto": + if requested not in self._engines: + raise UnsupportedCombinationError( + f"Engine {requested} neni v teto instanci k dispozici.", + {"available": sorted(self._engines)}, + ) + return requested + + if self._has_active_scripts(raw_html) and "chromium" in self._engines: + logger.info("Auto engine selected chromium because the document contains scripts") + return "chromium" + + if "weasyprint" in self._engines: + return "weasyprint" + + return next(iter(self._engines)) + + @staticmethod + def _has_active_scripts(raw_html: str) -> bool: + if SCRIPT_SRC.search(raw_html): + return True + return any(body.strip() for body in SCRIPT_TAG.findall(raw_html)) + + async def _load_source(self, request: ConvertRequest) -> tuple: + if request.source.html is not None: + html = request.source.html + limit = self._settings.max_html_bytes + if limit and len(html.encode("utf-8")) > limit: + raise LimitExceededError( + "Zdrojove HTML je vetsi nez nakonfigurovany limit MAX_HTML_BYTES.", + {"limit_bytes": limit}, + ) + return html, request.source.base_url + + fetched = await asyncio.to_thread( + fetch_document, request.source.url, self._guard, self._settings + ) + return fetched.html, request.source.base_url or fetched.base_url + + async def _execute( + self, + request: ConvertRequest, + raw_html: str, + base_url: str | None, + engine_name: str, + workdir: Path, + progress: ProgressCallback | None, + ) -> ConversionResult: + engine = self._engines[engine_name] + report = AssetReport() + gate = AssetGate( + guard=self._guard, + report=report, + allow_remote=request.assets.allow_remote, + timeout_seconds=request.assets.timeout_seconds, + ) + warnings: list[str] = [] + + numbering_mode = self._numbering_mode(request, engine_name) + + if request.toc.enabled and not getattr(engine, "supports_anchor_pages", False): + raise UnsupportedCombinationError( + "Generovani obsahu s cisly stranek podporuje pouze engine weasyprint.", + {"engine": engine_name}, + ) + + document = SourceDocument(raw_html) + document.set_base_url(base_url) + document.append_stylesheet( + build_page_css( + request.page, + request.page_numbers if numbering_mode == "css" else None, + total_pages=None, + outline=request.outline, + ) + ) + + headings = document.collect_headings(request.toc.depth) if request.toc.enabled else [] + if request.toc.enabled: + if not headings: + warnings.append("Dokument neobsahuje zadne nadpisy, obsah nebyl vygenerovan.") + request.toc.enabled = False + else: + document.insert_toc(headings, request.toc.title, pages=None) + + chunks = self._split(document, request) + direct_navigation = self._can_navigate_directly(request, engine_name, chunks) + if direct_navigation: + logger.info("Chromium will navigate to the source URL directly so its scripts run in context") + + renders = await self._render_pass( + engine, chunks, base_url, request, workdir, gate, progress, 1, direct_navigation + ) + + if request.toc.enabled: + anchor_pages = self._absolute_anchor_pages(renders) + missing = [item.anchor for item in headings if item.anchor not in anchor_pages] + if missing: + logger.warning( + "Some headings have no anchor position, their page numbers stay empty", + extra={"missing_anchors": len(missing)}, + ) + document.remove_toc() + document.insert_toc(headings, request.toc.title, pages=anchor_pages) + chunks = self._split(document, request) + renders = await self._render_pass( + engine, chunks, base_url, request, workdir, gate, progress, 2 + ) + + if len(chunks) > 1: + warnings.append( + "Dokument byl rozdelen na casti, odkazy v obsahu proto nejsou klikatelne. " + "Cisla stranek jsou spravna." + ) + + merged_path = workdir / "merged.pdf" + total_pages = merger.merge([item.path for item in renders], merged_path) + self._check_page_limit(total_pages) + + final_path = merged_path + if request.page_numbers.enabled and numbering_mode == "overlay": + final_path = await asyncio.to_thread( + self._stamp_numbers, merged_path, workdir, request, total_pages + ) + + on_job_finished = getattr(engine, "on_job_finished", None) + if callable(on_job_finished): + on_job_finished() + + return ConversionResult( + path=final_path, + page_count=total_pages, + engine_used=engine_name, + missing_assets=report.missing, + warnings=warnings, + ) + + @staticmethod + def _can_navigate_directly(request: ConvertRequest, engine_name: str, chunks: list) -> bool: + """Chromium renders a foreign page best when it loads the URL itself. + + Only possible when the document is not split and needs no injected + markup, otherwise the modified HTML has to be pushed into the page. + """ + return ( + engine_name == "chromium" + and request.source.url is not None + and not request.toc.enabled + and len(chunks) == 1 + ) + + def _numbering_mode(self, request: ConvertRequest, engine_name: str) -> str: + mode = request.page_numbers.mode + if mode == "auto": + if engine_name == "weasyprint" and not request.chunking.enabled: + return "css" + return "overlay" + + if mode == "css" and engine_name != "weasyprint": + raise UnsupportedCombinationError( + "Rezim cislovani css funguje pouze s enginem weasyprint. Pouzijte overlay nebo auto.", + {"engine": engine_name}, + ) + if mode == "css" and request.chunking.enabled: + raise UnsupportedCombinationError( + "Rezim cislovani css nelze kombinovat s chunkovanim, protoze citac stranek se v kazde " + "casti restartuje. Vypnete chunking nebo pouzijte overlay.", + ) + return mode + + def _split(self, document: SourceDocument, request: ConvertRequest) -> list: + if not request.chunking.enabled: + return [document.to_html()] + return split_document(document, request.chunking.pages_per_chunk) + + async def _render_pass( + self, + engine, + chunks: list, + base_url: str | None, + request: ConvertRequest, + workdir: Path, + gate: AssetGate, + progress: ProgressCallback | None, + pass_number: int, + direct_navigation: bool = False, + ) -> list: + renders = [] + pages_rendered = 0 + + for index, chunk_html in enumerate(chunks): + output = workdir / f"pass{pass_number}-chunk{index:04d}.pdf" + render = await engine.render_chunk( + None if direct_navigation else chunk_html, + base_url, + request, + output, + total_pages=None, + asset_gate=gate, + ) + renders.append(render) + pages_rendered += render.page_count + + if progress is not None: + progress(pages_rendered, index + 1, len(chunks), pass_number) + + self._check_page_limit(pages_rendered) + + logger.info( + "Render pass finished", + extra={"pass_number": pass_number, "chunks": len(chunks), "pages": pages_rendered}, + ) + return renders + + @staticmethod + def _absolute_anchor_pages(renders: list) -> dict: + pages: dict = {} + offset = 0 + for render in renders: + for anchor, local_page in render.anchor_pages.items(): + pages.setdefault(anchor, offset + local_page + 1) + offset += render.page_count + return pages + + def _check_page_limit(self, pages: int) -> None: + if self._settings.max_pages and pages > self._settings.max_pages: + raise LimitExceededError( + "Dokument ma vice stranek nez nakonfigurovany limit MAX_PAGES.", + {"pages": pages, "limit": self._settings.max_pages}, + ) + + @staticmethod + def _stamp_numbers(merged_path: Path, workdir: Path, request: ConvertRequest, total_pages: int) -> Path: + width, height = merger.first_page_size(merged_path) + overlay_path = workdir / "overlay.pdf" + paginator.build_overlay( + total_pages, width, height, request.page, request.page_numbers, overlay_path + ) + numbered_path = workdir / "numbered.pdf" + paginator.apply_overlay(merged_path, overlay_path, numbered_path) + return numbered_path diff --git a/app/services/security.py b/app/services/security.py new file mode 100644 index 0000000..35496ef --- /dev/null +++ b/app/services/security.py @@ -0,0 +1,132 @@ +"""SSRF protection. + +The service fetches arbitrary URLs on request, which is exactly the shape of an +SSRF vulnerability. Every URL is validated after DNS resolution, not on the +string alone, and the check is repeated on every redirect hop. +""" + +from __future__ import annotations + +import ipaddress +import logging +import socket +from urllib.parse import urlparse + +from ..config import Settings +from ..errors import BlockedTargetError + +logger = logging.getLogger(__name__) + +ALLOWED_SCHEMES = ("http", "https") + +# Ranges that must never be reachable from a user supplied URL. +PRIVATE_NETWORKS = [ + ipaddress.ip_network("127.0.0.0/8"), + ipaddress.ip_network("10.0.0.0/8"), + ipaddress.ip_network("172.16.0.0/12"), + ipaddress.ip_network("192.168.0.0/16"), + ipaddress.ip_network("169.254.0.0/16"), + ipaddress.ip_network("0.0.0.0/8"), + ipaddress.ip_network("100.64.0.0/10"), + ipaddress.ip_network("192.0.0.0/24"), + ipaddress.ip_network("198.18.0.0/15"), + ipaddress.ip_network("224.0.0.0/4"), + ipaddress.ip_network("240.0.0.0/4"), + ipaddress.ip_network("::1/128"), + ipaddress.ip_network("fc00::/7"), + ipaddress.ip_network("fe80::/10"), + ipaddress.ip_network("::/128"), +] + + +class UrlGuard: + """Validates URLs against the configured policy.""" + + def __init__(self, settings: Settings) -> None: + self._block_private = settings.ssrf_block_private + self._allowed_hosts = {host.lower() for host in settings.ssrf_allowed_hosts} + self._extra_blocked: list[ipaddress._BaseNetwork] = [] + + for cidr in settings.ssrf_extra_blocked_cidrs: + try: + self._extra_blocked.append(ipaddress.ip_network(cidr, strict=False)) + except ValueError: + logger.warning("Ignoring invalid CIDR in SSRF_EXTRA_BLOCKED_CIDRS", extra={"cidr": cidr}) + + def check(self, url: str) -> str: + """Raise BlockedTargetError when the URL must not be fetched. + + Returns the hostname so callers can reuse it without parsing again. + """ + parsed = urlparse(url) + scheme = (parsed.scheme or "").lower() + + if scheme not in ALLOWED_SCHEMES: + raise BlockedTargetError( + "Povolena jsou pouze schemata http a https.", + {"url": url, "scheme": scheme or None}, + ) + + host = parsed.hostname + if not host: + raise BlockedTargetError("Adresa neobsahuje hostname.", {"url": url}) + + if host.lower() in self._allowed_hosts: + logger.info("Host explicitly allowlisted", extra={"host": host}) + return host + + for address in self._resolve(host, url): + self._check_address(address, host, url) + + return host + + def _resolve(self, host: str, url: str) -> list[ipaddress.IPv4Address | ipaddress.IPv6Address]: + # A literal IP address needs no lookup. + try: + return [ipaddress.ip_address(host)] + except ValueError: + pass + + try: + infos = socket.getaddrinfo(host, None, proto=socket.IPPROTO_TCP) + except socket.gaierror as exc: + raise BlockedTargetError( + f"Hostname {host} se nepodarilo prelozit na IP adresu.", + {"url": url, "reason": str(exc)}, + ) from exc + + addresses = [] + for info in infos: + try: + addresses.append(ipaddress.ip_address(info[4][0])) + except ValueError: + continue + + if not addresses: + raise BlockedTargetError(f"Hostname {host} nema zadnou pouzitelnou IP adresu.", {"url": url}) + + return addresses + + def _check_address(self, address, host: str, url: str) -> None: + if self._block_private: + for network in PRIVATE_NETWORKS: + if address.version == network.version and address in network: + logger.warning( + "Blocked request to private address", + extra={"host": host, "address": str(address), "network": str(network)}, + ) + raise BlockedTargetError( + "Cilova adresa smeruje do privatniho nebo vyhrazeneho rozsahu a je zablokovana.", + {"url": url, "host": host, "address": str(address)}, + ) + + for network in self._extra_blocked: + if address.version == network.version and address in network: + logger.warning( + "Blocked request by configured CIDR", + extra={"host": host, "address": str(address), "network": str(network)}, + ) + raise BlockedTargetError( + "Cilova adresa je v konfigurovanem seznamu blokovanych rozsahu.", + {"url": url, "host": host, "address": str(address)}, + ) diff --git a/app/services/storage.py b/app/services/storage.py new file mode 100644 index 0000000..a8ef75c --- /dev/null +++ b/app/services/storage.py @@ -0,0 +1,65 @@ +"""Temporary storage of job working directories and results. + +Nothing is kept longer than needed. A result lives until it is picked up or +until its TTL expires, whichever comes first. +""" + +from __future__ import annotations + +import logging +import shutil +from pathlib import Path + +logger = logging.getLogger(__name__) + +RESULT_NAME = "result.pdf" + + +class Storage: + def __init__(self, root: str) -> None: + self.root = Path(root) + self.root.mkdir(parents=True, exist_ok=True) + + def job_dir(self, job_id: str) -> Path: + path = self.root / job_id + path.mkdir(parents=True, exist_ok=True) + return path + + def result_path(self, job_id: str) -> Path: + return self.root / job_id / RESULT_NAME + + def publish(self, job_id: str, produced: Path) -> Path: + """Move the produced file to its final name and drop the intermediates.""" + target = self.result_path(job_id) + if produced != target: + produced.replace(target) + + for item in self.job_dir(job_id).iterdir(): + if item.name == RESULT_NAME: + continue + self._remove(item) + + return target + + def discard(self, job_id: str) -> None: + self._remove(self.root / job_id) + + def _remove(self, path: Path) -> None: + try: + if path.is_dir(): + shutil.rmtree(path, ignore_errors=False) + elif path.exists(): + path.unlink() + except OSError as exc: + logger.warning("Could not remove temporary path", extra={"path": str(path)}, exc_info=exc) + + def sweep_orphans(self, known_job_ids: set[str]) -> int: + """Remove directories that belong to no known job, for example after a restart.""" + removed = 0 + for item in self.root.iterdir(): + if item.is_dir() and item.name not in known_job_ids: + self._remove(item) + removed += 1 + if removed: + logger.info("Removed orphaned job directories", extra={"count": removed}) + return removed diff --git a/documentation/README.md b/documentation/README.md new file mode 100644 index 0000000..ead78fb --- /dev/null +++ b/documentation/README.md @@ -0,0 +1,55 @@ +# Dokumentace sluzby html-to-pdf + +Sluzba prevadi HTML dokument na PDF. Prijme adresu dokumentu nebo HTML primo +v tele requestu a vrati soubor PDF. Zvlada dokumenty o tisicich stranek. + +## Obsah dokumentace + +- [api.md](api.md) popis endpointu a tel requestu +- [architektura.md](architektura.md) jak sluzba funguje uvnitr +- [konfigurace.md](konfigurace.md) environment variables +- [provoz.md](provoz.md) nasazeni, Docker, znama omezeni +- [zmeny.md](zmeny.md) zaznam zmen + +## Aktualni stav + +Verze 1.0.0, stav development. + +Hotovo: + +- synchronni endpoint POST /convert +- asynchronni endpoint POST /jobs se sledovanim stavu a stahovanim vysledku +- dva render enginy, weasyprint a chromium, plus rezim auto +- chunkovani velkych dokumentu a slucovani vysledku +- dvoupruchodovy render obsahu se skutecnymi cisly stranek +- cislovani stranek pres CSS countery nebo pres cislovaci vrstvu +- SSRF ochrana s kontrolou po DNS resolvu a na kazdem presmerovani +- hlaseni nedostupnych assetu v odpovedi jobu +- fronta s omezenym poctem paralelnich workeru +- volitelny callback po dokonceni jobu +- strukturovane JSON logovani s job_id + +Neni hotovo a neni ani v zadani: + +- autentizace. Sluzba je bez overovani, pristup resi reverse proxy. + Pokud ma byt chranena, je potreba se domluvit na zpusobu, typicky hlavicka + X-Api-Key. Do te doby zadna neni. +- trvala fronta. Restart sluzby znamena ztratu rozpracovanych jobu, ty se + oznaci jako failed s duvodem service_restarted. + +## Rychly priklad + +```bash +curl -X POST https://services.csbot.cz/apps/html-to-pdf/convert \ + -H "Content-Type: application/json" \ + -d '{"source": {"url": "https://example.com/dokument.html"}}' \ + --output dokument.pdf +``` + +Pro velky dokument se pouziva asynchronni cesta: + +```bash +curl -X POST https://services.csbot.cz/apps/html-to-pdf/jobs \ + -H "Content-Type: application/json" \ + -d '{"source": {"url": "https://example.com/velky.html"}, "toc": {"enabled": true}}' +``` diff --git a/documentation/api.md b/documentation/api.md new file mode 100644 index 0000000..9496753 --- /dev/null +++ b/documentation/api.md @@ -0,0 +1,245 @@ +# API + +Vsechny cesty jsou uvedene relativne. Verejne se volaji s prefixem +`/apps/html-to-pdf`, ktery Caddy pred predanim do containeru odstranuje. + +Odpovedi jsou JSON, vyjimkou je stazeni hotoveho PDF. + +## POST /convert + +Synchronni prevod. Vraci primo `application/pdf`. + +Urceno pro mensi dokumenty. Pokud konverze presahne `SYNC_TIMEOUT_SECONDS` +(vychozi 60 s), job se zrusi a sluzba vrati HTTP 413 s odkazem na asynchronni +endpoint. Zruseni se loguje, nikdy nezmizi potichu. + +Hlavicky odpovedi: + +| Hlavicka | Vyznam | +|---|---| +| `X-Page-Count` | pocet stranek vysledku | +| `X-Engine-Used` | engine, ktery dokument vyrenderoval | +| `X-Missing-Assets` | pocet assetu, ktere se nepodarilo nacist | +| `X-Warnings` | pocet varovani | + +Pokud je `X-Missing-Assets` nenulovy, v PDF neco chybi. Detaily jsou dostupne +jen u asynchronni cesty, kde se vraci cely seznam. + +## POST /jobs + +Zaradi konverzi do fronty. Vraci HTTP 202. + +```json +{ + "job_id": "1f0e...", + "status": "queued", + "created_at": "2026-08-27T10:00:00Z", + "result_url": "/apps/html-to-pdf/jobs/1f0e.../result" +} +``` + +## GET /jobs/{job_id} + +Stav jobu. + +```json +{ + "job_id": "1f0e...", + "status": "running", + "created_at": "2026-08-27T10:00:00Z", + "started_at": "2026-08-27T10:00:01Z", + "finished_at": null, + "expires_at": null, + "progress": { + "pages_rendered": 420, + "chunks_done": 9, + "chunks_total": 21, + "pass_number": 1 + }, + "engine_used": null, + "page_count": null, + "missing_assets": [], + "warnings": [], + "error": null, + "result_url": null +} +``` + +Stavy: `queued`, `running`, `done`, `failed`, `cancelled`, `expired`. + +Pole `pass_number` rozlisuje prvni a druhy pruchod. Druhy pruchod nastava jen +tehdy, kdyz se generuje obsah se skutecnymi cisly stranek, takze u takoveho +dokumentu ukazatel postupu probehne dvakrat. + +## GET /jobs/{job_id}/result + +Stahne hotove PDF. Soubor se streamuje, nenacita se cely do pameti. + +- 404 job neexistuje nebo uz expiroval +- 409 job jeste nedobehl nebo skoncil chybou + +## DELETE /jobs/{job_id} + +Zrusi bezici job nebo smaze hotovy vysledek. + +## GET /health + +Stav sluzby, verze, dostupnost enginu a stav fronty. U Chromia se dostupnost +overuje skutecnym nastartovanim prohlizece. + +`status` je `ok`, pokud je k dispozici alespon jeden engine, jinak `degraded`. + +## GET /version + +Nazev aplikace, verze a aktualni `root_path`. + +## GET /docs + +Swagger UI. OpenAPI dokument obsahuje `servers` s prefixem `/apps/html-to-pdf`, +takze tlacitko Try it out vola spravnou verejnou cestu. + +## Telo requestu + +Stejne pro `/convert` i `/jobs`. Povinne je pouze `source`, vsechno ostatni ma +pouzitelnou vychozi hodnotu. + +```json +{ + "source": { + "url": "https://example.com/dokument.html", + "html": null, + "base_url": null + }, + "engine": "auto", + "page": { + "format": "A4", + "orientation": "portrait", + "margin": { "top": "20mm", "right": "15mm", "bottom": "20mm", "left": "15mm" } + }, + "page_numbers": { + "enabled": false, + "format": "{page} / {pages}", + "position": "bottom-center", + "mode": "auto", + "start_at": 1 + }, + "toc": { "enabled": false, "depth": 3, "title": "Obsah" }, + "outline": true, + "pdf_profile": null, + "assets": { "allow_remote": true, "timeout_seconds": 10 }, + "chunking": { "enabled": true, "pages_per_chunk": 50 }, + "wait_for": { "state": "load", "selector": null, "timeout_seconds": 30 }, + "filename": null, + "callback_url": null +} +``` + +### source + +Vyplnene musi byt prave jedno z poli `url` a `html`, jinak sluzba vraci 422. +`base_url` slouzi k rozpadu relativnich cest a pouziva se hlavne spolu s `html`. +Pri pouziti `url` se `base_url` odvodi z finalni adresy po presmerovanich. + +### engine + +- `weasyprint` spravne strankovani, nizka pametova narocnost, nespousti JavaScript +- `chromium` zvladne i dokumenty dokreslovane JavaScriptem +- `auto` zvoli chromium, pokud dokument obsahuje aktivni skripty, jinak + weasyprint. Pokud render pres weasyprint selze, sluzba to zaloguje a zopakuje + ho pres chromium, coz se objevi ve `warnings`. + +### page.format + +Nazev formatu (`A4`, `A5`, `Letter`) nebo explicitni rozmer (`210mm 297mm`). +U nazvu se uplatni i `orientation`, u explicitniho rozmeru je orientace dana +poradim hodnot. + +### page_numbers.mode + +- `css` cislovani resi CSS countery pri renderu. Nejhezci vysledek, ale funguje + jen kdyz se cely dokument renderuje najednou, protoze v kazde casti se citac + stranek restartuje. Vyzaduje `engine: weasyprint` a `chunking.enabled: false`, + jinak sluzba vraci 400 s kodem `unsupported_combination`. +- `overlay` cisla se dopisi do hotoveho PDF jako pruhledna vrstva. Jedina + moznost u clenenych dokumentu a u Chromia. +- `auto` zvoli `css` u necleneneho dokumentu renderovaneho WeasyPrintem, + jinak `overlay`. + +Cislovaci vrstvu kresli WeasyPrint i tehdy, kdyz dokument vyrenderovalo +Chromium. Bez nainstalovaneho WeasyPrintu proto cislovani stranek nefunguje. + +### toc + +Generuje obsah s odkazy a skutecnymi cisly stranek. Vyzaduje +`engine: weasyprint`, protoze Chromium neumi rict, na ktere strance nadpis +skoncil. Pri jine kombinaci sluzba vraci 400. + +Nadpisy bez atributu `id` ho dostanou automaticky. + +Pokud je dokument rozdelen na casti, odkazy v obsahu nejsou klikatelne, protoze +cil lezi v jine casti. Cisla stranek jsou spravna. Sluzba na to upozorni ve +`warnings`. + +### outline + +Zalozky PDF generovane z nadpisu h1 az h6. Vypnuti se resi CSS pravidlem +`bookmark-level: none`, takze funguje i uvnitr jednotlivych casti. + +### pdf_profile + +Predava se WeasyPrintu jako `pdf_variant`. Povolene hodnoty: `pdf/a-1b`, +`pdf/a-2b`, `pdf/a-3b`, `pdf/a-4b`, `pdf/ua-1`. + +### assets + +`allow_remote: false` zakaze stahovani externich assetu. Nedostupne assety +render nezastavi, ale objevi se v `missing_assets` a v logu. + +### chunking + +`pages_per_chunk` je cilova velikost casti. Skutecny pocet stranek je znamy az +po renderu, deleni proto vychazi z odhadu podle mnozstvi textu, obrazku a radku +tabulek. Rez vznika vzdy jen mezi primymi potomky kontejneru, takze tabulka ani +odstavec se nikdy nerozdeli. + +Delici body jsou elementy `section`, `article`, `h1`, elementy s atributem +`data-chunk` a elementy se stylem obsahujicim `page-break-before` nebo +`break-before`. Pokud dokument zadny takovy nema, renderuje se vcelku a sluzba +to zaloguje. + +### wait_for + +Pouziva jen Chromium. `state` je `load`, `domcontentloaded` nebo `networkidle`, +`selector` navic ceka na konkretni element. + +### callback_url + +Po dokonceni nebo selhani jobu na nej sluzba posle POST se stejnym telem, jake +vraci `GET /jobs/{job_id}`. Neuspech se loguje a nekolikrat zopakuje, job se +kvuli nemu neoznaci jako failed. + +## Chyby + +```json +{ + "error_code": "blocked_target", + "message": "Cilova adresa smeruje do privatniho nebo vyhrazeneho rozsahu a je zablokovana.", + "detail": { "url": "http://127.0.0.1/a.html", "host": "127.0.0.1" } +} +``` + +| error_code | HTTP | Kdy nastane | +|---|---|---| +| `invalid_request` | 400 | chybny vstup | +| `blocked_target` | 400 | adresa smeruje do zakazaneho rozsahu nebo ma nepovolene schema | +| `unsupported_combination` | 400 | nepodporovana kombinace parametru, napriklad obsah s Chromiem | +| `limit_exceeded` | 400 | prekrocen nakonfigurovany limit | +| `source_unavailable` | 502 | zdrojovy dokument se nepodarilo stahnout | +| `render_timeout` | 504 | render presahl casovy limit | +| `sync_too_long` | 413 | synchronni konverze presahla limit, pouzijte POST /jobs | +| `engine_unavailable` | 503 | pozadovany engine neni k dispozici | +| `queue_full` | 503 | fronta je plna | +| `job_not_found` | 404 | job neexistuje nebo expiroval | +| `result_not_ready` | 409 | job jeste nedobehl | +| `service_restarted` | v tele jobu | sluzba se restartovala drive, nez job dobehl | +| `internal_error` | 500 | neocekavana chyba | diff --git a/documentation/architektura.md b/documentation/architektura.md new file mode 100644 index 0000000..d35cae9 --- /dev/null +++ b/documentation/architektura.md @@ -0,0 +1,115 @@ +# Architektura + +## Struktura projektu + +```text +app/ + main.py vytvoreni aplikace, lifespan, obsluha chyb + config.py konfigurace z environment variables + models.py Pydantic modely requestu a odpovedi + errors.py typove chyby a mapovani na HTTP kody + logging_setup.py strukturovane JSON logovani + deps.py kontejner sluzeb sestaveny pri startu + routers/ + health.py /health a /version + convert.py synchronni /convert + jobs.py asynchronni /jobs + services/ + security.py SSRF kontroly, resolv adres + fetcher.py stahovani dokumentu a assetu, hlaseni vypadku + pipeline.py cely prubeh konverze + jobs.py fronta jobu a workeri + storage.py docasne ulozeni vysledku + engines/ + base.py spolecne rozhrani enginu + weasy.py WeasyPrint + chromium.py Playwright a headless Chromium + pdf/ + document.py parsovani, nadpisy, obsah + chunker.py deleni HTML na casti + merger.py slucovani PDF + paginator.py cislovaci vrstva + styles.py generovani stylu stranky +``` + +## Prubeh konverze + +1. Nacteni zdroje. U `source.url` se dokument stahne, kazdy hop presmerovani + projde SSRF kontrolou. U `source.html` se pouzije telo requestu. +2. Volba enginu. Rezim `auto` hleda aktivni skripty, jinak bere WeasyPrint. +3. Volba rezimu cislovani stranek. Musi padnout pred injektazi stylu, protoze + styl se lisi podle toho, jestli cisla resi CSS countery nebo cislovaci vrstva. +4. Sestaveni dokumentu. Doplni se `base`, prida se styl stranky, nadpisy dostanou + `id` a pokud je zapnuty obsah, vlozi se jeho zastupna verze. +5. Deleni na casti. +6. Prvni pruchod renderu. Vysledkem jsou hotova PDF casti, jejich pocty stranek + a pozice kotev. +7. Druhy pruchod, pokud se generuje obsah. Zastupna cisla se nahradi skutecnymi. +8. Slouceni casti. +9. Nastampovani cislovaci vrstvy, pokud cisla neresi CSS countery. + +## Proc dvoupruchodovy render + +Cisla stranek v obsahu nejsou znama drive, nez se dokument vyrenderuje. Zaroven +plati, ze vlozeni obsahu posune cisla stranek, ktera obsah uvadi. + +Reseni je vlozit obsah uz v prvnim pruchodu, jen s pevne sirokou vyplni misto +cisel. Rozvrzeni je proto v obou pruchodech stejne a cisla zjistena v prvnim +pruchodu plati i po druhem. Cislo stranky je v tabulce zarovnane doprava ve +sloupci pevne sirky, takze ani jiny pocet cislic rozvrzeni nezmeni. + +Pozice nadpisu se ctou z kotev, ktere WeasyPrint hlasi u kazde stranky. To je +presnejsi nez odhad z poradi zalozek. + +## Proc se cisla stranek u clenenych dokumentu dopisuji az nakonec + +CSS counter `page` se v kazde renderovane casti restartuje od jednicky a +`counter(pages)` zna jen pocet stranek dane casti. Ve sloucenem dokumentu by +proto cislovani bylo nesmyslne. + +Cislovaci vrstva je samostatny dokument o stejnem poctu stranek a stejnem +rozmeru, ktery obsahuje jen cisla v okraji. Ten se pres hotove PDF nastampuje +stranku po strance. Rozmer stranky se cte z prvni stranky vysledneho PDF, takze +vrstva sedi i kdyz si dokument nastavil vlastni `@page size`. + +Fonty vrstvy kresli WeasyPrint a vklada je do souboru, vysledek proto nezavisi +na fontech v systemu, ktery PDF otevira. + +## Pamet + +Kriticky bod u tisicistrankovych dokumentu. + +- Render probiha po castech, v pameti je vzdy jen jedna cast. +- Chromium se restartuje po `CHROMIUM_RESTART_AFTER_JOBS` jobech, protoze + postupne unika pamet. +- Slucovani pouziva pypdf, ktere drzi stranky vysledku v pameti. U velmi velkych + dokumentu je to nejnarocnejsi krok cele konverze. V image je nainstalovany + i `qpdf`, ktery by slo pouzit jako nahradu, pokud by pamet prestala stacit. +- Vysledek se ke klientovi streamuje, nenacita se cely do pameti. + +## Fronta + +Fronta je v pameti procesu. Zadna databaze, zadny broker. + +- Soubezne bezi `WORKERS` jobu, ostatni cekaji ve fronte. +- Plna fronta vraci 503 s kodem `queue_full`. +- Pri ukonceni sluzby se rozpracovane joby oznaci jako failed s kodem + `service_restarted`. Nezmizi potichu. +- Vysledky se po `JOB_RESULT_TTL_SECONDS` smazou a job se odstrani. +- Pri startu se smazou adresare jobu, ktere po restartu zustaly bez zaznamu. + +## Bezpecnost + +Sluzba na pozadani stahuje libovolnou adresu, coz je presne tvar SSRF +zranitelnosti. + +- Povolena jsou jen schemata `http` a `https`. +- Kontrola probiha az po DNS resolvu, takze verejna domena mirici na 127.0.0.1 + neprojde. +- Kontrola se opakuje na kazdem presmerovani. +- Blokovane jsou loopback, privatni rozsahy, link local vcetne 169.254.169.254, + CGNAT, multicast a rezervovane rozsahy. +- U Chromia jde kazdy pozadavek prohlizece pres `context.route`, takze stejnou + kontrolou projdou i assety, ktere si stranka dotahne sama. +- Seznam blokovanych rozsahu jde rozsirit i zuzit konfiguraci. Vychozi stav je + blokovat. diff --git a/documentation/konfigurace.md b/documentation/konfigurace.md new file mode 100644 index 0000000..cd8f976 --- /dev/null +++ b/documentation/konfigurace.md @@ -0,0 +1,73 @@ +# Konfigurace + +Vsechno se cte z environment variables. AppFactory je predava pres vygenerovany +runtime `.env`. Zadna promenna neni povinna, sluzba nastartuje i bez nich. + +## Aplikace + +| Promenna | Vychozi | Vyznam | +|---|---|---| +| `APP_NAME` | `html-to-pdf` | nazev v dokumentaci a v odpovedi /version | +| `APP_VERSION` | `1.0.0` | verze | +| `ROOT_PATH` | prazdne | prefix reverse proxy, napriklad `/apps/html-to-pdf` | +| `BASE_PATH` | prazdne | pouzije se, kdyz `ROOT_PATH` neni nastavene | +| `LOG_LEVEL` | `INFO` | uroven logovani | + +## Fronta a joby + +| Promenna | Vychozi | Vyznam | +|---|---|---| +| `WORKERS` | `2` | pocet soubezne bezicich konverzi | +| `QUEUE_MAX_SIZE` | `100` | kapacita fronty, pri prekroceni se vraci 503 | +| `SYNC_TIMEOUT_SECONDS` | `60` | limit pro POST /convert | +| `JOB_RESULT_TTL_SECONDS` | `3600` | jak dlouho je vysledek k dispozici ke stazeni | +| `STORAGE_DIR` | `/tmp/html-to-pdf` | adresar pro docasne soubory | + +## Enginy + +| Promenna | Vychozi | Vyznam | +|---|---|---| +| `DEFAULT_ENGINE` | `auto` | engine pouzity, kdyz ho request neuvede | +| `CHROMIUM_ENABLED` | `true` | vypnuti Chromia usetri pamet, ale ztrati podporu JavaScriptu | +| `CHROMIUM_RESTART_AFTER_JOBS` | `50` | po kolika jobech se prohlizec restartuje | + +## Sit + +| Promenna | Vychozi | Vyznam | +|---|---|---| +| `FETCH_TIMEOUT_SECONDS` | `30` | timeout stazeni zdrojoveho dokumentu | +| `ASSET_TIMEOUT_SECONDS` | `10` | vychozi timeout stazeni jednoho assetu | +| `MAX_REDIRECTS` | `5` | maximalni pocet presmerovani | + +## SSRF ochrana + +| Promenna | Vychozi | Vyznam | +|---|---|---| +| `SSRF_BLOCK_PRIVATE` | `true` | blokovat loopback, privatni a rezervovane rozsahy | +| `SSRF_EXTRA_BLOCKED_CIDRS` | prazdne | dalsi blokovane rozsahy, oddelene carkou | +| `SSRF_ALLOWED_HOSTS` | prazdne | hostnames, ktere kontrolou neprochazi | + +`SSRF_ALLOWED_HOSTS` je urcene pro vyjimky typu interniho generatoru HTML ve +stejne siti. Kazdy zaznam obchazi celou kontrolu, pouzivat opatrne. + +Vypnuti `SSRF_BLOCK_PRIVATE` otevre sluzbe cestu do cele vnitrni site. Delat +jen tam, kde to ma duvod. + +## Limity + +Vsechny limity jsou ve vychozim stavu vypnute. Nula znamena bez limitu. +Pri prekroceni se vraci explicitni chyba `limit_exceeded`, dokument se nikdy +tise neorezava. + +| Promenna | Vychozi | Vyznam | +|---|---|---| +| `MAX_PAGES` | `0` | maximalni pocet stranek vysledku | +| `MAX_HTML_BYTES` | `0` | maximalni velikost zdrojoveho HTML | +| `MAX_RENDER_SECONDS` | `0` | maximalni doba jednoho renderu | + +## Callback + +| Promenna | Vychozi | Vyznam | +|---|---|---| +| `CALLBACK_TIMEOUT_SECONDS` | `15` | timeout jednoho pokusu | +| `CALLBACK_RETRIES` | `3` | pocet pokusu o doruceni | diff --git a/documentation/provoz.md b/documentation/provoz.md new file mode 100644 index 0000000..f1b9dbe --- /dev/null +++ b/documentation/provoz.md @@ -0,0 +1,100 @@ +# Provoz + +## Docker image + +Image je dvoufazovy. Prvni faze stavi zavislosti, do vysledneho image se +prekladace nedostanou. + +Runtime obsahuje: + +- systemove knihovny WeasyPrintu: cairo, pango, gdk-pixbuf, harfbuzz +- Chromium nainstalovany pres `playwright install --with-deps chromium` +- fonty DejaVu, Liberation a Noto vcetne ceske diakritiky +- `qpdf` jako zaloha pro praci s PDF +- `curl` pro healthcheck + +Image je velky, radove jednotky GB. To je u sluzby, ktera v sobe ma cely +prohlizec a dve renderovaci knihovny, ocekavane. + +## Port + +Sluzba posloucha na `0.0.0.0:8000`, coz odpovida `app.yml`. Port se nemeni bez +odpovidajici upravy metadat aplikace v AppFactory. + +## Healthcheck + +Dockerfile ma `HEALTHCHECK`, ktery vola `/health` na `127.0.0.1:8000`. +Endpoint pri kazdem volani overuje dostupnost enginu vcetne skutecneho +nastartovani Chromia. + +## Sdilena pamet pro Chromium + +Chromium se spousti s `--disable-dev-shm-usage`, takze si vystaci s malym +`/dev/shm`. Pokud by se v logu objevovaly pady rendereru, je potreba containeru +zvysit `--shm-size`. + +## Overeni po nasazeni + +```bash +curl -i https://services.csbot.cz/apps/html-to-pdf/health +curl -i https://services.csbot.cz/apps/html-to-pdf/docs +``` + +Ve Swagger UI overit, ze Try it out vola adresy s prefixem +`/apps/html-to-pdf`, ne bez nej. + +Pozor na znamou vlastnost AppFactory: Caddy propousti GET pozadavky jen +z omezeneho seznamu IP adres. Try it out ve Swaggeru proto muze z bezneho +prohlizece vracet 403 jako `text/plain`, i kdyz je sluzba v poradku. Overovat +curlem primo ze serveru. + +## Logy + +Kazdy zaznam je jeden radek JSON. Vsechno, co patri k jednomu jobu, nese +`job_id`. + +Co stoji za sledovani: + +- `Asset could not be loaded` v PDF neco chybi +- `WeasyPrint render failed, falling back to Chromium` dokument neprosel + primarnim enginem +- `Restarting Chromium to release memory` bezna udrzba, ne chyba +- `Callback could not be delivered` job dobehl, ale klient se to nedozvedel +- `Job failed` konverze selhala, kod chyby je v poli `error_code` + +Secrets se do logu nezapisuji. + +## Znama omezeni + +- Fronta je v pameti procesu. Restart znamena ztratu rozpracovanych jobu, ty se + oznaci jako failed s kodem `service_restarted`. +- Obsah s cisly stranek umi jen WeasyPrint. Chromium neumi rict, na ktere + strance nadpis skoncil. +- U cleneneho dokumentu nejsou odkazy v obsahu klikatelne, protoze cil lezi + v jine casti souboru. Cisla stranek jsou spravna a sluzba na to upozorni ve + `warnings`. +- Slucovani pres pypdf drzi stranky vysledku v pameti. Je to nejnarocnejsi krok + konverze. +- WeasyPrint nespousti JavaScript. Pro dokumenty dokreslovane skripty je nutne + Chromium. + +## Testy + +Testy se spousti jen na vyslovne pozadani, nikdy automaticky. + +```bash +pip install -r requirements-dev.txt +pytest # bez pomalych testu spusti vse ostatni +pytest -m slow # jen mereni nad dokumentem o zhruba 1200 strankach +``` + +Testy, ktere potrebuji WeasyPrint, se same preskoci, pokud neni nainstalovany. + +Co je pokryte: + +- SSRF vcetne domeny mirici na 127.0.0.1 +- deleni dokumentu, ktery se delit da, i toho, ktery se delit neda +- spravnost cisel stranek po slouceni +- obsah ukazuje na skutecne stranky +- nedostupny asset render nezastavi a objevi se v odpovedi +- zruseni beziciho jobu, plna fronta, chovani pri ukonceni sluzby diff --git a/documentation/zmeny.md b/documentation/zmeny.md new file mode 100644 index 0000000..f1315b8 --- /dev/null +++ b/documentation/zmeny.md @@ -0,0 +1,34 @@ +# Zaznam zmen + +## 1.0.0 + +Prvni implementace sluzby. + +Pridano: + +- synchronni endpoint `POST /convert` s limitem a odkazem na asynchronni cestu +- asynchronni endpointy `POST /jobs`, `GET /jobs/{id}`, + `GET /jobs/{id}/result`, `DELETE /jobs/{id}` +- `GET /health` s overenim dostupnosti obou enginu a stavem fronty +- engine WeasyPrint jako vychozi, engine Chromium pres Playwright, rezim `auto` + s detekci skriptu a naslednym fallbackem +- deleni dokumentu na casti na strukturalnich hranicich a slucovani vysledku +- dvoupruchodovy render obsahu se skutecnymi cisly stranek +- cislovani stranek pres CSS countery nebo pres cislovaci vrstvu +- zalozky PDF z nadpisu, vypinatelne pres `outline` +- profily PDF/A a PDF/UA pres `pdf_profile` +- SSRF ochrana s kontrolou po DNS resolvu, na kazdem presmerovani a u vsech + pozadavku prohlizece +- hlaseni nedostupnych assetu v odpovedi jobu a v logu +- fronta s omezenym poctem workeru, zruseni jobu, expirace vysledku +- volitelny callback po dokonceni jobu +- strukturovane JSON logovani s `job_id` +- Dockerfile s WeasyPrintem, Chromiem, fonty s ceskou diakritikou a healthcheckem +- testy vcetne fixture o zhruba 1200 strankach + +Zamerne neimplementovano: + +- autentizace, zpusob neni domluveny +- databaze, Redis ani message broker +- webove UI +- jakekoliv limity zapnute ve vychozim stavu diff --git a/pytest.ini b/pytest.ini new file mode 100644 index 0000000..309e3ad --- /dev/null +++ b/pytest.ini @@ -0,0 +1,6 @@ +[pytest] +asyncio_mode = auto +pythonpath = . +testpaths = tests +markers = + slow: dlouho bezici testy nad velkym dokumentem diff --git a/requirements-dev.txt b/requirements-dev.txt new file mode 100644 index 0000000..5c74beb --- /dev/null +++ b/requirements-dev.txt @@ -0,0 +1,3 @@ +-r requirements.txt +pytest~=8.3 +pytest-asyncio~=0.24 diff --git a/requirements.txt b/requirements.txt index 364e2ee..219e366 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,2 +1,8 @@ -fastapi -uvicorn[standard] +fastapi~=0.115 +uvicorn[standard]~=0.34 +pydantic~=2.9 +httpx~=0.28 +lxml~=5.3 +weasyprint~=63.1 +pypdf~=5.1 +playwright~=1.49 diff --git a/tests/__init__.py b/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/tests/conftest.py b/tests/conftest.py new file mode 100644 index 0000000..67477b7 --- /dev/null +++ b/tests/conftest.py @@ -0,0 +1,46 @@ +"""Shared fixtures. + +Nothing here starts a real browser. Engine level tests skip themselves when the +engine is not installed in the current environment. +""" + +from __future__ import annotations + +import pytest + + +def build_document(sections: int, paragraphs_per_section: int = 12) -> str: + """Synthetic document with predictable structure and size.""" + parts = [ + "", + "", + "", + ] + for index in range(sections): + parts.append(f"

Kapitola {index + 1}

") + for paragraph in range(paragraphs_per_section): + parts.append( + f"

Odstavec {paragraph + 1} kapitoly {index + 1}. " + + ("Text s ceskou diakritikou pro overeni fontu. " * 8) + + "

" + ) + parts.append("
") + parts.append("") + return "".join(parts) + + +@pytest.fixture +def small_document() -> str: + return build_document(sections=3, paragraphs_per_section=2) + + +@pytest.fixture +def large_document() -> str: + """Roughly twelve hundred pages, used for memory and timing measurements.""" + return build_document(sections=400, paragraphs_per_section=14) + + +@pytest.fixture +def unsplittable_document() -> str: + body = "".join(f"

Odstavec {index}. " + ("Text. " * 40) + "

" for index in range(200)) + return f"{body}" diff --git a/tests/test_api.py b/tests/test_api.py new file mode 100644 index 0000000..f2d0c5e --- /dev/null +++ b/tests/test_api.py @@ -0,0 +1,69 @@ +"""HTTP surface of the service.""" + +from __future__ import annotations + +import pytest +from fastapi.testclient import TestClient + +from app.main import app + + +@pytest.fixture +def client(): + with TestClient(app) as test_client: + yield test_client + + +def test_health_is_available(client) -> None: + response = client.get("/health") + + assert response.status_code == 200 + body = response.json() + assert body["status"] in ("ok", "degraded") + assert "engines" in body + + +def test_version_reports_the_application(client) -> None: + response = client.get("/version") + + assert response.status_code == 200 + assert response.json()["language"] == "python" + + +def test_openapi_is_generated(client) -> None: + response = client.get("/openapi.json") + + assert response.status_code == 200 + paths = response.json()["paths"] + assert "/convert" in paths + assert "/jobs" in paths + assert "/jobs/{job_id}/result" in paths + + +def test_source_requires_exactly_one_input(client) -> None: + response = client.post( + "/convert", + json={"source": {"url": "https://example.com/a.html", "html": ""}}, + ) + + assert response.status_code == 422 + + +def test_empty_source_is_rejected(client) -> None: + response = client.post("/convert", json={"source": {}}) + + assert response.status_code == 422 + + +def test_private_address_is_refused(client) -> None: + response = client.post("/convert", json={"source": {"url": "http://127.0.0.1:8000/a.html"}}) + + assert response.status_code == 400 + assert response.json()["error_code"] == "blocked_target" + + +def test_unknown_job_returns_404(client) -> None: + response = client.get("/jobs/00000000-0000-0000-0000-000000000000") + + assert response.status_code == 404 + assert response.json()["error_code"] == "job_not_found" diff --git a/tests/test_assets.py b/tests/test_assets.py new file mode 100644 index 0000000..84f211f --- /dev/null +++ b/tests/test_assets.py @@ -0,0 +1,42 @@ +"""Behaviour when an asset cannot be loaded. + +A missing image must not abort the render, but it must never disappear +silently either. +""" + +from __future__ import annotations + +import pytest + +pytest.importorskip("weasyprint") + +from app.config import Settings # noqa: E402 +from app.engines.weasy import WeasyPrintEngine # noqa: E402 +from app.models import ConvertRequest, Source # noqa: E402 +from app.services.pipeline import ConversionPipeline # noqa: E402 + +pytestmark = pytest.mark.asyncio + +HTML_WITH_BLOCKED_IMAGE = """ + + +

Dokument s nedostupnym obrazkem

+

Text pokracuje i kdyz se obrazek nenacte.

+ chybi + +""" + + +async def test_blocked_asset_is_reported_and_render_continues(tmp_path) -> None: + pipeline = ConversionPipeline({"weasyprint": WeasyPrintEngine()}, Settings()) + request = ConvertRequest( + source=Source(html=HTML_WITH_BLOCKED_IMAGE), + engine="weasyprint", + ) + + result = await pipeline.run(request, tmp_path) + + assert result.page_count >= 1 + assert result.path.exists() + assert result.missing_assets, "nedostupny asset musi byt hlaseny v odpovedi" + assert "127.0.0.1" in result.missing_assets[0].url diff --git a/tests/test_chunker.py b/tests/test_chunker.py new file mode 100644 index 0000000..47adb19 --- /dev/null +++ b/tests/test_chunker.py @@ -0,0 +1,56 @@ +"""Splitting of the document into chunks.""" + +from __future__ import annotations + +from app.pdf.chunker import split_document +from app.pdf.document import SourceDocument +from tests.conftest import build_document + + +def test_document_with_sections_is_split() -> None: + document = SourceDocument(build_document(sections=40, paragraphs_per_section=10)) + chunks = split_document(document, pages_per_chunk=5) + + assert len(chunks) > 1 + for chunk in chunks: + assert chunk.startswith("") + assert "" in chunk + + +def test_every_section_appears_exactly_once() -> None: + document = SourceDocument(build_document(sections=30, paragraphs_per_section=8)) + chunks = split_document(document, pages_per_chunk=3) + + joined = "".join(chunks) + for index in range(30): + assert joined.count(f"id=\"sekce-{index}\"") == 1 + + +def test_head_is_repeated_in_every_chunk() -> None: + document = SourceDocument(build_document(sections=20, paragraphs_per_section=10)) + chunks = split_document(document, pages_per_chunk=2) + + assert len(chunks) > 1 + for chunk in chunks: + assert "font-family:sans-serif" in chunk + + +def test_document_without_split_points_stays_whole(unsplittable_document: str) -> None: + document = SourceDocument(unsplittable_document) + chunks = split_document(document, pages_per_chunk=1) + + assert len(chunks) == 1 + + +def test_single_wrapper_is_traversed() -> None: + sections = "".join(f"

Kapitola {i}

Text

" for i in range(10)) + html = f"
{sections}
" + + document = SourceDocument(html) + assert document.container.tag == "main" + + chunks = split_document(document, pages_per_chunk=1) + assert len(chunks) > 1 + for chunk in chunks: + assert "class=\"obal\"" in chunk diff --git a/tests/test_jobs.py b/tests/test_jobs.py new file mode 100644 index 0000000..4c7df9b --- /dev/null +++ b/tests/test_jobs.py @@ -0,0 +1,133 @@ +"""Job queue behaviour.""" + +from __future__ import annotations + +import asyncio +from pathlib import Path + +import pytest + +from app.config import Settings +from app.errors import LimitExceededError, QueueFullError +from app.models import ConvertRequest, Source +from app.services.jobs import JobManager +from app.services.pipeline import ConversionResult +from app.services.storage import Storage + +pytestmark = pytest.mark.asyncio + + +class StubPipeline: + def __init__(self, behaviour="ok", delay=0.0) -> None: + self.behaviour = behaviour + self.delay = delay + + async def run(self, request, workdir: Path, progress=None) -> ConversionResult: + if progress is not None: + progress(1, 1, 1, 1) + if self.delay: + await asyncio.sleep(self.delay) + if self.behaviour == "conversion_error": + raise LimitExceededError("Prekrocen limit stranek.", {"limit": 10}) + if self.behaviour == "crash": + raise RuntimeError("necekana chyba enginu") + + workdir.mkdir(parents=True, exist_ok=True) + produced = workdir / "out.pdf" + produced.write_bytes(b"%PDF-1.7\n%fake\n") + return ConversionResult(path=produced, page_count=1, engine_used="stub") + + +def make_manager(tmp_path, behaviour="ok", delay=0.0, **overrides) -> JobManager: + settings = Settings() + for key, value in overrides.items(): + object.__setattr__(settings, key, value) + object.__setattr__(settings, "storage_dir", str(tmp_path)) + + storage = Storage(str(tmp_path)) + return JobManager(StubPipeline(behaviour, delay), storage, settings) + + +def simple_request() -> ConvertRequest: + return ConvertRequest(source=Source(html="

ahoj

")) + + +async def test_successful_job_publishes_a_result(tmp_path) -> None: + manager = make_manager(tmp_path) + await manager.start() + try: + job = manager.submit(simple_request()) + await asyncio.wait_for(job.done.wait(), timeout=5) + + assert job.state.status == "done" + assert job.state.page_count == 1 + assert (Path(tmp_path) / job.id / "result.pdf").exists() + finally: + await manager.stop() + + +async def test_conversion_error_keeps_its_code(tmp_path) -> None: + manager = make_manager(tmp_path, behaviour="conversion_error") + await manager.start() + try: + job = manager.submit(simple_request()) + await asyncio.wait_for(job.done.wait(), timeout=5) + + assert job.state.status == "failed" + assert job.state.error is not None + assert job.state.error.error_code == "limit_exceeded" + finally: + await manager.stop() + + +async def test_unexpected_error_is_reported_not_swallowed(tmp_path) -> None: + manager = make_manager(tmp_path, behaviour="crash") + await manager.start() + try: + job = manager.submit(simple_request()) + await asyncio.wait_for(job.done.wait(), timeout=5) + + assert job.state.status == "failed" + assert job.state.error.error_code == "internal_error" + finally: + await manager.stop() + + +async def test_running_job_can_be_cancelled(tmp_path) -> None: + manager = make_manager(tmp_path, delay=5.0) + await manager.start() + try: + job = manager.submit(simple_request()) + await asyncio.sleep(0.2) + manager.cancel(job.id) + await asyncio.wait_for(job.done.wait(), timeout=5) + + assert job.state.status == "cancelled" + assert not (Path(tmp_path) / job.id).exists() + finally: + await manager.stop() + + +async def test_full_queue_is_rejected(tmp_path) -> None: + manager = make_manager(tmp_path, delay=5.0, queue_max_size=1, workers=1) + await manager.start() + try: + manager.submit(simple_request()) + await asyncio.sleep(0.1) + manager.submit(simple_request()) + + with pytest.raises(QueueFullError): + manager.submit(simple_request()) + finally: + await manager.stop() + + +async def test_shutdown_marks_pending_jobs_as_failed(tmp_path) -> None: + manager = make_manager(tmp_path, delay=5.0, workers=1) + await manager.start() + job = manager.submit(simple_request()) + await asyncio.sleep(0.1) + await manager.stop() + + assert job.state.status == "failed" + assert job.state.error.error_code == "service_restarted" diff --git a/tests/test_large_document.py b/tests/test_large_document.py new file mode 100644 index 0000000..43baf0f --- /dev/null +++ b/tests/test_large_document.py @@ -0,0 +1,39 @@ +"""Measurement fixture for a document of roughly twelve hundred pages. + +Marked slow on purpose. It is here to measure memory and wall clock time of the +chunked pipeline, not to assert exact numbers. +""" + +from __future__ import annotations + +import time + +import pytest + +pytest.importorskip("weasyprint") + +from app.config import Settings # noqa: E402 +from app.engines.weasy import WeasyPrintEngine # noqa: E402 +from app.models import ChunkSettings, ConvertRequest, PageNumbers, Source # noqa: E402 +from app.services.pipeline import ConversionPipeline # noqa: E402 + +pytestmark = [pytest.mark.asyncio, pytest.mark.slow] + + +async def test_large_document_renders_in_chunks(tmp_path, large_document: str) -> None: + pipeline = ConversionPipeline({"weasyprint": WeasyPrintEngine()}, Settings()) + request = ConvertRequest( + source=Source(html=large_document), + engine="weasyprint", + page_numbers=PageNumbers(enabled=True), + chunking=ChunkSettings(enabled=True, pages_per_chunk=50), + ) + + started = time.monotonic() + result = await pipeline.run(request, tmp_path) + duration = time.monotonic() - started + + print(f"stranek: {result.page_count}, cas: {duration:.1f} s") + + assert result.page_count > 1000 + assert result.path.exists() diff --git a/tests/test_pagination.py b/tests/test_pagination.py new file mode 100644 index 0000000..58ba205 --- /dev/null +++ b/tests/test_pagination.py @@ -0,0 +1,90 @@ +"""Page numbering and table of contents of a merged document. + +This is where the chunking approach usually breaks, so the numbers are read back +out of the produced PDF instead of being trusted. +""" + +from __future__ import annotations + +import re + +import pytest + +pytest.importorskip("weasyprint") +pytest.importorskip("pypdf") + +from app.config import Settings # noqa: E402 +from app.engines.weasy import WeasyPrintEngine # noqa: E402 +from app.models import ChunkSettings, ConvertRequest, PageNumbers, Source, TocSettings # noqa: E402 +from app.services.pipeline import ConversionPipeline # noqa: E402 +from tests.conftest import build_document # noqa: E402 + +pytestmark = pytest.mark.asyncio + + +def page_texts(path) -> list[str]: + from pypdf import PdfReader + + with open(path, "rb") as handle: + return [page.extract_text() or "" for page in PdfReader(handle).pages] + + +def make_pipeline() -> ConversionPipeline: + return ConversionPipeline({"weasyprint": WeasyPrintEngine()}, Settings()) + + +async def test_page_numbers_are_continuous_across_chunks(tmp_path) -> None: + request = ConvertRequest( + source=Source(html=build_document(sections=12, paragraphs_per_section=8)), + engine="weasyprint", + page_numbers=PageNumbers(enabled=True, format="{page} / {pages}"), + chunking=ChunkSettings(enabled=True, pages_per_chunk=2), + ) + + result = await make_pipeline().run(request, tmp_path) + texts = page_texts(result.path) + + assert result.page_count == len(texts) + assert result.page_count > 3, "dokument musi mit vic stranek, jinak test nic neoveruje" + + for index, text in enumerate(texts, start=1): + assert f"{index} / {result.page_count}" in text.replace("\n", " ") + + +async def test_table_of_contents_points_to_the_real_pages(tmp_path) -> None: + request = ConvertRequest( + source=Source(html=build_document(sections=10, paragraphs_per_section=8)), + engine="weasyprint", + toc=TocSettings(enabled=True, depth=1, title="Obsah"), + chunking=ChunkSettings(enabled=True, pages_per_chunk=2), + ) + + result = await make_pipeline().run(request, tmp_path) + texts = page_texts(result.path) + toc_text = " ".join(texts[:2]).replace("\n", " ") + + for index in range(10): + heading = f"Kapitola {index + 1}" + match = re.search(re.escape(heading) + r"\s+(\d+)", toc_text) + assert match, f"v obsahu chybi polozka {heading}" + + declared_page = int(match.group(1)) + actual_pages = [ + number for number, text in enumerate(texts, start=1) if heading in text.replace("\n", " ") + ] + # The first occurrence after the table of contents is the heading itself. + assert declared_page in actual_pages, f"{heading} deklaruje stranku {declared_page}" + + +async def test_unchunked_document_uses_css_counters(tmp_path) -> None: + request = ConvertRequest( + source=Source(html=build_document(sections=4, paragraphs_per_section=6)), + engine="weasyprint", + page_numbers=PageNumbers(enabled=True, format="{page} / {pages}"), + chunking=ChunkSettings(enabled=False), + ) + + result = await make_pipeline().run(request, tmp_path) + texts = page_texts(result.path) + + assert f"1 / {result.page_count}" in texts[0].replace("\n", " ") diff --git a/tests/test_security.py b/tests/test_security.py new file mode 100644 index 0000000..33405df --- /dev/null +++ b/tests/test_security.py @@ -0,0 +1,65 @@ +"""SSRF protection. + +The important case is a public hostname that resolves to a loopback address. +Checking the URL string alone would let it through. +""" + +from __future__ import annotations + +import socket + +import pytest + +from app.config import Settings +from app.errors import BlockedTargetError +from app.services.security import UrlGuard + + +@pytest.fixture +def guard() -> UrlGuard: + return UrlGuard(Settings()) + + +def test_literal_loopback_is_blocked(guard: UrlGuard) -> None: + with pytest.raises(BlockedTargetError): + guard.check("http://127.0.0.1:8000/dokument.html") + + +def test_private_range_is_blocked(guard: UrlGuard) -> None: + with pytest.raises(BlockedTargetError): + guard.check("http://192.168.1.10/dokument.html") + + +def test_link_local_metadata_endpoint_is_blocked(guard: UrlGuard) -> None: + with pytest.raises(BlockedTargetError): + guard.check("http://169.254.169.254/latest/meta-data/") + + +def test_hostname_resolving_to_loopback_is_blocked(guard: UrlGuard, monkeypatch) -> None: + def fake_getaddrinfo(host, *args, **kwargs): # noqa: ARG001 + return [(socket.AF_INET, socket.SOCK_STREAM, 6, "", ("127.0.0.1", 80))] + + monkeypatch.setattr(socket, "getaddrinfo", fake_getaddrinfo) + + with pytest.raises(BlockedTargetError): + guard.check("http://vlastni-domena.example.com/dokument.html") + + +def test_file_scheme_is_blocked(guard: UrlGuard) -> None: + with pytest.raises(BlockedTargetError): + guard.check("file:///etc/passwd") + + +def test_public_address_passes(guard: UrlGuard, monkeypatch) -> None: + def fake_getaddrinfo(host, *args, **kwargs): # noqa: ARG001 + return [(socket.AF_INET, socket.SOCK_STREAM, 6, "", ("93.184.216.34", 80))] + + monkeypatch.setattr(socket, "getaddrinfo", fake_getaddrinfo) + assert guard.check("https://example.com/dokument.html") == "example.com" + + +def test_allowlisted_host_skips_the_check() -> None: + settings = Settings() + object.__setattr__(settings, "ssrf_allowed_hosts", ["localhost"]) + guard = UrlGuard(settings) + assert guard.check("http://localhost:9000/dokument.html") == "localhost" diff --git a/tests/test_styles.py b/tests/test_styles.py new file mode 100644 index 0000000..866238a --- /dev/null +++ b/tests/test_styles.py @@ -0,0 +1,37 @@ +"""Building of the page stylesheet.""" + +from __future__ import annotations + +from app.models import PageNumbers, PageSettings +from app.pdf.styles import build_page_css, css_content_value, page_size_value + + +def test_named_format_keeps_orientation() -> None: + assert page_size_value(PageSettings(format="A4", orientation="landscape")) == "A4 landscape" + + +def test_explicit_dimensions_are_used_as_is() -> None: + assert page_size_value(PageSettings(format="210mm 297mm")) == "210mm 297mm" + + +def test_content_uses_counters_when_the_total_is_unknown() -> None: + assert css_content_value("{page} / {pages}", None) == 'counter(page) " / " counter(pages)' + + +def test_content_uses_a_literal_when_the_total_is_known() -> None: + assert css_content_value("Strana {page} z {pages}", 120) == '"Strana " counter(page) " z " "120"' + + +def test_page_numbers_are_absent_when_disabled() -> None: + css = build_page_css(PageSettings(), PageNumbers(enabled=False)) + assert "@bottom-center" not in css + + +def test_page_numbers_land_in_the_requested_box() -> None: + css = build_page_css(PageSettings(), PageNumbers(enabled=True, position="top-right")) + assert "@top-right" in css + + +def test_bookmarks_can_be_switched_off() -> None: + css = build_page_css(PageSettings(), None, outline=False) + assert "bookmark-level: none" in css