"""The conversion pipeline.
Order of operations:
1. load the source HTML, validating the target address
2. pick the engine
3. decide how page numbers will be produced
4. build the document, inject the page stylesheet, optionally insert the table
of contents placeholder
5. split into chunks
6. first render pass, which yields the real page count and anchor positions
7. second render pass when a table of contents needs real page numbers
8. merge the chunks
9. stamp the numbering overlay when CSS counters cannot be used
"""
from __future__ import annotations
import asyncio
import logging
import re
from dataclasses import dataclass, field
from pathlib import Path
from typing import Callable
from ..config import Settings, get_settings
from ..errors import (
ConversionError,
LimitExceededError,
RenderTimeoutError,
UnsupportedCombinationError,
)
from ..models import ConvertRequest, MissingAsset
from ..pdf import merger, paginator
from ..pdf.chunker import split_document
from ..pdf.document import SourceDocument
from ..pdf.styles import build_page_css
from .fetcher import AssetGate, AssetReport, fetch_document
from .security import UrlGuard
logger = logging.getLogger(__name__)
SCRIPT_TAG = re.compile(r"", re.IGNORECASE | re.DOTALL)
SCRIPT_SRC = re.compile(r"