"""How the OCR engine is set up for one deployment.

Separate from Settings because the engine is optional: a deployment that has not
installed PaddleOCR still starts, reads its CMS configuration, and reports
honestly that it cannot read documents — rather than failing at import with a
missing module and no explanation.
"""

from __future__ import annotations

import os
from dataclasses import dataclass
from pathlib import Path

#: The recognition model, named once.
#:
#: A module constant rather than a literal repeated in the field and again in
#: from_env(). The second copy is the one that decides, because every deployment
#: builds its configuration from the environment — and a default changed in the
#: field alone would look changed and behave exactly as before.
#:
#: `slots=True` also rules out reading it back off the class: OcrConfig.language
#: is the slot descriptor, not the value.
DEFAULT_LANGUAGE = "ur"

#: The detector, pinned rather than inherited from the language.
#:
#: Named here for the same reason as the language: from_env() would otherwise
#: hold the only copy that any real deployment reads.
DEFAULT_DETECTION_MODEL = "PP-OCRv5_mobile_det"

#: The alphabet to try when the first read cannot identify anybody.
#:
#: Empty disables the second pass entirely. See app/ocr/fallback.py for the
#: measurements this is worth.
DEFAULT_FALLBACK_LANGUAGE = "en"


@dataclass(frozen=True, slots=True)
class OcrConfig:
    """Everything the engine needs, and nothing about the rest of the platform."""

    language: str = DEFAULT_LANGUAGE
    """Recognition model. `ur` selects PaddleOCR's Arabic-script recogniser,
    which reads Urdu **and Latin** — it is not an Urdu-only setting.

    This was `en`, on the reasoning that Pakistani tax documents are
    overwhelmingly English and the fields extraction looks for (CNIC, IBAN,
    mobile, amounts) are Latin digits on every one of them. The second half of
    that is true. The first half was wrong about the documents that matter most.

    Measured on a real CNIC, both sides:

        FRONT   en: 248 chars    ur: 316 chars
        BACK    en:  91 chars    ur: 120 chars

    And the difference is not in the Urdu. The English model rendered the card's
    English headline as `A at ar` / `LAN`; the Arabic-script model read
    `PAKISTAN National Identity Card` and `ISLAMIC REPUBLIC OF PAKISTAN`
    correctly. Surrounding Urdu was making the English model misread the
    English — and "national identity card" is the strongest marker the
    classifier has, worth 3.0 on its own.

    The risk in switching was English-only documents, so that was measured too,
    on a rendered bank statement: **identical output**, 271 characters from
    both, confidence 0.997 against 0.996, all six expected markers found by
    each, both classified `bank_statement`.

    So this replaces the English model rather than joining it. One engine, no
    second pass, no extra memory — 608 MB peak was with both resident at once
    during the comparison, and only one is loaded now.

    Set TAXPILOT_OCR_LANGUAGE=en to go back, for a deployment whose documents
    really are English-only and who would rather have the dedicated model."""

    use_angle_classifier: bool = True
    """Detect and correct rotated text lines. Costs a little time per page and
    earns it back immediately: documents arrive as phone photographs taken at
    whatever angle the desk allowed."""

    correct_page_orientation: bool = True
    """Detect a page photographed sideways or upside down. A separate model from
    the line-level classifier above, and the one that matters more — a document
    rotated 90 degrees reads as nothing at all without it."""

    unwarp_pages: bool = False
    """Flatten a curved page. Off by default: it is another model to download and
    run, and it addresses photographs of bound books rather than the loose
    documents this actually receives."""

    enable_mkldnn: bool = False
    """oneDNN acceleration, off by default because in PaddlePaddle 3.3.1 it does
    not work.

    With it enabled, inference raises `ConvertPirAttribute2RuntimeAttribute not
    support` from the oneDNN executor and no text is read at all. Working beats
    fast, and this is a real fault in a released version rather than caution —
    verified on this platform, not assumed. A deployment whose Paddle build
    handles it can turn it back on for the speed."""

    model_dir: Path | None = None
    """Where the weights live. Set in the container image so nothing is
    downloaded at start-up — a service that fetches half a gigabyte of models on
    first run fails the day the mirror is slow, which is after a deploy, when
    nobody is watching."""

    threads: int = 2
    """CPU threads. Bounded on purpose: PaddleOCR will otherwise take every core
    on the box, and on a small VPS that starves the poller and the webhook
    listener while a single document is read."""

    fallback_language: str = DEFAULT_FALLBACK_LANGUAGE
    """A second alphabet, tried only when the first read found nothing that
    could identify a client.

    No single model reads every Pakistani identity document. On a real old
    all-Urdu card the Arabic-script model missed the identity number and the
    English model found it; on the card photographed next to it, the reverse
    held. The two reads are combined rather than one chosen — see
    app/ocr/fallback.py for the measurements.

    Empty to switch the second pass off, for a deployment that would rather have
    the ten seconds."""

    detection_model: str = DEFAULT_DETECTION_MODEL
    """Which model finds the text before another one reads it.

    Pinned, because choosing a recognition language also chooses a detector and
    the one that comes with `ur` is a *server* model. Measured on this box, with
    the Arabic recogniser held constant in both rows:

        server detector   103.7s   84 chars    (a card small in a phone photo)
        mobile detector    10.4s   97 chars

        server detector    25.1s  122 chars    (a card filling the frame)
        mobile detector     7.5s   98 chars

    Ten times faster, and not worse where it counts: it recovered *more* text
    from the difficult photograph, and both found the identity number on the
    document where either could. The server model's extra characters on an easy
    image are Urdu address lines nothing extracts from.

    Three minutes a document is not a performance note. The daemon runs its jobs
    on one thread, so a read that long also stalls the command poller — a
    customer pressing "Connect WhatsApp" would wait out the whole document
    before seeing a QR code.

    Set TAXPILOT_OCR_DETECTION_MODEL empty to let PaddleOCR pick from the
    language again."""

    max_side: int = 1600
    """Longest edge, in pixels, before an image is scaled down.

    A modern phone photograph is 3000-4000px wide. Feeding that in full costs
    seconds per document and gains nothing — the text is already far above the
    resolution the detector needs — while memory grows with the square of it.

    Passed to the detector as `text_det_limit_side_len`. It was configured and
    never used until the Phase 9 audit, which is worse than absent: a setting
    that promises a bound nobody applies."""

    max_file_bytes: int = 20 * 1024 * 1024
    """Largest document accepted, on disk.

    A compressed image expands to width × height × channels once decoded, so a
    few megabytes can become gigabytes resident. The input arrives from outside
    over WhatsApp, which makes this a bound on what a stranger can make this
    process allocate. Matches the CMS's own upload limit."""

    timeout_seconds: float = 300.0
    """How long one document may be read before the reader gives up.

    Five minutes, chosen from measurement rather than instinct. The slowest of
    twelve real documents took 69 seconds — a 468 KB multi-page PDF — so this is
    roughly four times the slowest legitimate read observed, and would have
    interrupted none of them.

    It exists because there was no deadline at all. A read that never returned
    blocked the inbox job, and since that job takes delivery of everything the
    webhook queued, one stuck document stopped **all** intake until somebody
    restarted the agent. The daemon's watchdog noticed at fifteen minutes and only
    reported it.

    Raise it for a deployment that handles large multi-page scans; the number to
    beat is its own slowest legitimate document, not this one.
    """

    minimum_confidence: float = 0.30
    """Regions the engine is less sure of than this are dropped.

    Not a quality judgement about the document: below roughly this level
    PaddleOCR emits noise from page edges and shadows, and one hallucinated
    line of digits in the middle of a CNIC is worse than a slightly shorter
    read."""

    @classmethod
    def from_env(cls, env: dict[str, str] | None = None) -> OcrConfig:
        env = env if env is not None else dict(os.environ)
        model_dir = env.get("TAXPILOT_OCR_MODEL_DIR")

        return cls(
            language=env.get("TAXPILOT_OCR_LANGUAGE") or DEFAULT_LANGUAGE,
            # Explicitly set-but-empty means "let PaddleOCR choose", which is
            # why this is not the `or` idiom used above.
            detection_model=env.get("TAXPILOT_OCR_DETECTION_MODEL", DEFAULT_DETECTION_MODEL),
            fallback_language=env.get(
                "TAXPILOT_OCR_FALLBACK_LANGUAGE", DEFAULT_FALLBACK_LANGUAGE
            ),
            use_angle_classifier=_flag(env, "TAXPILOT_OCR_ANGLE_CLASSIFIER", True),
            correct_page_orientation=_flag(env, "TAXPILOT_OCR_PAGE_ORIENTATION", True),
            unwarp_pages=_flag(env, "TAXPILOT_OCR_UNWARP", False),
            enable_mkldnn=_flag(env, "TAXPILOT_OCR_MKLDNN", False),
            model_dir=Path(model_dir) if model_dir else None,
            threads=int(env.get("TAXPILOT_OCR_THREADS", "2")),
            max_side=int(env.get("TAXPILOT_OCR_MAX_SIDE", "1600")),
            max_file_bytes=int(env.get("TAXPILOT_OCR_MAX_FILE_BYTES", str(20 * 1024 * 1024))),
            minimum_confidence=float(env.get("TAXPILOT_OCR_MIN_CONFIDENCE", "0.30")),
            timeout_seconds=float(env.get("TAXPILOT_OCR_TIMEOUT_SECONDS", "300")),
        )


def _flag(env: dict[str, str], name: str, default: bool) -> bool:
    raw = env.get(name)

    if raw is None:
        return default

    return raw.strip().lower() not in {"0", "false", "no", "off"}
