"""The OCR port.

ADR-0002 puts original documents at Level 3: they never leave the installation,
so text extraction happens locally. PaddleOCR is the intended production engine,
but it is a large dependency with platform-specific wheels, and nothing above
this line should care which engine is installed.

So the engine is a protocol. Everything downstream — classification, field
extraction, workflows — is written against plain text plus per-block confidence,
and is therefore fully testable without any engine present.
"""

from __future__ import annotations

from dataclasses import dataclass, field
from pathlib import Path
from typing import Protocol, runtime_checkable


@dataclass(frozen=True, slots=True)
class TextBlock:
    """One recognised region, with the engine's confidence in it."""

    text: str
    confidence: float

    def __post_init__(self) -> None:
        if not 0.0 <= self.confidence <= 1.0:
            raise ValueError(f"Confidence must be between 0 and 1, got {self.confidence}.")


@dataclass(frozen=True, slots=True)
class OcrResult:
    """Everything an engine produces for one document."""

    blocks: list[TextBlock] = field(default_factory=list)
    pages: int = 1
    engine: str = "unknown"

    failure: str | None = None
    """Why the engine could not read this, when it could not.

    An empty result has two completely different causes needing opposite
    responses: a page with nothing on it is finished with, while a file the
    engine could not decode needs somebody to ask the client to send it again.
    Without this they were indistinguishable — both arrived as zero blocks — so a
    corrupt photograph and a blank sheet produced identical records and identical
    advice.

    Recorded rather than raised, deliberately. A failed read must not end a
    workflow that should still reach a human, so the failure travels *with* the
    result instead of in place of it.
    """

    @property
    def failed(self) -> bool:
        """The engine could not read, as against read and found nothing."""
        return self.failure is not None

    @property
    def text(self) -> str:
        return "\n".join(block.text for block in self.blocks)

    @property
    def confidence(self) -> float:
        """Mean block confidence, weighted by text length.

        Weighted deliberately: a document whose long body scanned cleanly but
        whose one-word header did not is a good read, and an unweighted mean
        would report it as mediocre. The reverse — a clean header over an
        unreadable body — must not look good, and this reports it correctly.
        """
        if not self.blocks:
            return 0.0

        total_chars = sum(len(b.text) for b in self.blocks)

        if total_chars == 0:
            return 0.0

        return sum(b.confidence * len(b.text) for b in self.blocks) / total_chars

    @property
    def is_empty(self) -> bool:
        return not self.text.strip()


@runtime_checkable
class OcrEngine(Protocol):
    """What any OCR implementation must provide."""

    name: str

    def read(self, path: Path) -> OcrResult:
        """Extract text from a document.

        Implementations must not raise for an unreadable file: an empty
        OcrResult is the honest answer, and a workflow can act on it. An
        exception here would end a run that should have asked a human.
        """
        ...


class NullOcrEngine:
    """A engine that reads nothing, for environments without one installed.

    Deliberately not a fake that invents text. It reports an empty result with
    zero confidence, which every downstream consumer already handles as "could
    not read" — so a deployment missing its OCR engine degrades to asking a
    human rather than to confidently filing nonsense.
    """

    name = "null"

    def read(self, path: Path) -> OcrResult:  # noqa: ARG002 - deliberately ignored
        return OcrResult(blocks=[], engine=self.name)


def build_engine(config=None) -> OcrEngine:
    """The engine this deployment should use.

    PaddleOCR when it is installed, the null engine otherwise — and the
    difference is logged loudly, because the two behave identically from every
    caller's point of view and differ entirely in whether the product works.

    Degrading rather than refusing to start is deliberate. An installation whose
    OCR is missing still receives documents, still reaches its CMS, and still
    puts every one of them in front of a human — unread, and marked high risk
    because nothing could be read. That is a bad day, not an outage.
    """
    import logging

    from app.ocr.config import OcrConfig

    logger = logging.getLogger(__name__)

    try:
        from app.ocr.paddle import PaddleOcrEngine
    except ImportError:
        logger.error(
            "PaddleOCR is not installed. Every document will reach a reviewer unread."
        )

        return NullOcrEngine()

    from dataclasses import replace

    from app.ocr.text_layer import TextLayerFirst

    config = config or OcrConfig.from_env()
    primary = PaddleOcrEngine(config)

    # Wrapped outermost, so a PDF that already carries its text never reaches
    # either alphabet. Eight of the nine documents production failed to read in
    # early August were exports with a perfect text layer, being rendered to
    # bitmaps and guessed at until the deadline killed them.
    if not config.fallback_language or config.fallback_language == config.language:
        return TextLayerFirst(primary)

    # A second alphabet, read only when the first pass found nothing that could
    # identify a client. Both engines are built here, but PaddleOCR loads its
    # models lazily — a deployment whose documents all read first time never
    # pays for the second.
    from app.ocr.fallback import FallbackOcrEngine

    logger.info(
        "Reading with '%s', and again with '%s' when a document identifies nobody.",
        config.language,
        config.fallback_language,
    )

    return TextLayerFirst(
        FallbackOcrEngine(
            primary,
            PaddleOcrEngine(replace(config, language=config.fallback_language)),
        )
    )
