"""PaddleOCR — the production engine (ADR-0002).

Local, because original documents are Level 3 and never leave the installation.
That rules out Google Vision, Azure and Textract regardless of their accuracy:
each requires sending the image off-box, and a CNIC photograph is exactly what
must not go.

PaddleOCR over Tesseract because the input is phone photographs sent over
WhatsApp — rotated, shadowed, uneven — which is precisely where Tesseract's
accuracy collapses.

The cost is a ~2 GB install. Runtime memory was guessed at "a 4 GB floor" here
until it was measured: a real read peaks at **~1 GB** resident, 34s cold while
the models load and 18s warm (Xeon E5-2699 v4, 8 cores, single document). The
guess was four times the truth and it was being used to size hosts.

The import is deliberately local to this module. A deployment that has not
installed PaddleOCR still starts, still reaches its CMS, and reports honestly
that it cannot read documents.
"""

from __future__ import annotations

import logging
from concurrent.futures import ThreadPoolExecutor, TimeoutError as FutureTimeout
from pathlib import Path
from typing import Any

from app.ocr.config import OcrConfig
from app.ocr.engine import OcrResult, TextBlock
from app.ocr.layout import Region, from_polygon, reading_order

logger = logging.getLogger(__name__)

NAME = "paddleocr"

#: What each extra page of a scan is worth, in seconds.
#:
#: Deliberately far below the configured deadline. That deadline covers loading
#: the models and reading a page; every page after the first only adds the
#: reading. Measured against the August backlog, a page took roughly a minute.
SECONDS_PER_EXTRA_PAGE = 90.0

#: The longest any single document may hold the reader, whatever its length.
#:
#: The deadline exists because one document must never stop intake for everyone
#: else, and a per-page budget with no ceiling hands that back: sixteen pages at
#: the full deadline each would be eighty minutes during which nothing else is
#: read. Twenty minutes is enough for a long scan and short enough that a
#: pathological one is a delay rather than an outage.
#:
#: Past it the document goes to a reviewer unread, which is the outcome this
#: whole module treats as acceptable and a blocked queue is not.
MAX_DEADLINE_SECONDS = 1200.0


class _ReadTimedOut(RuntimeError):
    """The reader ran past its deadline. Internal to this module."""


class PaddleOcrUnavailable(RuntimeError):
    """PaddleOCR is not installed or cannot start."""


class PaddleOcrEngine:
    """Reads documents locally.

    The models load on first use, not on construction. Loading is slow and takes
    hundreds of megabytes, and doing it at start-up means a service that cannot
    report why it failed to start. First use is also when it is actually needed.
    """

    name = NAME

    @property
    def language(self) -> str:
        """The alphabet this instance reads.

        Exposed so a composite engine can say which of two passes produced
        what — both instances answer "paddleocr" for `name`, and "paddleocr+
        paddleocr" on a run tells a reviewer nothing.
        """
        return self._config.language

    def __init__(self, config: OcrConfig | None = None) -> None:
        self._config = config or OcrConfig()
        self._reader: Any | None = None

    # ── The port ──────────────────────────────────────────────────────────

    def read(self, path: Path) -> OcrResult:
        """Extract text from a document.

        Never raises. An unreadable file, a missing engine and a page of noise
        are all real outcomes with the same honest answer — an empty result,
        which every caller already handles by asking a human. An exception here
        would end a workflow that should have gone to the Approval Queue.
        """
        deadline = self._deadline_for(path)

        try:
            pages = self._recognise_within(path, deadline)
        except _ReadTimedOut:
            logger.error(
                "PaddleOCR did not finish %s within %.0fs; giving up on it.",
                path.name, deadline,
            )

            return OcrResult(
                blocks=[], engine=self.name,
                failure=(
                    f"The document could not be read within {deadline:.0f} seconds."
                ),
            )
        except PaddleOcrUnavailable:
            logger.error("PaddleOCR is not available; no text was read from %s.", path.name)

            return OcrResult(
                blocks=[], engine=self.name,
                failure="The OCR engine is not available on this deployment.",
            )
        except Exception as exc:  # noqa: BLE001 - see docstring
            # Logged with the filename only. The contents are Level 3 and a log
            # file is not where they belong.
            logger.exception("PaddleOCR failed on %s: %s", path.name, exc)

            # Recorded on the result rather than raised. Still an empty read, so
            # the workflow carries on to a human exactly as before — but now
            # distinguishable from a page that was read and had nothing on it,
            # which is the difference between "ask the client to send this again"
            # and "this page is blank".
            return OcrResult(
                blocks=[], engine=self.name,
                failure=f"The document could not be read ({type(exc).__name__}).",
            )

        blocks: list[TextBlock] = []

        for page in pages:
            blocks.extend(self._page_to_blocks(page))

        return OcrResult(blocks=blocks, pages=max(len(pages), 1), engine=self.name)

    def _recognise_within(self, path: Path, seconds: float):
        """Recognise, or give up after `seconds`.

        ## Why a thread rather than a real cancellation

        There is none to be had. PaddleOCR blocks inside native code, and neither
        a signal (Unix-only, and only on the main thread) nor any Python-level
        interrupt reaches it. The honest options were a subprocess per read —
        which reloads 335 MB of models each time, turning a 17-second photograph
        into a minute — or a deadline that stops *waiting* without stopping the
        work.

        This is the second. The caller gets its answer at the deadline and the
        pipeline moves on; the abandoned thread finishes in its own time and its
        result is discarded.

        ## What that costs, said plainly

        A hung read leaks a thread and whatever it holds until the process
        restarts. That is a real cost and it is the lesser one: before this, a
        single stuck document blocked the inbox job, and with it delivery of
        every other document the webhook had queued. One leaked thread against
        all intake stopping is not a close comparison.

        Rare by construction — the deadline is roughly four times the slowest
        real document measured.
        """
        if seconds <= 0:
            return self._recognise(path)

        executor = ThreadPoolExecutor(max_workers=1, thread_name_prefix="ocr")

        try:
            future = executor.submit(self._recognise, path)

            try:
                return future.result(timeout=seconds)
            except FutureTimeout as timed_out:
                raise _ReadTimedOut(str(path)) from timed_out
        finally:
            # Never blocks on the abandoned worker: wait=True here would wait out
            # precisely the read this exists to stop waiting for.
            executor.shutdown(wait=False)

    def _deadline_for(self, path: Path) -> float:
        """How long this particular document is allowed.

        The deadline was one number for every document, which is right for a
        photograph and impossible for a scan. A sixteen-page scanned statement
        was given the same five minutes as a single CNIC and could not finish —
        it was not stuck, it was doing sixteen pages of work against a budget
        sized for one. Every document production failed to read between 1 and 6
        August 2026 had four or more pages; everything that succeeded had one or
        two.

        So the budget follows the work: the configured deadline covers loading
        the models and reading one page, and each further page adds its own
        smaller allowance. Most PDFs never reach here at all now —
        app/ocr/text_layer.py reads their embedded text first — so this governs
        genuine scans, which are the only ones that need it.

        Capped hard. A deadline exists so that one document cannot stop intake
        for every other, and a per-page budget without a ceiling gives exactly
        that back.
        """
        base = self._config.timeout_seconds

        if path.suffix.lower() != ".pdf":
            return base

        try:
            import pypdfium2 as pdfium

            document = pdfium.PdfDocument(str(path))

            try:
                pages = max(1, len(document))
            finally:
                document.close()
        except Exception:  # noqa: BLE001 - unreadable here means "use the base deadline"
            return base

        return min(base + (pages - 1) * SECONDS_PER_EXTRA_PAGE, MAX_DEADLINE_SECONDS)

    # ── Internals ─────────────────────────────────────────────────────────

    @property
    def reader(self) -> Any:
        if self._reader is None:
            self._reader = self._build_reader()

        return self._reader

    def _build_reader(self) -> Any:
        try:
            from paddleocr import PaddleOCR
        except ImportError as exc:
            raise PaddleOcrUnavailable(
                "PaddleOCR is not installed. Install it, or run with the null engine "
                "and every document will go to a human unread."
            ) from exc

        kwargs: dict[str, Any] = {
            "lang": self._config.language,
            "use_textline_orientation": self._config.use_angle_classifier,
            "use_doc_orientation_classify": self._config.correct_page_orientation,
            "use_doc_unwarping": self._config.unwarp_pages,
            # Passed through to PaddleX. Off by default because PaddlePaddle
            # 3.3.1's oneDNN executor raises
            # "ConvertPirAttribute2RuntimeAttribute not support" and reads
            # nothing at all — verified on this platform, not assumed.
            "enable_mkldnn": self._config.enable_mkldnn,
            # Caps the longest edge before detection. Without it a 4000px phone
            # photograph is processed at full resolution for no accuracy gain and
            # quadratic memory.
            "text_det_limit_side_len": self._config.max_side,
            # "max", and this is not optional. PaddleOCR defaults to "min",
            # which scales the SHORTEST side *up* to the limit — so setting a
            # side length without this makes images larger, not smaller. Measured
            # on one 2400px document: "min" took 96.9s, "max" took 19.3s, with
            # identical output. The parameter did the exact opposite of its
            # purpose until this line existed.
            "text_det_limit_type": "max",
        }

        recogniser = self._recognition_model() if self._config.detection_model else None

        if recogniser:
            # BOTH names, together, or neither.
            #
            # Naming a detector alone makes PaddleOCR forget the recogniser the
            # language chose: it silently loaded PP-OCRv6_medium_rec instead of
            # the Arabic model, so Urdu stopped being read while every log line
            # still said the language was `ur`. That nearly passed as a
            # measurement showing the mobile detector was better — it was
            # reading a different alphabet.
            kwargs["text_detection_model_name"] = self._config.detection_model
            kwargs["text_recognition_model_name"] = recogniser

        if self._config.model_dir is not None:
            # Set in the container image. A service that downloads models on
            # first run fails the day the mirror is slow — which is after a
            # deploy, when nobody is watching.
            kwargs["text_detection_model_dir"] = str(self._config.model_dir / "det")
            kwargs["text_recognition_model_dir"] = str(self._config.model_dir / "rec")

        try:
            return PaddleOCR(**kwargs)
        except Exception as exc:  # noqa: BLE001
            raise PaddleOcrUnavailable(f"PaddleOCR could not start: {exc}") from exc

    def _recognition_model(self) -> str | None:
        """The recogniser the configured language would have chosen for itself.

        Asked of PaddleOCR rather than hard-coded here, so that a future release
        remapping a language moves this with it instead of leaving a copy of an
        old answer behind.

        A private method, called unbound with no instance — it takes `self` and
        never uses it, checked rather than assumed. If a release changes that,
        this returns None and the caller stops pinning the detector: slower, and
        still reading the right alphabet, which is the correct way round to fail.
        """
        try:
            from paddleocr._pipelines.ocr import PaddleOCR as _Pipeline

            _, recogniser = _Pipeline._get_ocr_model_names(None, self._config.language, None)
        except Exception:  # noqa: BLE001
            logger.warning(
                "Could not resolve the recognition model for '%s'; letting PaddleOCR "
                "choose both models. Reads will be slower.",
                self._config.language,
            )

            return None

        return recogniser

    def _recognise(self, path: Path) -> list[Any]:
        if not path.exists():
            logger.error("No document at %s.", path)

            return []

        size = path.stat().st_size

        if size > self._config.max_file_bytes:
            # Refused before the file reaches the decoder. A compressed image
            # expands to width × height × channels in memory, so a few megabytes
            # on disk can become gigabytes resident — and the input here arrives
            # from outside, over WhatsApp.
            logger.error(
                "Refusing a %d byte document; the limit is %d.", size, self._config.max_file_bytes
            )

            return []

        result = self.reader.predict(str(path))

        # One entry per page. A single-page image still returns a list, so
        # callers never branch on how many pages a document happened to have.
        return list(result or [])

    def _page_to_blocks(self, page: Any) -> list[TextBlock]:
        """Turn one page of engine output into ordered text blocks.

        This is where reading order is imposed. The engine returns regions in
        detection order, and everything downstream reads the joined text — so a
        label and its value arriving out of order is a field that silently fails
        to extract, for reasons nothing about the scan explains.
        """
        regions = self._page_to_regions(page)

        if not regions:
            return []

        return [
            TextBlock(text=" ".join(r.text for r in line), confidence=_line_confidence(line))
            for line in reading_order(regions)
        ]

    def _page_to_regions(self, page: Any) -> list[Region]:
        """Read PaddleOCR's output shape.

        Isolated deliberately: this is the only part coupled to the library's
        result format, which has changed between major versions. Everything else
        here works on Regions.
        """
        texts, scores, polys = _unpack(page)

        regions: list[Region] = []

        for text, score, polygon in zip(texts, scores, polys, strict=False):
            cleaned = str(text).strip()

            if not cleaned:
                continue

            confidence = _clamp(float(score))

            if confidence < self._config.minimum_confidence:
                # Below this the engine emits noise from page edges and shadows.
                # One hallucinated line of digits inside a CNIC is worse than a
                # slightly shorter read.
                continue

            regions.append(from_polygon(cleaned, confidence, polygon))

        return regions


def _unpack(page: Any) -> tuple[list, list, list]:
    """Get texts, scores and polygons out of a page result.

    PaddleOCR 3.x returns a mapping with ``rec_texts``/``rec_scores``/``rec_polys``.
    Handled by key rather than by position so a library that adds a field does
    not shift everything by one.
    """
    if hasattr(page, "get"):
        return (
            list(page.get("rec_texts") or []),
            list(page.get("rec_scores") or []),
            list(page.get("rec_polys") or []),
        )

    # 2.x shape: [[polygon, (text, score)], ...]. Supported because an
    # installation pinned to the older library should degrade to working, not to
    # a crash nobody can interpret.
    texts, scores, polys = [], [], []

    for entry in page or []:
        try:
            polygon, (text, score) = entry[0], entry[1]
        except (TypeError, ValueError, IndexError):
            continue

        polys.append(polygon)
        texts.append(text)
        scores.append(score)

    return texts, scores, polys


def _line_confidence(line: list[Region]) -> float:
    """Confidence for a joined line, weighted by how much text each region held.

    The same weighting as OcrResult.confidence, and for the same reason: a long
    accurate value beside a short uncertain label is a good read, and an
    unweighted mean would report it as mediocre.
    """
    total = sum(len(r.text) for r in line)

    if total == 0:
        return 0.0

    return _clamp(sum(r.confidence * len(r.text) for r in line) / total)


def _clamp(value: float) -> float:
    """TextBlock refuses anything outside 0–1, and floating point drifts."""
    return max(0.0, min(1.0, value))
