"""Reading a document twice, in two alphabets, when once was not enough.

Pakistani identity documents are bilingual, and no single recognition model
reads all of them. Measured on four real cards photographed by a customer:

    doc1  old all-Urdu card   urdu  95 chars, no CNIC   english  70 chars, CNIC
    doc2  old all-Urdu card   urdu 153 chars, CNIC      english  55 chars, CNIC
    doc3  new card, back      urdu  91 chars, CNIC      english  86 chars, CNIC
    doc4  new card, front     urdu 317 chars, CNIC      english 298 chars, CNIC

The Arabic-script model is the better primary — it reads far more, and it reads
the English on a bilingual card that the English model garbles. But on doc1 it
missed the identity number that the English model found, and an identity number
is the difference between filing a document against a named client and handing a
reviewer a stranger's card.

## Combined, not replaced

Both reads are kept. On the same four documents, combining was never worse than
the better single read, recovered the number on doc1, and changed no
classification — no type was gained by luck and none was lost to noise.

Replacing would have been a choice between two partial reads. There is no need
to choose: the classifier and the field extractors both work on text, and more
correct text can only help them.

## Only when the first read leaves us unable to say whose document this is

A second pass costs another OCR, roughly ten seconds. It buys nothing on a
document already carrying an identifier, so it does not run there — doc2, doc3
and doc4 above all stop after one read.

The trigger is deliberately *not* "did we classify it". A type nobody
recognised is a real outcome that a second alphabet rarely fixes, and doc2
proves it: unknown in both, with the CNIC found either way. Reading again there
would spend ten seconds to learn nothing.
"""

from __future__ import annotations

import logging
from pathlib import Path

from app.ocr.engine import OcrEngine, OcrResult

logger = logging.getLogger(__name__)


def _describe(engine) -> str:
    """An engine's name, qualified by its alphabet when it has one."""
    language = getattr(engine, "language", None)

    return f"{engine.name}:{language}" if language else engine.name


def identifies_somebody(result: OcrResult) -> bool:
    """Whether this read found anything that could name a client.

    The same identifiers the intake workflow searches on, asked of the
    extractors rather than re-implemented here — a second copy of "what counts
    as a CNIC" would drift from the one that matters.
    """
    from app.ocr.extraction import extract_cnic, extract_iban, extract_mobile, extract_passport

    text = result.text

    return any(
        find(text) is not None
        for find in (extract_cnic, extract_iban, extract_mobile, extract_passport)
    )


class FallbackOcrEngine:
    """Reads with one engine, and again with another when the first is not enough.

    An OcrEngine itself, so nothing downstream knows this is happening — the
    workflow asks for text and gets text.
    """

    def __init__(
        self,
        primary: OcrEngine,
        secondary: OcrEngine,
        is_sufficient=identifies_somebody,
    ) -> None:
        self._primary = primary
        self._secondary = secondary
        self._is_sufficient = is_sufficient

    @property
    def name(self) -> str:
        """Which two engines read it, in a form worth reading back.

        By alphabet where an engine can say — both PaddleOCR engines answer
        "paddleocr", so the obvious composite name is "paddleocr+paddleocr",
        which tells a reviewer nothing about why a document took two passes.
        """
        return f"{_describe(self._primary)}+{_describe(self._secondary)}"

    def read(self, path: Path) -> OcrResult:
        first = self._primary.read(path)

        if self._is_sufficient(first):
            return first

        second = self._secondary.read(path)

        if second.is_empty:
            # Nothing to add. Returning `first` rather than a merge keeps the
            # result identical to the single-engine one, which is what a
            # reviewer comparing two runs would expect.
            return first

        logger.info(
            "Read %s a second time: the first pass found nothing to identify a client by.",
            path.name,
        )

        return OcrResult(
            blocks=[*first.blocks, *second.blocks],
            pages=max(first.pages, second.pages),
            engine=self.name,
        )
