"""Reading a document twice, in two alphabets, when once was not enough.

No single recognition model reads every Pakistani identity document. Measured
on four real cards a customer photographed:

    doc1  old all-Urdu card   urdu  95 chars, no CNIC   english  70 chars, CNIC
    doc2  old all-Urdu card   urdu 153 chars, CNIC      english  55 chars, CNIC
    doc3  new card, back      urdu  91 chars, CNIC      english  86 chars, CNIC
    doc4  new card, front     urdu 317 chars, CNIC      english 298 chars, CNIC

The Arabic-script model is the better primary and stays that way. But on doc1
it missed the identity number the English model found, and that number is the
difference between filing against a named client and handing a reviewer a
stranger's card.
"""

from __future__ import annotations

from pathlib import Path

from app.ocr.engine import OcrResult, TextBlock
from app.ocr.fallback import FallbackOcrEngine, identifies_somebody

DOCUMENT = Path("card.jpg")

#: Real, and the reason this module exists: valid under the province-code rule.
CNIC = "35103-7998239-7"


class Recorder:
    """An engine that returns a fixed result and counts how often it was asked."""

    def __init__(self, name: str, text: str = "") -> None:
        self.name = name
        self.reads = 0
        self._text = text

    def read(self, path: Path) -> OcrResult:
        self.reads += 1
        blocks = [TextBlock(text=line, confidence=0.9) for line in self._text.splitlines() if line]

        return OcrResult(blocks=blocks, engine=self.name)


class TestWhenTheSecondPassRuns:
    def test_a_first_read_that_names_nobody_is_read_again(self):
        # doc1: Urdu found the address lines and no identity number.
        primary = Recorder("urdu", "شادی\n47502 ضلع قصور")
        secondary = Recorder("english", f"Identity Number {CNIC}")

        result = FallbackOcrEngine(primary, secondary).read(DOCUMENT)

        assert (primary.reads, secondary.reads) == (1, 1)
        assert CNIC in result.text

    def test_a_first_read_carrying_an_identifier_is_not_read_again(self):
        # doc2, doc3, doc4. A second pass costs another ten seconds and buys
        # nothing on a document that already says whose it is.
        primary = Recorder("urdu", f"شناختی نمبر {CNIC}")
        secondary = Recorder("english", "should not be needed")

        FallbackOcrEngine(primary, secondary).read(DOCUMENT)

        assert (primary.reads, secondary.reads) == (1, 0)

    def test_an_unclassifiable_document_alone_does_not_trigger_it(self):
        """The trigger is "we cannot say whose this is", not "we cannot type it".

        doc2 is unknown in both alphabets and carries its CNIC in both. Reading
        it again would spend ten seconds to learn nothing, so the type is
        deliberately not part of the decision.
        """
        primary = Recorder("urdu", f"{CNIC}\nnothing here matches any marker")
        secondary = Recorder("english", "unused")

        FallbackOcrEngine(primary, secondary).read(DOCUMENT)

        assert secondary.reads == 0


class TestWhatComesBack:
    def test_both_reads_are_kept_rather_than_one_chosen(self):
        """Combining was never worse than the better single read.

        Measured across all four documents, and it recovered the number on
        doc1. Replacing would be choosing between two partial reads, and there
        is no need to choose: the classifier and the extractors both work on
        text, and more correct text can only help them.
        """
        primary = Recorder("urdu", "مستقل پتہ\nضلع قصور")
        secondary = Recorder("english", f"Registrar General of Pakistan\n{CNIC}")

        result = FallbackOcrEngine(primary, secondary).read(DOCUMENT)

        assert "ضلع قصور" in result.text
        assert "Registrar General of Pakistan" in result.text
        assert CNIC in result.text

    def test_the_combined_engine_names_both(self):
        # So a reviewer looking at a run can see that two passes happened.
        result = FallbackOcrEngine(Recorder("urdu"), Recorder("english", "x")).read(DOCUMENT)

        assert result.engine == "urdu+english"

    def test_two_engines_of_the_same_name_are_told_apart_by_alphabet(self):
        """Both PaddleOCR instances answer "paddleocr".

        Named only by that, a two-pass run reported `paddleocr+paddleocr`,
        which tells a reviewer nothing about why the document took two passes.
        """
        primary = Recorder("paddleocr")
        secondary = Recorder("paddleocr", "x")
        primary.language, secondary.language = "ur", "en"

        result = FallbackOcrEngine(primary, secondary).read(DOCUMENT)

        assert result.engine == "paddleocr:ur+paddleocr:en"

    def test_a_second_read_that_finds_nothing_changes_nothing(self):
        # Identical to the single-engine result, which is what somebody
        # comparing two runs would expect.
        primary = Recorder("urdu", "some urdu")
        secondary = Recorder("english", "")

        result = FallbackOcrEngine(primary, secondary).read(DOCUMENT)

        assert result.text == "some urdu"
        assert result.engine == "urdu"


class TestWhatCountsAsIdentifyingSomebody:
    def test_a_cnic_does(self):
        assert identifies_somebody(_result(f"Identity Number {CNIC}"))

    def test_an_iban_does(self):
        assert identifies_somebody(_result("IBAN PK36SCBL0000001123456702"))

    def test_a_mobile_number_does(self):
        assert identifies_somebody(_result("Contact 0300-1234567"))

    def test_dates_and_addresses_do_not(self):
        # Exactly doc1's first pass: real text, none of it a person.
        assert not identifies_somebody(_result("28/08/2019\n28/08/2029\nضلع قصور"))

    def test_an_empty_read_does_not(self):
        assert not identifies_somebody(_result(""))


def _ocr_part(engine):
    """The recognition engine inside the text-layer wrapper.

    build_engine wraps whatever it assembles in TextLayerFirst, so a PDF that
    already carries its text is never rendered and guessed at. That wrapper is
    not what these tests are about — they are about how the recognisers
    underneath are composed — so they look straight through it.
    """
    from app.ocr.text_layer import TextLayerFirst

    return engine._engine if isinstance(engine, TextLayerFirst) else engine  # noqa: SLF001


class TestHowItIsBuilt:
    def test_every_deployment_reads_a_pdf_text_layer_before_reaching_for_ocr(self):
        from app.ocr.config import OcrConfig
        from app.ocr.engine import build_engine
        from app.ocr.text_layer import TextLayerFirst

        # Both shapes: with a second alphabet and without.
        assert isinstance(build_engine(OcrConfig()), TextLayerFirst)
        assert isinstance(build_engine(OcrConfig(fallback_language="")), TextLayerFirst)

    def test_a_deployment_gets_the_second_pass_by_default(self):
        from app.ocr.config import OcrConfig

        config = OcrConfig.from_env({})

        assert config.language == "ur"
        assert config.fallback_language == "en"

    def test_the_second_pass_can_be_switched_off(self):
        from app.ocr.config import OcrConfig
        from app.ocr.engine import build_engine
        from app.ocr.paddle import PaddleOcrEngine

        engine = build_engine(OcrConfig(fallback_language=""))

        assert isinstance(_ocr_part(engine), PaddleOcrEngine)

    def test_a_fallback_matching_the_primary_is_not_a_fallback(self):
        # Reading the same document twice with the same model is pure cost.
        from app.ocr.config import OcrConfig
        from app.ocr.engine import build_engine
        from app.ocr.paddle import PaddleOcrEngine

        engine = build_engine(OcrConfig(language="en", fallback_language="en"))

        assert isinstance(_ocr_part(engine), PaddleOcrEngine)

    def test_the_two_engines_differ_only_in_language(self):
        from app.ocr.config import OcrConfig
        from app.ocr.engine import build_engine

        engine = _ocr_part(build_engine(OcrConfig(max_side=1234)))

        assert isinstance(engine, FallbackOcrEngine)
        # Everything else — thresholds, the detector, the size bound — has to
        # travel to the second engine, or the fallback reads under rules the
        # deployment never chose.
        assert engine._secondary._config.max_side == 1234  # noqa: SLF001
        assert engine._secondary._config.language == "en"  # noqa: SLF001
        assert engine._primary._config.language == "ur"  # noqa: SLF001


def _result(text: str) -> OcrResult:
    return OcrResult(blocks=[TextBlock(text=text, confidence=0.9)] if text else [])
