"""Document classification.

"Never guess" is the requirement, so most of these tests are about the classifier
declining. A wrong type files a document where nobody will look for it; an
``unknown`` puts it in a review queue where someone will.
"""

from __future__ import annotations

from app.ocr.classification import UNKNOWN, classify
from app.ocr.engine import NullOcrEngine, OcrResult, TextBlock


class TestConfidentClassification:
    def test_a_cnic_front(self):
        result = classify(
            "ISLAMIC REPUBLIC OF PAKISTAN\nNATIONAL IDENTITY CARD\nNADRA\n"
            "Identity Number 35202-1234567-1\nDate of Birth 01.01.1990"
        )

        assert result.document_type == "cnic_front"
        assert result.is_confident

    def test_the_back_of_a_card_as_the_engine_actually_reads_it(self):
        """Captured from a real CNIC back, digits changed.

        The back is mostly Urdu and this engine reads English, so what comes out
        is 91 characters of noise with two legible things in it: the identity
        number and "Registrar General of Pakistan". The tidy English labels this
        table relied on — permanent address, present address, NADRA — appeared
        nowhere on it.

        It classified as unknown, with cnic_front the runner-up on 2.0 against a
        threshold of 3.0, and the document reached a reviewer with no type.
        """
        result = classify(
            "Ctrl\n3513620 35202-1234567-1\n190.\n7132\n47502 1.5\n"
            "880000875655\nRegistrar General of Pakistan"
        )

        assert result.document_type == "cnic_back"

    def test_a_property_document_is_not_mistaken_for_a_card_back(self):
        # Both mention a registrar. Only one says "Registrar General", which is
        # why the marker is the phrase and not the word.
        result = classify(
            "SALE DEED\nPlot No 42, Block C\nSub Registrar Lahore\nMutation No 1187"
        )

        assert result.document_type == "property_document"

    def test_a_bank_statement(self):
        result = classify(
            "STATEMENT OF ACCOUNT\nOpening Balance 10,000\nClosing Balance 45,000\n"
            "PK36SCBL0000001123456702"
        )

        assert result.document_type == "bank_statement"

    def test_a_utility_bill(self):
        result = classify("K-ELECTRIC\nBill Month: June 2024\nUnits Consumed: 320\nDue Date 15/07/2024")

        assert result.document_type == "utility_bill"

    def test_a_salary_slip(self):
        result = classify("SALARY SLIP\nGross Salary 150,000\nDeductions 12,000\nNet Pay 138,000")

        assert result.document_type == "salary_slip"

    def test_an_fbr_notice(self):
        result = classify("FEDERAL BOARD OF REVENUE\nNotice u/s 114\nCommissioner Inland Revenue")

        assert result.document_type == "fbr_notice"

    def test_a_tax_return(self):
        result = classify("INCOME TAX RETURN\nTaxable Income 1,200,000\nIRIS Acknowledgement")

        assert result.document_type == "tax_return"


class TestRefusalToGuess:
    def test_empty_text_is_unknown(self):
        result = classify("")

        assert result.document_type == UNKNOWN
        assert "no text" in result.evidence[0]

    def test_unrecognisable_text_is_unknown(self):
        result = classify("Dear Sir, please find attached the file we discussed. Regards.")

        assert result.document_type == UNKNOWN

    def test_a_single_keyword_in_prose_is_not_a_classification(self):
        # This one caught a real weakness: the word "invoice" alone scored
        # exactly the threshold and won outright. One mention is a mention.
        result = classify("Please see the invoice discussion in our last meeting.")

        assert result.document_type == UNKNOWN
        assert "single marker" in result.evidence[0]

    def test_two_close_scores_produce_unknown(self):
        # Genuinely ambiguous: both types have corroborated evidence and neither
        # is clearly ahead. When the text cannot separate them, a human should.
        result = classify("INVOICE\nAmount Due 50,000\nRECEIPT\nPaid")

        assert result.document_type == UNKNOWN
        assert "too close" in result.evidence[0]

    def test_an_unknown_result_still_names_its_best_candidate(self):
        # Useless to a filing rule, useful to the human reviewing the queue.
        result = classify("Please see the invoice discussion in our last meeting.")

        assert result.runner_up is not None


class TestConfidence:
    def test_confidence_never_reaches_certainty(self):
        # Keyword evidence cannot justify 100%, and claiming it would invite a
        # caller to skip the human.
        result = classify(
            "NATIONAL IDENTITY CARD NADRA Identity Number Date of Birth 35202-1234567-1"
        )

        assert result.confidence <= 0.99

    def test_stronger_evidence_scores_higher(self):
        weak = classify("STATEMENT OF ACCOUNT\nClosing Balance 45,000")
        strong = classify(
            "STATEMENT OF ACCOUNT\nOpening Balance 10,000\nClosing Balance 45,000\n"
            "Debit and Credit\nPK36SCBL0000001123456702"
        )

        assert strong.confidence > weak.confidence

    def test_evidence_is_reported(self):
        # A classification a human cannot audit is a classification they must
        # take on trust.
        result = classify("SALARY SLIP\nGross Salary 150,000\nNet Pay 138,000")

        assert result.evidence


class TestOcrResult:
    def test_confidence_is_weighted_by_text_length(self):
        # A long clean body with a misread one-word header is a good read.
        result = OcrResult(
            blocks=[TextBlock("x" * 100, 0.99), TextBlock("hdr", 0.20)],
            engine="test",
        )

        assert result.confidence > 0.9

    def test_a_clean_header_over_an_unreadable_body_is_not_a_good_read(self):
        result = OcrResult(
            blocks=[TextBlock("hdr", 0.99), TextBlock("x" * 100, 0.20)],
            engine="test",
        )

        assert result.confidence < 0.3

    def test_an_empty_result_reports_zero_confidence(self):
        assert OcrResult().confidence == 0.0
        assert OcrResult().is_empty

    def test_an_impossible_confidence_is_rejected(self):
        import pytest

        with pytest.raises(ValueError, match="between 0 and 1"):
            TextBlock("text", 1.5)


class TestNullEngine:
    def test_it_reads_nothing_rather_than_inventing_text(self):
        from pathlib import Path

        result = NullOcrEngine().read(Path("anything.pdf"))

        # A deployment with no OCR engine must degrade to asking a human, not
        # to confidently filing nonsense.
        assert result.is_empty
        assert result.confidence == 0.0
        assert classify(result.text).document_type == UNKNOWN


class TestUrduOnlyDocuments:
    """An older CNIC carries no English at all.

    PaddleOCR's Arabic model is trained on Arabic, and Urdu uses letters Arabic
    does not — it returns ك for ک, ي or ى for ی, ه for ہ. Urdu written
    correctly matches none of that, so everything is folded to one spelling
    before any marker is tried.

    Folding is not enough on its own. Measured on a real card,
    "قومی شناختی کارڈ" came back as "قويى شنايى كارد" — letters *missing*, not
    merely substituted. A phrase pattern matches nothing on real output, which
    is why the markers are single words: the ones that survived intact.
    """

    #: Captured from a real old all-Urdu CNIC front, identity number changed.
    REAL_FRONT = (
        "ctr\n"
        "حكومت ياكستان\n"
        "قويى شنايى كارد\n"
        "35202-1234567-1 ٠م ٠متاز\n"
        "جنس مرد\n"
        "والدكانام ارت خال\n"
        "شاخى صامتكوني سي عان يفسين\n"
        "تاريت بيداش وشط جسشر ارجضرال"
    )

    def test_an_all_urdu_card_front_is_recognised(self):
        result = classify(self.REAL_FRONT)

        assert result.document_type == "cnic_front"
        assert result.is_confident

    def test_the_letters_the_recogniser_substitutes_still_match(self):
        """The whole point of folding.

        This text spells "حکومت" with the Arabic kaf the model actually emits.
        Without folding it is a different string from the marker and matches
        nothing.
        """
        from app.ocr.classification import fold_arabic

        assert "ك" in self.REAL_FRONT           # what the recogniser produced
        assert "حکومت" in fold_arabic(self.REAL_FRONT)   # what the marker asks for

    def test_folding_leaves_english_alone(self):
        from app.ocr.classification import fold_arabic

        english = "STATEMENT OF ACCOUNT\nClosing Balance 45,000"

        assert fold_arabic(english) == english

    def test_an_english_document_still_classifies_the_same(self):
        # Folding runs on every document now, so the English path is worth
        # asserting rather than assuming.
        result = classify(
            "STATEMENT OF ACCOUNT\nOpening Balance 10,000\nClosing Balance 45,000"
        )

        assert result.document_type == "bank_statement"

    def test_the_back_of_an_older_card_is_still_not_recognised(self):
        """Recorded because it is a real limit, not an oversight.

        The back of an old card came through as 95 characters of noise — two
        dates, a family number, and no Urdu word intact. Six candidate markers
        were tested against it and none survived, so there is nothing honest to
        match on.

        It is not lost: the second OCR pass recovers the identity number, so the
        document reaches a reviewer attached to the right client, with the type
        left for them to choose.
        """
        real_back = (
            "caps lock\nshirt t\nCtPy\nنافى ا\nر C06547\n"
            "اري ن غء28/08/2019\n28/08/2029 دكار رقربى يجز س ي ذال وين"
        )

        assert classify(real_back).document_type == "unknown"
