"""The classification pipeline — four stages, ordered by what they cost.

The behaviour worth protecting is not "it classifies things". It is that the
expensive stage runs only when the cheap ones could not answer, and that the
cheap ones never answer by guessing.

`_reads` counts OCR calls in every test that could trigger one, because "it
classified correctly" and "it classified correctly without opening the file" are
different results and only one of them is the design.
"""

from __future__ import annotations

from pathlib import Path

import pytest

from app.documents import registry
from app.documents.classifier import UNKNOWN, Method, classify
from app.documents.signals import Signals


@pytest.fixture
def document(tmp_path) -> Path:
    path = tmp_path / "scan.pdf"
    path.write_bytes(b"%PDF-1.4 test")

    return path


def signals(document: Path, *, caption="", filename="", text=None, **kwargs) -> tuple[Signals, list]:
    """Signals plus a record of how often the document was read."""
    reads: list[int] = []

    def read() -> str:
        reads.append(1)

        return text or ""

    return (
        Signals(
            path=document,
            caption=caption,
            filename=filename or document.name,
            read_first_page=read if text is not None else None,
            **kwargs,
        ),
        reads,
    )


class TestTheCaptionDecidesFirst:
    def test_a_caption_naming_the_type_settles_it(self, document):
        s, reads = signals(document, caption="Bank Statement", text="ignored")

        result = classify(s)

        assert result.filing_type == "bank_statement"
        assert result.method is Method.CAPTION
        # The whole point. Nothing opened the file.
        assert reads == []
        assert result.was_read is False

    def test_a_caption_naming_a_refinement_keeps_both_names(self, document):
        s, _ = signals(document, caption="Sale Deed for the Lahore plot")

        result = classify(s)

        # Filed as a property document; known to be a sale deed.
        assert result.filing_type == "property_document"
        assert result.refinement == "sale_deed"
        assert result.label == "Sale Deed"

    @pytest.mark.parametrize(
        "caption,expected",
        [
            ("salary slip", "salary_slip"),
            ("Salary_Slip", "salary_slip"),
            ("SALARYSLIP july", "salary_slip"),
            ("utility bill", "utility_bill"),
            ("wealth statement", "wealth_statement"),
            ("FBR Notice", "fbr_notice"),
            ("passport", "passport"),
        ],
    )
    def test_labels_are_matched_however_they_are_typed(self, document, caption, expected):
        s, _ = signals(document, caption=caption)

        assert classify(s).filing_type == expected

    def test_the_longer_name_wins(self, document):
        # "Income Tax Return" contains "Tax Return". A caption saying the first
        # means the first, whichever entry sits earlier in the registry.
        s, _ = signals(document, caption="Income Tax Return 2025")

        result = classify(s)

        assert result.filing_type == "tax_return"
        assert result.refinement == "income_tax_return"

    def test_a_cnic_used_to_name_the_client_is_not_a_cnic_document(self, document):
        """The trap the alias rule exists for.

        This is the brief's own example of a captioned bank statement, and a
        bare `cnic` alias would have two types fighting over one caption.
        """
        s, _ = signals(document, caption="Bank Statement\nCNIC: 35202-1234567-1")

        result = classify(s)

        assert result.filing_type == "bank_statement"
        assert result.refinement is None

    def test_a_bare_cnic_with_no_number_does_name_the_document(self, document):
        s, _ = signals(document, caption="CNIC front please file")

        assert classify(s).filing_type == "cnic_front"

    def test_a_caption_naming_two_types_settles_nothing(self, document):
        # A person being unclear. Picking one would be a guess dressed as a rule.
        s, _ = signals(document, caption="salary slip and utility bill", text="")

        assert classify(s).filing_type == UNKNOWN

    def test_a_caption_naming_nothing_falls_through(self, document):
        s, reads = signals(
            document,
            caption="here you go",
            text="SALARY SLIP\nGross Salary 150,000\nNet Pay 138,000",
        )

        result = classify(s)

        assert result.filing_type == "salary_slip"
        assert result.method is Method.RULES
        assert reads == [1]


class TestTheFilenameIsEvidenceNotProof:
    def test_a_filename_classifies_when_nothing_else_can(self, document):
        s, _ = signals(document, filename="Meezan_Bank_Statement_July2026.pdf", text="")

        result = classify(s)

        assert result.filing_type == "bank_statement"
        assert result.method is Method.FILENAME

    def test_a_bank_names_it_an_account_statement_not_a_bank_statement(self, document):
        """The real filename from the first live forward, 31 July 2026.

        It classified as unknown, so the router sent it to the reading workflow
        and a client's bank statement was OCR'd — the one thing this type's
        handling exists to prevent. The phrase was in the text markers all
        along; it was the filename list, which is what decides routing before
        the file is ever downloaded, that had only the other word order.
        """
        s, _ = signals(
            document,
            filename="Account Statement - 12-Jul-26 - baajbacfdh_unl.pdf",
            text="",
        )

        result = classify(s)

        assert result.filing_type == "bank_statement"
        assert result.method is Method.FILENAME

    def test_a_filename_does_not_stop_the_document_being_read(self, document):
        """Below the accept threshold on purpose.

        A filename is wrong in ways that leave no trace — a template reused from
        another client, a phone renaming a scan — so it must never prevent the
        text being consulted.
        """
        s, reads = signals(
            document,
            filename="SalarySlip_July.pdf",
            text="STATEMENT OF ACCOUNT\nClosing Balance 45,000\nOpening Balance 30,000",
        )

        result = classify(s)

        assert reads == [1]
        # The text wins: the file was misnamed.
        assert result.filing_type == "bank_statement"
        assert result.method is Method.RULES

    def test_the_filename_stands_when_the_text_says_nothing(self, document):
        s, reads = signals(document, filename="IncomeTaxReturn2025.pdf", text="")

        result = classify(s)

        assert reads == [1]
        assert result.filing_type == "tax_return"
        assert result.method is Method.FILENAME

    def test_a_filename_can_name_a_refinement(self, document):
        s, _ = signals(document, filename="sale-deed-plot-42.pdf", text="")

        result = classify(s)

        assert result.filing_type == "property_document"
        assert result.refinement == "sale_deed"


class TestTheRuleStage:
    def test_it_refines_within_the_type_it_chose(self, document):
        s, _ = signals(
            document,
            text="SALE DEED\nPlot No 42, Khasra 118\nRegistrar of Properties",
        )

        result = classify(s)

        assert result.filing_type == "property_document"
        assert result.refinement == "sale_deed"
        assert result.method is Method.RULES

    def test_it_carries_the_rule_engine_s_reason_for_refusing(self, document):
        s, _ = signals(document, text="Please see the invoice discussion in our last meeting.")

        result = classify(s)

        assert result.filing_type == UNKNOWN
        # "rests on a single marker" is more use to a reviewer than "unknown".
        assert "single marker" in result.reason

    def test_it_reports_the_document_as_read(self, document):
        s, _ = signals(document, text="SALARY SLIP\nGross Salary 1\nNet Pay 2")

        assert classify(s).was_read is True


class TestWhenNothingKnows:
    def test_an_unrecognisable_document_is_unknown_not_a_guess(self, document):
        s, _ = signals(document, filename="scan001.pdf", text="Dear Sir, regards.")

        result = classify(s)

        assert result.filing_type == UNKNOWN
        assert result.confidence == 0.0

    def test_a_deployment_with_no_reader_still_classifies(self, document):
        # read_first_page is None — no OCR available at all.
        s, _ = signals(document, filename="Bank_Statement.pdf")

        result = classify(s)

        assert result.filing_type == "bank_statement"
        assert result.was_read is False

    def test_a_reader_that_fails_is_an_answer_not_a_crash(self, document):
        def explode() -> str:
            raise RuntimeError("PaddleOCR fell over")

        s = Signals(path=document, filename="scan.pdf", read_first_page=explode)

        result = classify(s)

        assert result.filing_type == UNKNOWN
        # It was opened, and it could not be read. The CMS scores those apart.
        assert result.was_read is True

    def test_the_document_is_read_at_most_once(self, document):
        s, reads = signals(document, text="")

        classify(s)
        s.text()
        s.text()

        assert reads == [1]


class TestEveryTypeIsReachable:
    """A registered type nothing can produce is configuration, not a feature."""

    @pytest.mark.parametrize("filing", [t for t in registry.filing_types() if t.slug != "other"])
    def test_each_filing_type_can_be_named_in_a_caption(self, document, filing):
        s, _ = signals(document, caption=filing.label)

        result = classify(s)

        assert result.filing_type == filing.slug, f"{filing.label!r} named {result.filing_type}"

    @pytest.mark.parametrize("refinement", registry.REFINEMENTS)
    def test_each_refinement_can_be_named_in_a_caption(self, document, refinement):
        s, _ = signals(document, caption=refinement.label)

        result = classify(s)

        assert result.refinement == refinement.slug, (
            f"{refinement.label!r} refined to {result.refinement}"
        )
        assert result.filing_type == refinement.files_as
