"""Workflows for documents worth reading in detail.

Two things carry the weight here.

**Extraction misses rather than guesses.** Every pattern anchors on a label
printed on the document, so wording nobody anticipated returns nothing for that
field. A blank is the correct failure: an extracted field is shown to a reviewer
as fact, and a plausible wrong number is worse than no number, because nobody
checks a field that is already filled in.

**A mobile match is never an answer.** It is the weakest way to identify a
client — phones are shared, accountants forward on a client's behalf — so a
single match on one still leaves the client for a reviewer to choose. That is
the difference between this tier and the two above it, and it is easy to lose.
"""

from __future__ import annotations

from pathlib import Path

import pytest

from app.api.client import AgentApiError, CmsClient
from app.config.settings import Settings
from app.documents import extractors
from app.memory.repository import InMemoryRepository
from app.ocr.engine import OcrResult, TextBlock
from app.tools.client_tools import FindClientByIdentifierTool
from app.tools.document_tools import ExtractDocumentFieldsTool, OcrTool, SubmitFilingProposalTool
from app.tools.memory_tools import RememberTool
from app.tools.registry import ToolRegistry
from app.workflow.engine import WorkflowEngine
from app.workflow.extraction_intake import (
    IDENTITY_INTAKE,
    PROPERTY_INTAKE,
    SALARY_SLIP_INTAKE,
    TAX_DOCUMENT_INTAKE,
)
from app.workflow.state import RunState

ALI = {"id": 42, "name": "Muhammad Ali", "file_number": "1420"}

SALARY_SLIP = """
ACME TEXTILES LIMITED
SALARY SLIP
Employer: Acme Textiles Limited
Employee Name: Muhammad Ali
Designation: Senior Accountant
Pay Period: July 2026
Gross Salary: Rs. 150,000
Total Deductions: 12,000
Net Pay: Rs. 138,000
"""

CNIC = """
NATIONAL IDENTITY CARD
Name: Muhammad Ali
Father Name: Abdul Rahman
Identity Number: 35202-1234567-1
Date of Birth: 01.01.1985
Date of Issue: 12.03.2019
Date of Expiry: 12.03.2029
"""

DEED = """
SALE DEED
Registry No: LHR-2026-4471
Vendor: Fatima Khan
Vendee: Muhammad Ali
Plot No: 42
Area: 10 Marla
Situated at: Model Town, Lahore
"""

TAX_CERTIFICATE = """
WITHHOLDING TAX CERTIFICATE
Tax Year 2025
NTN: 1234567-8
Certificate of: Tax Deducted at Source
Taxable Income: Rs. 1,200,000
Tax Deducted: 45,000
"""


class FakeOcr:
    name = "fake"

    def __init__(self, text: str) -> None:
        self._text = text

    def read(self, path: Path) -> OcrResult:  # noqa: ARG002
        if not self._text:
            return OcrResult(blocks=[], engine=self.name)

        return OcrResult(
            blocks=[TextBlock(text=line, confidence=0.94) for line in self._text.strip().splitlines()],
            engine=self.name,
        )


class FakeCms(CmsClient):
    def __init__(self, *, matches=None, by_mobile=None):
        super().__init__(
            Settings(
                cms_base_url="https://cms.test",
                agent_api_key="tpa_test",
                agent_api_secret="s" * 64,
            )
        )
        self.matches = [ALI] if matches is None else matches
        self.by_mobile = [ALI] if by_mobile is None else by_mobile
        self.submitted: list[dict] = []
        self.searches: list[tuple[str, str]] = []

    def _request(self, method, path, params=None, body=None):  # noqa: ANN001
        if path == "clients":
            search_type = (params or {}).get("search_type")
            self.searches.append((search_type, (params or {}).get("search")))
            found = self.by_mobile if search_type == "mobile" else self.matches

            return {"ok": True, "data": found, "meta": {"total": len(found)}}

        if path == "proposals" and method == "POST":
            record = dict(body or {})
            record["id"] = 900 + len(self.submitted) + 1
            self.submitted.append(record)

            return {"ok": True, "data": {"id": record["id"], "status": "pending",
                                         "risk_level": "medium"}}

        raise AssertionError(f"Unexpected call to {method} {path}")


@pytest.fixture
def document(tmp_path) -> Path:
    path = tmp_path / "scan.pdf"
    path.write_bytes(b"%PDF-1.4")

    return path


def build(cms: FakeCms, text: str) -> WorkflowEngine:
    registry = ToolRegistry()
    registry.register(OcrTool(FakeOcr(text)))
    registry.register(ExtractDocumentFieldsTool())
    registry.register(FindClientByIdentifierTool(cms))
    registry.register(SubmitFilingProposalTool(cms))
    registry.register(RememberTool(InMemoryRepository()))

    return WorkflowEngine(registry)


def run(workflow, cms, document, text, document_type: str = "salary_slip", **context):
    """Start a workflow the way the inbox does.

    `document_type` is always supplied in reality — the router decided it, and a
    workflow that re-decided it could disagree with the route that chose it — so
    it has a default here rather than being optional.
    """
    return build(cms, text).start(
        workflow,
        {
            "path": str(document),
            "workflow_name": workflow.name,
            "document_type": document_type,
            "classification": {"method": "caption", "confidence": 0.95,
                               "reason": "named in the message", "refinement": None},
            "source": {"channel": "whatsapp", "message_id": "WA-1"},
            **context,
        },
    )


# ── Extraction ────────────────────────────────────────────────────────────


class TestSalarySlip:
    def test_it_pulls_the_fields_a_reviewer_needs(self):
        fields = extractors.extract("salary_slip", SALARY_SLIP)

        assert fields["employer"]["value"] == "Acme Textiles Limited"
        assert fields["employee"]["value"] == "Muhammad Ali"
        assert fields["salary_period"]["value"] == "July 2026"
        assert fields["gross_salary"]["value"] == "150,000"
        assert fields["net_salary"]["value"] == "138,000"

    def test_gross_and_net_come_from_labels_not_from_size(self):
        """The tempting shortcut is picking the two largest numbers.

        It is wrong on any slip carrying a year-to-date total, which is larger
        than both and is neither.
        """
        with_ytd = SALARY_SLIP + "\nYear to Date Earnings: Rs. 1,800,000\n"

        fields = extractors.extract("salary_slip", with_ytd)

        assert fields["gross_salary"]["value"] == "150,000"
        assert fields["net_salary"]["value"] == "138,000"

    def test_wording_nobody_anticipated_returns_nothing_rather_than_a_guess(self):
        fields = extractors.extract("salary_slip", "REMUNERATION ADVICE\nEmoluments 90,000")

        assert "gross_salary" not in fields
        assert "net_salary" not in fields


class TestIdentity:
    def test_it_pulls_the_card_fields(self):
        fields = extractors.extract("cnic_front", CNIC)

        assert fields["cnic"]["value"] == "35202-1234567-1"
        assert fields["name"]["value"] == "Muhammad Ali"
        assert fields["father_name"]["value"] == "Abdul Rahman"
        assert fields["date_of_birth"]["value"] == "01.01.1985"
        assert fields["date_of_expiry"]["value"] == "12.03.2029"


class TestProperty:
    def test_it_pulls_the_deed_fields(self):
        fields = extractors.extract("property_document", DEED)

        assert fields["registry_number"]["value"] == "LHR-2026-4471"
        assert fields["owner"]["value"] == "Muhammad Ali"
        assert fields["area"]["value"] == "10 Marla"
        assert fields["plot_number"]["value"] == "42"

    def test_it_names_the_seller_as_well_as_the_buyer(self):
        # A deed names both. Showing only "owner" when the extractor took the
        # wrong line is how a document lands on the other party's file.
        fields = extractors.extract("property_document", DEED)

        assert fields["seller"]["value"] == "Fatima Khan"


class TestTaxDocuments:
    def test_it_pulls_the_tax_fields(self):
        fields = extractors.extract("tax_certificate", TAX_CERTIFICATE)

        assert fields["ntn"]["value"] == "1234567-8"
        assert fields["tax_year"]["value"] == "2025"
        assert fields["taxable_income"]["value"] == "1,200,000"


class TestTypesWithNothingToExtract:
    def test_a_receipt_has_no_specific_extractor(self):
        assert extractors.has_extractor("receipt") is False

    def test_the_general_fields_are_still_offered(self):
        # A mobile or a CNIC is useful on any document — the client lookup falls
        # back to a mobile when no exact identifier was supplied.
        fields = extractors.extract("receipt", "Paid. Contact 0300-1234567")

        assert fields["mobile"]["value"] == "03001234567"


# ── The workflows ─────────────────────────────────────────────────────────


class TestSalarySlipIntake:
    def test_it_reads_extracts_identifies_and_proposes(self, document):
        cms = FakeCms()
        result = run(SALARY_SLIP_INTAKE, cms, document, SALARY_SLIP, file_number="1420")

        assert result.state is RunState.AWAITING_APPROVAL

        submitted = cms.submitted[0]

        assert submitted["payload"]["document_type"] == "salary_slip"
        assert submitted["payload"]["client_id"] == 42
        assert submitted["evidence"]["fields"]["net_salary"]["value"] == "138,000"

    def test_the_proposal_says_how_the_type_was_decided(self, document):
        cms = FakeCms()
        run(SALARY_SLIP_INTAKE, cms, document, SALARY_SLIP, file_number="1420")

        classification = cms.submitted[0]["payload"]["classification"]

        # The first question a reviewer asks when a type looks wrong.
        assert classification["method"] == "caption"
        assert classification["reason"] == "named in the message"

    def test_a_typed_client_beats_one_printed_on_the_document(self, document):
        cms = FakeCms()
        run(SALARY_SLIP_INTAKE, cms, document, SALARY_SLIP, file_number="1420", cnic="3520212345671")

        # Both were offered; the file number is the firm's own key and wins.
        assert ("file_number", "1420") in cms.searches


class TestIdentityIntake:
    def test_a_cnic_identifies_its_own_client(self, document):
        cms = FakeCms()
        run(IDENTITY_INTAKE, cms, document, CNIC, document_type="cnic_front")

        # No caption identifier, so the number printed on the card was used.
        assert ("cnic", "35202-1234567-1") in cms.searches
        assert cms.submitted[0]["payload"]["client_id"] == 42

    def test_it_writes_no_memory_note(self, document):
        """A note saying whose identity papers arrived is a sentence about
        somebody's documents sitting in an embedded store, and buys nothing a
        later run needs."""
        assert "remember" not in [s.name for s in IDENTITY_INTAKE.steps]


class TestPropertyAndTaxIntake:
    def test_a_deed_proposes_with_its_fields(self, document):
        cms = FakeCms()
        run(PROPERTY_INTAKE, cms, document, DEED, document_type="property_document",
            file_number="1420")

        fields = cms.submitted[0]["evidence"]["fields"]

        assert fields["registry_number"]["value"] == "LHR-2026-4471"
        assert cms.submitted[0]["payload"]["title"] == "Property Document"

    def test_a_certificate_proposes_with_its_fields(self, document):
        cms = FakeCms()
        run(TAX_DOCUMENT_INTAKE, cms, document, TAX_CERTIFICATE,
            document_type="tax_certificate", file_number="1420")

        assert cms.submitted[0]["evidence"]["fields"]["ntn"]["value"] == "1234567-8"


class TestIdentifyingByPhoneIsNeverAnAnswer:
    def test_a_single_mobile_match_still_leaves_the_client_unchosen(self, document):
        """The rule this tier exists under.

        A file number matching one client is an answer. A mobile matching one
        client is a suggestion — phones are shared, and an accountant forwards
        on a client's behalf.
        """
        cms = FakeCms(matches=[], by_mobile=[ALI])

        run(SALARY_SLIP_INTAKE, cms, document, "SALARY SLIP\nContact 0300-1234567",
            sender="923001234567")

        payload = cms.submitted[0]["payload"]

        assert payload["client_id"] is None
        assert payload["candidates"][0]["id"] == 42

    def test_the_reviewer_is_told_the_match_is_weak(self, document):
        cms = FakeCms(matches=[], by_mobile=[ALI])

        run(SALARY_SLIP_INTAKE, cms, document, "SALARY SLIP\nContact 0300-1234567",
            sender="923001234567")

        warnings = cms.submitted[0]["warnings"]

        assert any("phone number" in w for w in warnings)

    def test_the_match_reason_says_so_too(self, document):
        cms = FakeCms(matches=[], by_mobile=[ALI])

        run(SALARY_SLIP_INTAKE, cms, document, "SALARY SLIP\nContact 0300-1234567",
            sender="923001234567")

        reason = cms.submitted[0]["payload"]["candidates"][0]["match_reason"]

        assert "confirm" in reason.lower()

    def test_an_exact_match_is_not_downgraded(self, document):
        cms = FakeCms()

        run(SALARY_SLIP_INTAKE, cms, document, SALARY_SLIP, file_number="1420")

        assert cms.submitted[0]["payload"]["client_id"] == 42


class TestEvidenceForATypeThatMustNotBeKept:
    def test_a_bank_statement_read_only_to_classify_keeps_no_text(self, document):
        """The case the caption and filename stages normally prevent.

        A bank statement with no caption and a filename of `scan001.pdf` is
        recognisable only from its text. Reading it is legitimate — OCR is local
        and nothing leaves — but keeping an account number and every transaction
        in a stored evidence blob buys nothing.
        """
        cms = FakeCms()

        run(TAX_DOCUMENT_INTAKE, cms, document,
            "STATEMENT OF ACCOUNT\nClosing Balance 45,000",
            document_type="bank_statement", file_number="1420")

        evidence = cms.submitted[0]["evidence"]

        assert "text" not in evidence
        assert "fields" not in evidence
        # Honest about what happened: it was opened, and the contents are not kept.
        assert evidence["read_attempted"] is True
        assert "not kept" in evidence["withheld"]

    def test_an_ordinary_type_keeps_its_evidence(self, document):
        cms = FakeCms()

        run(SALARY_SLIP_INTAKE, cms, document, SALARY_SLIP, file_number="1420")

        assert "Gross Salary" in cms.submitted[0]["evidence"]["text"]


class TestWhenTheDocumentCannotBeRead:
    def test_an_unreadable_document_still_reaches_a_reviewer(self, document):
        cms = FakeCms()

        result = run(SALARY_SLIP_INTAKE, cms, document, "", file_number="1420")

        assert result.state is RunState.AWAITING_APPROVAL
        assert any("could not be read" in w for w in cms.submitted[0]["warnings"])
