"""Document intake, end to end (ADR-0008).

A client sends a photograph on WhatsApp; it ends up filed against the right
client — but the filing happens in the CMS, after a human approves. This platform
reads the document, works out whose it is, asks, and waits.

The CMS is faked at the HTTP boundary — the real CmsClient, the real signing, the
real tools, the real engine, with only the network replaced. A fake tool would
test the test.
"""

from __future__ import annotations

from pathlib import Path

import pytest

from app.api.client import AgentApiError, CmsClient
from app.config.settings import Settings
from app.memory.records import MemoryKind
from app.memory.repository import InMemoryRepository
from app.ocr.engine import OcrResult, TextBlock
from app.tools.client_tools import SearchClientTool
from app.tools.document_tools import (
    ClassifyDocumentTool,
    ExtractFieldsTool,
    OcrTool,
    SubmitFilingProposalTool,
)
from app.tools.memory_tools import RememberTool
from app.tools.registry import ToolRegistry
from app.workflow.decisions import DecisionPoller
from app.workflow.document_intake import DOCUMENT_INTAKE
from app.workflow.engine import WorkflowEngine
from app.workflow.state import RunState, StepState

CNIC_TEXT = """
GOVERNMENT OF PAKISTAN
NATIONAL IDENTITY CARD
Name: Muhammad Ali
Identity Number: 35202-1234567-1
Date of Birth: 01.01.1985
"""


class FakeOcr:
    """An OCR engine that returns whatever the test decided the document says."""

    name = "fake"

    def __init__(self, text: str = CNIC_TEXT, confidence: float = 0.94) -> None:
        self._text = text
        self._confidence = confidence

    def read(self, path: Path) -> OcrResult:  # noqa: ARG002
        if not self._text:
            return OcrResult(blocks=[], engine=self.name)

        return OcrResult(
            blocks=[TextBlock(text=line, confidence=self._confidence)
                    for line in self._text.strip().splitlines()],
            engine=self.name,
        )


class FakeCms(CmsClient):
    """The real client with only the wire replaced.

    Subclassing rather than duck-typing on purpose: everything above the
    transport — signing, error mapping, payload shapes — is the code under test.
    """

    def __init__(self, *, matches: list[dict] | None = None, fail_submit: str | None = None):
        super().__init__(
            Settings(
                cms_base_url="https://cms.test",
                agent_api_key="tpa_test",
                agent_api_secret="s" * 64,
            )
        )
        self.matches = matches if matches is not None else [
            {"id": 42, "name": "Muhammad Ali", "file_number": "LA-0042"}
        ]
        self.submitted: list[dict] = []
        self.decisions: list[dict] = []
        self.fail_submit = fail_submit
        self.last_search_type: str | None = None
        self.last_search: str | None = None

    def _request(self, method, path, params=None, body=None):  # noqa: ANN001
        if path == "clients":
            self.last_search_type = (params or {}).get("search_type")
            self.last_search = (params or {}).get("search")

            return {"ok": True, "data": self.matches, "meta": {"total": len(self.matches)}}

        if path == "proposals" and method == "POST":
            if self.fail_submit:
                raise AgentApiError(self.fail_submit)

            key = (body or {}).get("idempotency_key")

            # The CMS returns the SAME proposal for a repeated key. Modelled here
            # because a workflow that resubmits with a stale key would silently
            # get back the answer a reviewer had already rejected.
            for existing in self.submitted:
                if existing["idempotency_key"] == key:
                    return {"ok": True, "data": {"id": existing["id"], "status": "pending"}}

            record = dict(body or {})
            record["id"] = 900 + len(self.submitted) + 1
            self.submitted.append(record)

            return {"ok": True, "data": {"id": record["id"], "status": "pending",
                                         "risk_level": "medium"}}

        if path == "proposals" and method == "GET":
            return {"ok": True, "data": self.decisions}

        raise AssertionError(f"Unexpected call to {method} {path}")

    # ── Standing in for a reviewer ────────────────────────────────────────

    def decide(self, proposal_id: int, status: str, **extra) -> None:
        self.decisions = [
            {"id": proposal_id, "status": status, "updated_at": "2026-07-29T10:00:00+00:00", **extra}
        ]


@pytest.fixture
def document(tmp_path) -> Path:
    path = tmp_path / "cnic.jpg"
    path.write_bytes(b"pretend-image-bytes")

    return path


def build(cms: FakeCms, ocr: FakeOcr | None = None, memory=None):
    registry = ToolRegistry()
    registry.register(OcrTool(ocr or FakeOcr()))
    registry.register(ClassifyDocumentTool())
    registry.register(ExtractFieldsTool())
    registry.register(SearchClientTool(cms))
    registry.register(SubmitFilingProposalTool(cms))
    registry.register(RememberTool(memory if memory is not None else InMemoryRepository()))

    return WorkflowEngine(registry)


def poller(cms: FakeCms, engine: WorkflowEngine) -> DecisionPoller:
    return DecisionPoller(cms, engine, DOCUMENT_INTAKE)


class TestSubmitting:
    def test_a_document_is_proposed_not_filed(self, document):
        cms = FakeCms()
        run = build(cms).start(DOCUMENT_INTAKE, {"path": str(document), "sender": "923001234567"})

        # The whole design in one assertion: everything up to the write happened
        # autonomously, and the write did not happen here at all.
        assert run.state is RunState.AWAITING_APPROVAL
        assert len(cms.submitted) == 1

    def test_it_read_classified_and_identified_first(self, document):
        run = build(FakeCms()).start(
            DOCUMENT_INTAKE, {"path": str(document), "sender": "923001234567"}
        )

        assert run.step("read").state is StepState.SUCCEEDED
        assert run.context["classify"]["document_type"] == "cnic_front"
        assert run.context["extract"]["fields"]["cnic"]["value"] == "35202-1234567-1"

    def test_the_run_holds_the_proposal_it_is_waiting_on(self, document):
        cms = FakeCms()
        run = build(cms).start(DOCUMENT_INTAKE, {"path": str(document)})

        # Without this the poller cannot match a decision back to a run.
        assert run.step("file").output["proposal_id"] == cms.submitted[0]["id"]

    def test_a_failed_submission_does_not_pause_the_run(self, document):
        cms = FakeCms(fail_submit="TaxPilot CMS is unreachable: timed out")
        run = build(cms).start(DOCUMENT_INTAKE, {"path": str(document)})

        # Nothing is waiting on a decision that was never asked for. The run
        # fails, visibly, rather than hanging forever.
        assert run.state is RunState.FAILED


class TestWhatTheReviewerGets:
    def test_the_proposal_carries_the_document(self, document):
        import base64

        cms = FakeCms()
        build(cms).start(DOCUMENT_INTAKE, {"path": str(document)})

        assert base64.b64decode(cms.submitted[0]["file_base64"]) == b"pretend-image-bytes"

    def test_it_carries_the_text_and_the_fields(self, document):
        cms = FakeCms()
        build(cms).start(DOCUMENT_INTAKE, {"path": str(document)})
        evidence = cms.submitted[0]["evidence"]

        # A reviewer deciding whether to trust a match needs to see what it was
        # based on. This is Level 3 and stays inside the installation — the CMS
        # is not across that boundary, it is the boundary.
        assert "NATIONAL IDENTITY CARD" in evidence["text"]
        assert evidence["fields"]["cnic"]["value"] == "35202-1234567-1"
        assert evidence["readable"] is True

    def test_it_says_why_each_candidate_matched(self, document):
        cms = FakeCms()
        build(cms).start(DOCUMENT_INTAKE, {"path": str(document)})

        # A CNIC printed on the document and a phone number the message happened
        # to arrive from are very different levels of evidence.
        assert cms.submitted[0]["payload"]["candidates"][0]["match_reason"] == (
            "CNIC on the document matched exactly"
        )

    def test_a_weaker_match_says_so(self, tmp_path):
        path = tmp_path / "slip.jpg"
        path.write_bytes(b"x")
        cms = FakeCms()

        build(cms, FakeOcr("SALARY SLIP\nBasic Pay 120,000\nGross Salary 150,000")).start(
            DOCUMENT_INTAKE, {"path": str(path), "sender": "923001234567"}
        )

        assert cms.submitted[0]["payload"]["candidates"][0]["match_reason"] == (
            "Matched on the number the message came from"
        )

    def test_it_replays_the_steps_that_ran_here(self, document):
        cms = FakeCms()
        build(cms).start(DOCUMENT_INTAKE, {"path": str(document)})
        events = [e["event"] for e in cms.submitted[0]["events"]]

        # So the trail a person reads covers the whole chain rather than starting
        # at the moment the proposal arrived.
        assert events == ["ocr", "classified", "extracted", "identified"]
        assert all(e["occurred_at"] for e in cms.submitted[0]["events"])

    def test_ambiguity_is_flagged_as_a_warning(self, document):
        cms = FakeCms(matches=[
            {"id": 42, "name": "Muhammad Ali", "file_number": "LA-0042"},
            {"id": 77, "name": "Muhammad Ali", "file_number": "LA-0077"},
        ])
        build(cms).start(DOCUMENT_INTAKE, {"path": str(document)})

        assert cms.submitted[0]["payload"]["client_id"] is None
        assert "2 clients matched — choose one." in cms.submitted[0]["warnings"]

    def test_no_match_is_flagged_too(self, document):
        cms = FakeCms(matches=[])
        build(cms).start(DOCUMENT_INTAKE, {"path": str(document)})

        assert "No client matched. Search for the right one." in cms.submitted[0]["warnings"]

    def test_an_unreadable_document_reaches_a_human_without_a_type(self, tmp_path):
        """Reverses an earlier decision, on the evidence of real documents.

        This asserted the opposite: unclassifiable meant the run failed and
        nothing reached the queue, because asking a reviewer to pick a type from
        a dropdown "is not a proposal".

        What it did in practice was throw the whole run away. Seen on a real
        CNIC: the number was read, the client was identified as a specific
        person, the image was on disk — and because the classifier could not
        name an old all-Urdu card, none of it reached anybody. The customer put
        a document in their self-chat and nothing happened, with no error
        anywhere to look at.

        Choosing the type is exactly what a person looking at the document can
        do and this cannot, and the CMS already lets a reviewer amend
        `document_type` when they approve. So it goes to them carrying its own
        uncertainty — the same answer _propose_filing already gives for an
        ambiguous identification.
        """
        path = tmp_path / "blurry.jpg"
        path.write_bytes(b"x")
        cms = FakeCms()

        run = build(cms, FakeOcr("")).start(
            DOCUMENT_INTAKE, {"path": str(path), "sender": "923001234567"}
        )

        assert run.state is not RunState.FAILED
        assert len(cms.submitted) == 1

        submitted = cms.submitted[0]

        # No invented type — not on the submission, and not in the payload the
        # CMS reads when the reviewer approves.
        assert submitted["payload"]["document_type"] is None
        assert any("type could not be determined" in w for w in submitted["warnings"])


class TestDecisions:
    def test_an_approved_filing_completes_the_run(self, document):
        cms = FakeCms()
        engine = build(cms)
        run = engine.start(DOCUMENT_INTAKE, {"path": str(document)})

        cms.decide(run.step("file").output["proposal_id"], "executed", result_id=901)
        poller(cms, engine).poll()

        # Reloaded, not the object the test is holding. The poller works on what
        # the store handed it, exactly as it will against Postgres — so this
        # asserts the outcome was persisted, not merely computed.
        run = engine.store.load(run.id)

        assert run.state is RunState.COMPLETED
        assert run.context["file"]["document_id"] == 901

    def test_it_remembers_what_happened(self, document):
        memory = InMemoryRepository()
        cms = FakeCms()
        engine = build(cms, memory=memory)
        run = engine.start(DOCUMENT_INTAKE, {"path": str(document)})

        cms.decide(run.step("file").output["proposal_id"], "executed", result_id=901)
        poller(cms, engine).poll()

        remembered = memory.recall("cnic document filed", client_id=42)

        assert len(remembered) == 1
        assert remembered[0][0].kind is MemoryKind.DOCUMENT

    def test_a_reviewers_correction_reaches_the_memory_note(self, document):
        memory = InMemoryRepository()
        cms = FakeCms()
        engine = build(cms, memory=memory)
        run = engine.start(DOCUMENT_INTAKE, {"path": str(document)})

        cms.decide(
            run.step("file").output["proposal_id"], "executed",
            result_id=901, amendments={"client_id": 77},
        )
        poller(cms, engine).poll()

        # The note must describe where the document actually went, not where the
        # AI proposed sending it.
        assert memory.recall("document", client_id=77)

    def test_a_rejection_ends_the_run_with_its_reason(self, document):
        cms = FakeCms()
        engine = build(cms)
        run = engine.start(DOCUMENT_INTAKE, {"path": str(document)})

        cms.decide(
            run.step("file").output["proposal_id"], "rejected",
            decision_note="Blurry — ask the client to resend.",
        )
        poller(cms, engine).poll()
        run = engine.store.load(run.id)

        # A rejection is a decision, not a malfunction — but it is terminal, and
        # recording it otherwise would leave the run looking like it might yet
        # complete.
        assert run.state is RunState.FAILED
        assert "Blurry" in run.error

    def test_an_undecided_proposal_leaves_the_run_waiting(self, document):
        cms = FakeCms()
        engine = build(cms)
        run = engine.start(DOCUMENT_INTAKE, {"path": str(document)})

        cms.decide(run.step("file").output["proposal_id"], "approved")
        outcomes = poller(cms, engine).poll()

        # Approved but not yet filed. Advancing here would claim an outcome the
        # CMS has not reached.
        assert run.state is RunState.AWAITING_APPROVAL
        assert outcomes[0].applied is False

    def test_a_decision_for_an_unknown_run_is_ignored(self, document):
        cms = FakeCms()
        engine = build(cms)
        engine.start(DOCUMENT_INTAKE, {"path": str(document)})

        cms.decide(99999, "executed", result_id=1)

        # Most often a run lost in a restart. The document is filed either way
        # and the CMS holds the record; crashing over it would help nobody.
        assert poller(cms, engine).poll() == []

    def test_an_unreachable_cms_is_survived(self, document):
        cms = FakeCms()
        engine = build(cms)
        engine.start(DOCUMENT_INTAKE, {"path": str(document)})

        def unreachable(*args, **kwargs):
            raise AgentApiError("TaxPilot CMS is unreachable")

        cms._request = unreachable  # noqa: SLF001

        # A long-running poller must not die because the CMS restarted.
        assert poller(cms, engine).poll() == []


class TestSendingItBack:
    def _submitted_and_sent_back(self, document, directive: str):
        cms = FakeCms()
        engine = build(cms)
        run = engine.start(DOCUMENT_INTAKE, {"path": str(document)})

        cms.decide(
            run.step("file").output["proposal_id"], "changes_requested",
            directive=directive, decision_note="Try again.",
        )
        poller(cms, engine).poll()

        return cms, run

    def test_reading_it_again_reruns_everything_downstream(self, document):
        cms, run = self._submitted_and_sent_back(document, "rerun_ocr")

        # A different reading changes the classification, the fields and possibly
        # the client, so all of it runs again.
        assert run.state is RunState.AWAITING_APPROVAL
        assert len(cms.submitted) == 2

    def test_looking_for_the_client_again_keeps_the_reading(self, document):
        cms, run = self._submitted_and_sent_back(document, "rerun_client_search")

        # Nothing about the OCR was questioned, so its result stands.
        assert run.step("read").state is StepState.SUCCEEDED
        assert run.step("read").attempts == 1
        assert len(cms.submitted) == 2

    def test_a_resubmission_uses_a_new_key(self, document):
        cms, _ = self._submitted_and_sent_back(document, "rerun_ocr")

        keys = [s["idempotency_key"] for s in cms.submitted]

        # Reusing the key returns the proposal the reviewer just sent back —
        # the CMS is idempotent, and that is exactly the trap.
        assert keys[0] != keys[1]
        assert cms.submitted[0]["id"] != cms.submitted[1]["id"]

    def test_the_reviewers_reason_reaches_the_new_proposal(self, document):
        cms, _ = self._submitted_and_sent_back(document, "rerun_ocr")

        assert any("Try again." in w for w in cms.submitted[1]["warnings"])

    def test_a_send_back_with_no_directive_starts_from_the_top(self, document):
        cms, run = self._submitted_and_sent_back(document, "")

        assert len(cms.submitted) == 2
        assert run.state is RunState.AWAITING_APPROVAL


class TestIdentification:
    def test_the_cnic_on_the_document_beats_the_sender_number(self, document):
        cms = FakeCms()
        build(cms).start(DOCUMENT_INTAKE, {"path": str(document), "sender": "923009999999"})

        assert cms.last_search_type == "cnic"
        assert cms.last_search == "35202-1234567-1"

    def test_a_document_identifying_nobody_fails_rather_than_guessing(self, tmp_path):
        path = tmp_path / "blank.jpg"
        path.write_bytes(b"x")

        run = build(FakeCms(), FakeOcr("SALARY SLIP\nBasic Pay 120,000")).start(
            DOCUMENT_INTAKE, {"path": str(path)}   # no sender either
        )

        assert run.state is RunState.FAILED
        assert "identifies a client" in run.error


class TestResumability:
    def test_a_resumed_run_does_not_submit_twice(self, document):
        cms = FakeCms()
        engine = build(cms)
        run = engine.start(DOCUMENT_INTAKE, {"path": str(document)})

        engine.resume(DOCUMENT_INTAKE, run)
        engine.resume(DOCUMENT_INTAKE, run)

        # A paused step is not re-run, so a reviewer never sees the same document
        # twice in their queue.
        assert len(cms.submitted) == 1

    def test_a_decision_applied_twice_files_one_memory_note(self, document):
        memory = InMemoryRepository()
        cms = FakeCms()
        engine = build(cms, memory=memory)
        run = engine.start(DOCUMENT_INTAKE, {"path": str(document)})

        cms.decide(run.step("file").output["proposal_id"], "executed", result_id=901)
        p = poller(cms, engine)
        p.poll()
        p.poll()

        assert len(memory) == 1

    def test_losing_the_memory_note_does_not_fail_a_filed_document(self, document):
        class BrokenMemory(InMemoryRepository):
            def remember(self, record):
                raise RuntimeError("Postgres is down")

        cms = FakeCms()
        engine = build(cms, memory=BrokenMemory())
        run = engine.start(DOCUMENT_INTAKE, {"path": str(document)})

        cms.decide(run.step("file").output["proposal_id"], "executed", result_id=901)
        poller(cms, engine).poll()
        run = engine.store.load(run.id)

        # The document is filed by this point. Failing the run over a note would
        # report a real success as a failure and invite someone to re-run it.
        assert run.state is RunState.COMPLETED
        assert run.step("remember").state is StepState.SKIPPED


class TestDataPolicy:
    """ADR-0002, at the place this workflow could break it."""

    def _filed(self, document, memory):
        cms = FakeCms()
        engine = build(cms, memory=memory)
        run = engine.start(DOCUMENT_INTAKE, {"path": str(document)})
        cms.decide(run.step("file").output["proposal_id"], "executed", result_id=901)
        poller(cms, engine).poll()

    def test_the_memory_note_carries_no_cnic(self, document):
        memory = InMemoryRepository()
        self._filed(document, memory)

        stored = memory.recall("document", client_id=42)[0][0]

        # Memory is embedded and may be sent to a hosted model to reason over.
        # A CNIC here leaves the installation the first time anyone asks a
        # question about this client.
        assert "35202-1234567-1" not in stored.content

    def test_the_memory_note_carries_no_raw_document_text(self, document):
        memory = InMemoryRepository()
        self._filed(document, memory)

        stored = memory.recall("document", client_id=42)[0][0]

        assert "NATIONAL IDENTITY CARD" not in stored.content
        assert len(stored.content) < 200


class TestConfidenceReachesTheReviewer:
    """What the classifier knew has to survive the trip to the queue.

    A tool reports its confidence alongside its data rather than inside it, so
    it landed on the step record and went no further. The proposal reads its
    confidence out of the context, so every document arrived carrying none —
    which meant every one of them scored high risk on the "we do not know how
    sure it was" branch, and the queue's confidence column showed "—" for
    everything. The classifier had been perfectly sure. Nothing downstream could
    tell.
    """

    def test_a_step_publishes_its_confidence_to_later_steps(self, document):
        cms = FakeCms()
        run = build(cms).start(DOCUMENT_INTAKE, {"path": str(document)})

        assert run.context["classify"]["confidence"] > 0

    def test_the_proposal_carries_it(self, document):
        cms = FakeCms()
        build(cms).start(DOCUMENT_INTAKE, {"path": str(document)})

        assert cms.submitted[0]["confidence"] > 0

    def test_it_matches_what_the_step_recorded(self, document):
        # The step record and the context must not disagree — a reviewer and an
        # operations screen reading the same run should see one number.
        cms = FakeCms()
        run = build(cms).start(DOCUMENT_INTAKE, {"path": str(document)})

        assert run.step("classify").confidence == run.context["classify"]["confidence"]

    def test_an_unreadable_document_still_reports_a_confidence(self, tmp_path):
        # Zero is an answer. None is the absence of one, and the risk rule
        # treats those very differently.
        path = tmp_path / "blurry.jpg"
        path.write_bytes(b"x")
        cms = FakeCms()

        build(cms, FakeOcr("")).start(
            DOCUMENT_INTAKE, {"path": str(path), "sender": "923001234567"}
        )

        assert cms.submitted[0]["confidence"] == 0.0

    def test_a_tool_that_reports_its_own_confidence_keeps_it(self):
        """The engine fills a gap; it does not overwrite an answer.

        A tool putting a confidence inside its own data means something more
        specific by it than "how sure was this step".
        """
        from app.tools.base import ToolResult
        from app.workflow.engine import WorkflowEngine

        result = ToolResult.success({"confidence": 0.11}, confidence=0.99)
        published = dict(result.data)
        published.setdefault("confidence", result.confidence)

        assert published["confidence"] == 0.11
