"""Bank statement intake — filing a document nobody reads.

A firm forwards a bank statement to their own WhatsApp self-chat and writes the
client's file number or CNIC in the caption. It ends up on that client's record,
after a human approves it.

    identify → file (approval) → done

Two steps, against document intake's six. Everything that is missing is missing
on purpose, and the omissions are the design:

**No OCR.** A bank statement is Level 3 (ADR-0002) — account numbers, balances,
every transaction. It is not read here, not extracted from, and not sent
anywhere. The file is treated as an opaque attachment from the moment it arrives
until the CMS writes it to disk.

**No classification.** The type is known: somebody forwarded a bank statement.
Running a classifier over text nobody extracted would produce "unknown" and put a
question in front of a reviewer that has already been answered.

**No memory note.** Document intake writes one because a later run benefits from
knowing a document arrived. Memory is embedded and may be reasoned over by a
hosted model, and there is nothing about this filing worth carrying forward that
is not already in the proposal and the client's timeline.

**No sender fallback.** Document intake falls back to identifying the client by
the number the message came from. Here the message always comes from the firm's
own account — that is what a self-chat is — so the sender identifies nobody, and
a fallback would resolve every forward to the same wrong client.

## Why the confidence is 1.0 and it is not a lie

Confidence here is not a model's estimate. Nothing inferred anything: a person
typed an identifier and the CMS found exactly one client holding it. What the
number reports is *how much of this was guesswork*, and the honest answer is
none. The risk that remains — that they typed the wrong number — is not
uncertainty this system can measure, and is exactly what the reviewer is for.
"""

from __future__ import annotations

from typing import Any

from app.workflow.engine import Step, Workflow

INTAKE = "bank_statement_intake"

#: What every document filed through this workflow is.
#:
#: Fixed rather than parsed from the caption. A reviewer can change it before
#: approving — the CMS has allowed that since the queue was built — and reading a
#: type out of free text would be a second thing to get wrong on a path whose
#: whole premise is that the human already said what they meant.
DOCUMENT_TYPE = "bank_statement"


def _identify(ctx: dict[str, Any]) -> dict[str, Any]:
    """Look the client up from what the sender typed.

    Both identifiers go to the Tool rather than one being chosen here. Choosing
    would mean deciding precedence before knowing whether the two agree, and
    whether they agree is the most useful thing either of them can tell us.

    The number it arrived from is offered as a last resort, below both. This
    workflow is no longer reached only by a captioned forward — since the router
    began deciding from the classification, a statement recognised by its
    filename or its text arrives here with no caption at all. Without a fallback
    it had nothing to search on.
    """
    return {
        "file_number": ctx.get("file_number"),
        "cnic": ctx.get("cnic"),
        "mobile": ctx.get("sender"),
    }


def _propose_filing(ctx: dict[str, Any]) -> dict[str, Any]:
    """Build the proposal a human will review (ADR-0008).

    Ambiguity is carried in rather than failing the run, for document intake's
    reason: a run that failed here would throw the forward away, and the person
    who sent it would see nothing happen with nothing to look at.
    """
    identify = ctx.get("identify", {})
    matches = identify.get("matches") or []
    attempt = int(ctx.get("attempt", 1))
    run_id = ctx.get("run_id") or "run"
    source = ctx.get("source") or {}

    return {
        # A fresh key per attempt, so a resubmission after a reviewer sent it
        # back returns the new answer rather than the one they just refused.
        "idempotency_key": f"{run_id}:file" if attempt == 1 else f"{run_id}:file:{attempt}",
        "workflow_id": run_id,
        "workflow_name": INTAKE,
        "step_name": "file",
        "path": ctx.get("path"),
        "document_type": DOCUMENT_TYPE,
        # Not an inference. See the module docstring.
        "confidence": 1.0,
        "payload": {
            "client_id": matches[0].get("id") if identify.get("unambiguous") else None,
            "document_type": DOCUMENT_TYPE,
            "title": "Bank Statement",
            "candidates": [
                {
                    "id": m.get("id"),
                    "name": m.get("name"),
                    "file_number": m.get("file_number"),
                    "match_reason": _match_reason(identify, m),
                }
                for m in matches
            ],
            "identifier": identify.get("identifier", {}),
            "source": source,
            # How the type was decided, the same as every other workflow
            # reports it. Easy to leave out here, because this path predates the
            # classifier — and the case a reviewer most needs it for lands here:
            # a file called `SalarySlip_July.pdf` recognised as a bank statement
            # from its text is exactly when "how did it decide that?" is the
            # first question.
            "classification": ctx.get("classification", {}),
        },
        "evidence": _evidence(ctx),
        "events": _agent_events(ctx),
        "warnings": _warnings(ctx, matches),
    }


def _evidence(ctx: dict[str, Any]) -> dict[str, Any]:
    """What was read — which is usually nothing, and must not claim so wrongly.

    `read_attempted: false` was hardcoded here, and was true for as long as this
    workflow was reached only by a captioned forward. It stopped being true when
    the router began deciding from the classification: a statement recognised
    from its *text* was opened by the classifier before anything routed it, and
    reporting that as "never read" is a false claim on the one screen where a
    reviewer is deciding how much to trust the filing.

    Either way the contents are not kept. That is the guarantee — ADR-0002 is
    about a bank statement's transactions not being stored or sent, not about
    the file never having been opened locally — and it holds in both branches.
    """
    was_read = bool(ctx.get("classification", {}).get("was_read"))

    if not was_read:
        return {"read_attempted": False}

    return {
        "read_attempted": True,
        "readable": True,
        "withheld": "This type's contents are not kept (ADR-0002).",
    }


def _match_reason(identify: dict[str, Any], match: dict[str, Any]) -> str:
    """Why this client is a candidate.

    Stated from what was actually searched. A reviewer choosing between two
    people needs to know which number produced which name — that is the entire
    content of the disagreement they are being asked to settle.
    """
    identifier = identify.get("identifier", {})

    if identify.get("conflict"):
        by_file = match.get("file_number") and str(match["file_number"]) == str(identifier.get("value"))

        return (
            "Matched the file number in the message"
            if by_file
            else "Matched the CNIC in the message"
        )

    return "Matched using user supplied identifier"


def _warnings(ctx: dict[str, Any], matches: list[dict[str, Any]]) -> list[str]:
    """What a reviewer should be told before they look at anything else."""
    warnings: list[str] = []
    identify = ctx.get("identify", {})

    if identify.get("conflict"):
        warnings.append(
            "The file number and the CNIC in this message identify different clients. "
            "Neither was chosen."
        )
    elif len(matches) > 1:
        warnings.append(f"{len(matches)} clients matched that identifier — choose one.")
    elif not matches:
        warnings.append(
            "No client matched the identifier in the message. Search for the right one."
        )

    if reason := ctx.get("redo_reason"):
        warnings.append(f"Resubmitted after review: {reason}")

    return warnings


def _agent_events(ctx: dict[str, Any]) -> list[dict[str, Any]]:
    """The steps that ran here, for the CMS to replay into its own history.

    Two of them, and the first is not a step at all: `received` is the moment the
    message arrived, which is what somebody asking "when did this come in?" wants
    and is otherwise lost — the proposal's own timestamp is when the run finished,
    not when the file was sent.
    """
    times = ctx.get("step_times") or {}
    took = ctx.get("step_durations") or {}
    source = ctx.get("source") or {}
    identify = ctx.get("identify", {})

    events = [
        ("received", None, source.get("received_at"), "Forwarded to the firm's WhatsApp self-chat."),
        ("identified", "identify", times.get("identify"), _identified_message(identify)),
    ]

    return [
        {
            "event": event,
            "occurred_at": occurred,
            "message": message,
            "duration_ms": took.get(step) if step else None,
        }
        for event, step, occurred, message in events
        if occurred
    ]


def _identified_message(identify: dict[str, Any]) -> str:
    """One line on what the lookup found. No identifier value — this is stored.

    The number itself is on the review screen, where the person looking at it has
    already been authorised to see that client. An event message is replayed into
    a history that a wider set of people read.
    """
    matches = identify.get("matches") or []

    if identify.get("conflict"):
        return "Two identifiers named different clients."

    if len(matches) == 1:
        return "One client matched the supplied identifier."

    return f"{len(matches)} clients matched the supplied identifier."


BANK_STATEMENT_INTAKE = Workflow(
    name=INTAKE,
    steps=(
        Step(name="identify", tool="find_client_by_identifier", build_input=_identify),
        # Submits to the Approval Queue and waits. The CMS files it once a human
        # approves (ADR-0008); this step's result arrives from there rather than
        # from running the tool a second time.
        Step(name="file", tool="submit_filing_proposal", build_input=_propose_filing),
    ),
)
