"""Workflows for documents worth reading in detail.

    read → extract → identify → file (approval)

One definition, built four times. Salary slip, identity, property and tax intake
differ in exactly two things — which fields are worth pulling out, and what the
proposal is called — and both are arguments. Four hand-written definitions of
the same four steps would drift the first time one of them was fixed.

## What these do that the general reading workflow does not

`document_intake` reads a document, works out what it is, and files it. These
run *after* the type is already known, so they skip classification entirely and
spend the effort on extraction instead: employer and net pay from a salary slip,
registry number and area from a deed.

The fields are evidence for a reviewer, never a decision. Nothing extracted here
files anything, and the only one that influences anything at all is the CNIC,
which is offered to the client lookup as a candidate identifier.

## Client identification, in order of how much it is worth

    1. what the sender typed      exact, and a person said it
    2. a CNIC on the document     exact, and the document said it
    3. the number it arrived from a suggestion, never an answer

Tier 3 is marked weak and is never treated as unambiguous, whatever comes back.
A phone is shared within a family and an accountant forwards on a client's
behalf, so a single match on a mobile number is something for a reviewer to
confirm rather than something to act on.
"""

from __future__ import annotations

from typing import Any

from app.documents import registry
from app.workflow.document_intake import image_to_file, image_to_read, is_photograph, nobody_named_it
from app.memory.records import MemoryKind
from app.workflow.engine import Step, Workflow


def _text(ctx: dict[str, Any]) -> str:
    return str(ctx.get("read", {}).get("text") or "")


def _document_type(ctx: dict[str, Any]) -> str:
    """The type the classifier settled on before this workflow started.

    Passed in as context rather than recomputed. Re-classifying here would let
    a workflow disagree with the router that chose it, and the document would be
    filed as something other than what it was routed as.
    """
    return str(ctx.get("document_type") or "")


def _extract_input(ctx: dict[str, Any]) -> dict[str, Any]:
    return {"text": _text(ctx), "document_type": _document_type(ctx)}


def _identify_input(ctx: dict[str, Any]) -> dict[str, Any]:
    """Everything that might name the client, best first.

    All three go to the Tool together rather than being chosen here, because
    which one is usable depends on what the CMS finds — and the Tool already
    holds that precedence for the forwarding path. One place, one set of rules.
    """
    fields = ctx.get("extract", {}).get("fields", {})
    extracted_cnic = fields.get("cnic", {}).get("value") if isinstance(fields.get("cnic"), dict) else None
    extracted_mobile = fields.get("mobile", {}).get("value") if isinstance(fields.get("mobile"), dict) else None

    return {
        "file_number": ctx.get("file_number"),
        # What the sender typed wins over what the document says: a person
        # naming a client is better evidence than a number printed on a page
        # that may belong to a spouse, an employer or a previous owner.
        "cnic": ctx.get("cnic") or extracted_cnic,
        "mobile": extracted_mobile or ctx.get("sender"),
    }


def _match_reason(ctx: dict[str, Any]) -> str:
    """Why these candidates, stated from what was actually searched.

    A reviewer choosing between two people needs to know whether the match came
    from a number somebody typed, a CNIC printed on the document, or the phone
    it happened to arrive on. Those are very different levels of evidence and
    the third is barely evidence at all.
    """
    identifier = ctx.get("identify", {}).get("identifier", {})

    return {
        "file_number": "Matched the file number in the message",
        "cnic": "Matched a CNIC exactly",
        "mobile": "Matched the number this arrived from — confirm this is the right client",
    }.get(str(identifier.get("type")), "Matched on the supplied identifier")


def _warnings(ctx: dict[str, Any]) -> list[str]:
    warnings: list[str] = []
    identify = ctx.get("identify", {})
    matches = identify.get("matches") or []

    if not ctx.get("read", {}).get("readable", False):
        warnings.append("The document could not be read.")

    if identify.get("weak") and matches:
        # The single most useful thing this workflow can tell a reviewer. A
        # mobile match looks exactly like an exact one on screen otherwise.
        warnings.append(
            "The client was suggested by the phone number this arrived from, "
            "not by anything on the document. Confirm before approving."
        )

    if len(matches) > 1:
        warnings.append(f"{len(matches)} clients matched — choose one.")

    if not matches:
        warnings.append("No client matched. Search for the right one.")

    if not ctx.get("extract", {}).get("specific", False):
        warnings.append("No fields were extracted — this type carries none worth reading.")

    if reason := ctx.get("redo_reason"):
        warnings.append(f"Resubmitted after review: {reason}")

    return warnings


def _agent_events(ctx: dict[str, Any]) -> list[dict[str, Any]]:
    times = ctx.get("step_times") or {}
    took = ctx.get("step_durations") or {}

    return [
        {
            "event": event,
            "occurred_at": occurred,
            "message": message,
            "duration_ms": took.get(step),
        }
        for event, step, occurred, message in (
            ("ocr", "read", times.get("read"), ctx.get("read", {}).get("engine")),
            ("classified", "classify", times.get("classified"), ctx.get("classification", {}).get("reason")),
            ("extracted", "extract", times.get("extract"),
             ", ".join(ctx.get("extract", {}).get("found", [])) or None),
            ("identified", "identify", times.get("identify"), _match_reason(ctx)),
        )
        if occurred
    ]


def _proposal(ctx: dict[str, Any], title: str) -> dict[str, Any]:
    """The proposal a human reviews (ADR-0008).

    Ambiguity is carried in rather than failing the run — a run that failed here
    would throw away the reading, the extraction and the candidate list, and
    make somebody start again to correct one id.
    """
    identify = ctx.get("identify", {})
    matches = identify.get("matches") or []
    attempt = int(ctx.get("attempt", 1))
    run_id = ctx.get("run_id") or "run"
    document_type = _document_type(ctx)
    filing = registry.get(document_type)

    # A single match on a mobile number is a suggestion, not an answer, so the
    # client is left for a reviewer to choose even though exactly one came back.
    unambiguous = identify.get("unambiguous") and not identify.get("weak")

    return {
        "idempotency_key": f"{run_id}:file" if attempt == 1 else f"{run_id}:file:{attempt}",
        "workflow_id": run_id,
        "workflow_name": ctx.get("workflow_name", "extraction_intake"),
        "step_name": "file",
        "path": image_to_file(ctx),
        "document_type": document_type,
        "confidence": ctx.get("classification", {}).get("confidence"),
        "payload": {
            "client_id": matches[0].get("id") if unambiguous else None,
            "document_type": document_type,
            "title": title,
            "candidates": [
                {
                    "id": m.get("id"),
                    "name": m.get("name"),
                    "file_number": m.get("file_number"),
                    "match_reason": _match_reason(ctx),
                }
                for m in matches
            ],
            "identifier": identify.get("identifier", {}),
            "source": ctx.get("source") or {},
            # How the type was decided, which is the question a reviewer asks
            # first when the type looks wrong. Level 1 throughout — a method
            # name and a sentence, no client data.
            "classification": ctx.get("classification", {}),
        },
        "evidence": _evidence(ctx, filing),
        "events": _agent_events(ctx),
        "warnings": _warnings(ctx),
    }


def _evidence(ctx: dict[str, Any], filing: registry.FilingType | None) -> dict[str, Any]:
    """What the AI read, for the reviewer — unless it must not be kept.

    A type marked `reads_document = False` is one whose contents are Level 3
    and stay inside the installation (ADR-0002). Such a document normally never
    reaches this workflow at all, because the caption or the filename settles it
    first. But it *can*: a bank statement with no caption and a filename of
    `scan001.pdf` is only recognisable from its text.

    When that happens the reading is legitimate — OCR is local and nothing
    leaves — but keeping the text would put an account number and every
    transaction into a stored evidence blob for no benefit. So the text and the
    fields are dropped, and the fact that it was read is reported honestly.
    """
    read = ctx.get("read", {})

    if filing is not None and not filing.reads_document:
        return {
            "readable": bool(read.get("readable")),
            "read_attempted": True,
            "withheld": "This type's contents are not kept (ADR-0002).",
        }

    return {
        "readable": bool(read.get("readable")),
        "read_attempted": True,
        "text": _text(ctx),
        "fields": ctx.get("extract", {}).get("fields", {}),
        "engine": read.get("engine"),
    }


def _note(ctx: dict[str, Any]) -> dict[str, Any]:
    """A summary for a later run. Never a transcript.

    Memory is embedded and may be reasoned over by a hosted model, so the raw
    text stays out of it (ADR-0002). What is worth carrying forward is that a
    document of a type arrived and where it went.
    """
    filed = ctx.get("file", {})
    document_type = _document_type(ctx)

    return {
        "kind": MemoryKind.DOCUMENT,
        "content": f"Received and filed a {document_type.replace('_', ' ')}.",
        "client_id": filed.get("client_id"),
        "document_id": filed.get("document_id"),
        "workflow_id": ctx.get("run_id"),
        "confidence": ctx.get("classification", {}).get("confidence"),
        "metadata": {
            "document_type": document_type,
            "source": (ctx.get("source") or {}).get("channel", "whatsapp"),
            "approved": bool(filed.get("approved")),
        },
    }


def build(name: str, title: str, *, remember: bool = True) -> Workflow:
    """One extraction workflow.

    `remember` is off for identity documents. A memory note saying a CNIC
    arrived for a named client is a sentence about somebody's identity papers
    sitting in an embedded store, and the note buys nothing a later run needs.
    """
    steps = [
        Step(name="prepare", tool="prepare_image", optional=True,
             build_input=lambda ctx: {"path": ctx.get("path")},
             skip_when=lambda ctx: not is_photograph(ctx)),
        Step(name="read", tool="ocr_document", build_input=lambda ctx: {"path": image_to_read(ctx)}),
        Step(name="extract", tool="extract_document_fields", build_input=_extract_input),
        Step(name="identify", tool="find_client_by_identifier", build_input=_identify_input),
        Step(
            name="file",
            skip_when=nobody_named_it,
            tool="submit_filing_proposal",
            build_input=lambda ctx, _title=title: _proposal(ctx, _title),
        ),
    ]

    if remember:
        # Optional on purpose: by the time it runs the document is already
        # filed, and failing a run over a lost note would report a real success
        # as a failure.
        steps.append(Step(name="remember", tool="remember", build_input=_note, optional=True))

    return Workflow(name=name, steps=tuple(steps))


SALARY_SLIP_INTAKE = build("salary_slip_intake", "Salary Slip")
IDENTITY_INTAKE = build("identity_intake", "Identity Document", remember=False)
PROPERTY_INTAKE = build("property_intake", "Property Document")
TAX_DOCUMENT_INTAKE = build("tax_document_intake", "Tax Document")

ALL = (SALARY_SLIP_INTAKE, IDENTITY_INTAKE, PROPERTY_INTAKE, TAX_DOCUMENT_INTAKE)
