"""Document intake — the workflow everything else was built for.

A client sends a photograph of a document on WhatsApp. It has to end up filed
against the right client, under the right type, with a note that says what
happened. That is this:

    read → classify → extract → identify → file (approval) → remember

Written as a definition rather than a function so the engine owns sequencing:
each step is persisted as it completes, so a restart resumes rather than
re-filing, and the one step that writes to a client's record stops for a human.

The interesting decisions are all about *not* being confident. A classifier that
returns "unknown" and an identification that returns three candidates are normal
outcomes, and the workflow has to reach a human without pretending otherwise.
"""

from __future__ import annotations

from typing import Any

from app.memory.records import MemoryKind
from app.ocr.classification import UNKNOWN
from app.workflow.engine import Step, Workflow

INTAKE = "document_intake"


def is_photograph(ctx: dict[str, Any]) -> bool:
    """Whether this document is the kind the preprocessing stage is for.

    A PDF is not a photograph of anything, and the registry can turn the stage
    off per type — so a new type opts in by declaration rather than by an engine
    change.
    """
    import os

    from app.documents import registry

    # Off unless a deployment asks for it.
    #
    # Cropping decides what becomes the permanent client document, and on real
    # photographs it currently succeeds on 3 in 11 — the rest fall back
    # untouched, which is safe but unproven at scale. A switch means staging can
    # run it against real traffic while a customer's installation does not, on
    # the same code, without a separate build.
    #
    # Default off rather than on: a deployment that has not been told to crop
    # should behave exactly as it did before this feature existed.
    if os.getenv("TAXPILOT_IMAGE_PREPROCESSING", "").strip().lower() not in {"1", "true", "yes", "on"}:
        return False

    suffix = str(ctx.get("path") or "").lower()

    if not suffix.endswith((".jpg", ".jpeg", ".png")):
        return False

    filing = registry.get(str(ctx.get("document_type") or ""))

    return filing is None or filing.preprocess


def image_to_read(ctx: dict[str, Any]) -> str | None:
    """The image OCR should read: always the original.

    Measured on real photographs, 1 August 2026: reading the crop instead of the
    source returned 68 characters at 0.67 confidence where the original returned
    166 at 0.82 — and with it no CNIC, no type, and no client. Cropping to the
    card necessarily discards pixels, and a recogniser needs pixels per
    character far more than it needs a tidy frame.

    So the reader gets everything that was captured.
    """
    return ctx.get("path")


def image_to_file(ctx: dict[str, Any]) -> str | None:
    """The image to put in the client's record: the crop when there is one.

    Deliberately NOT the same image the reader sees, which is a rule worth
    stating because the two were one path until today. They serve different
    ends: OCR wants every pixel that was captured, and a person opening a
    client's file wants the document rather than a photograph of a desk.

    The risk of them differing is that the evidence attached to a proposal came
    from an image the reviewer is not looking at. That is acceptable only
    because a crop is refused unless it is confidently the whole document —
    every gate in app/images/prepare.py exists to make "the crop contains the
    same document" true before this divergence is allowed.
    """
    prepared = (ctx.get("prepare") or {}).get("path")

    return str(prepared) if prepared else ctx.get("path")


def _text(ctx: dict[str, Any]) -> str:
    return str(ctx.get("read", {}).get("text") or "")


def _identify_query(ctx: dict[str, Any]) -> dict[str, Any]:
    """Whose document is this?

    A CNIC first: it is exact, and an exact match is the difference between
    filing a document and guessing. The sender's number is the fallback, which is
    weaker — phones are shared and an accountant may forward on a client's
    behalf — so the result is treated as a candidate list, never as an answer.

    Both stay inside the installation: this query goes to the CMS, which is
    Level 3's boundary, not across it (ADR-0002).
    """
    fields = ctx.get("extract", {}).get("fields", {})

    if cnic := fields.get("cnic"):
        return {"query": str(cnic["value"]), "search_type": "cnic"}

    if mobile := fields.get("mobile"):
        return {"query": str(mobile["value"]), "search_type": "mobile"}

    if sender := ctx.get("sender"):
        return {"query": str(sender), "search_type": "mobile"}

    raise ValueError("Nothing in the document or the message identifies a client.")


def nobody_named_it(ctx: dict[str, Any]) -> bool:
    """Whether to stop before proposing, and ask whose this is instead.

    The fallback half of OCR_WITH_METADATA_FALLBACK. The document was read, the
    reading named nobody, and there is a person on the other end of a WhatsApp
    conversation who knows the answer — so ask them, rather than handing a
    reviewer a proposal with an empty client field.

    Two conditions, and the second matters more than it looks.
    ``may_wait_for_metadata`` is set only on the first pass. Once an identifier
    has been supplied and *still* named nobody, the document goes to a reviewer:
    asking again for something already answered is how a document waits for ever.
    """
    if not ctx.get("may_wait_for_metadata"):
        return False

    return not (ctx.get("identify") or {}).get("unambiguous")


def _propose_filing(ctx: dict[str, Any]) -> dict[str, Any]:
    """Build the proposal a human will review (ADR-0008).

    Deliberately does *not* refuse when identification was ambiguous. A run that
    failed here would throw away the OCR, the classification and the candidate
    list, and a human would have to start again to correct one id. Instead the
    ambiguity is carried into the proposal: ``client_id`` is None, the candidates
    are attached, and the reviewer picks one.

    Everything the reviewer needs to judge it goes with it — the text that was
    read, the fields extracted, why each candidate matched. That data is Level 3
    and stays inside the installation, which is where it is going: the CMS is not
    across the boundary, it is the boundary.
    """
    identify = ctx.get("identify", {})
    classify = ctx.get("classify", {})
    matches = identify.get("matches") or []
    attempt = int(ctx.get("attempt", 1))
    run_id = ctx.get("run_id") or "run"

    return {
        # A fresh key per attempt. Reusing one returns the proposal the reviewer
        # has already seen — and just sent back — instead of the new answer.
        "idempotency_key": f"{run_id}:file" if attempt == 1 else f"{run_id}:file:{attempt}",
        "workflow_id": run_id,
        "workflow_name": INTAKE,
        "step_name": "file",
        "path": image_to_file(ctx),
        "document_type": classify.get("document_type", UNKNOWN),
        "confidence": classify.get("confidence"),
        "payload": {
            "client_id": matches[0].get("id") if identify.get("unambiguous") else None,
            "document_type": classify.get("document_type", UNKNOWN),
            "title": _title_for(classify.get("document_type")),
            "candidates": [
                {
                    "id": m.get("id"),
                    "name": m.get("name"),
                    "file_number": m.get("file_number"),
                    "match_reason": _match_reason(ctx),
                }
                for m in matches
            ],
        },
        "evidence": {
            "readable": bool(ctx.get("read", {}).get("readable")),
            "text": _text(ctx),
            "fields": ctx.get("extract", {}).get("fields", {}),
            "engine": ctx.get("read", {}).get("engine"),
        },
        "events": _agent_events(ctx),
        "warnings": _warnings(ctx),
    }


def _title_for(document_type: str | None) -> str:
    return (document_type or "document").replace("_", " ").title()


def _match_reason(ctx: dict[str, Any]) -> str:
    """Why the AI thinks these candidates are the right ones.

    Stated from what was actually used to search, not invented. A reviewer
    deciding between two people needs to know whether the match came from a CNIC
    printed on the document or from the phone it happened to arrive on — those
    are very different levels of evidence.
    """
    fields = ctx.get("extract", {}).get("fields", {})

    if fields.get("cnic"):
        return "CNIC on the document matched exactly"

    if fields.get("mobile"):
        return "Mobile number on the document matched"

    return "Matched on the number the message came from"


def _warnings(ctx: dict[str, Any]) -> list[str]:
    """What a reviewer should be told before they look at anything else."""
    warnings: list[str] = []
    identify = ctx.get("identify", {})
    matches = identify.get("matches") or []

    if not ctx.get("read", {}).get("readable", False):
        warnings.append("The document could not be read.")

    if len(matches) > 1:
        warnings.append(f"{len(matches)} clients matched — choose one.")

    if not matches:
        warnings.append("No client matched. Search for the right one.")

    if reason := ctx.get("redo_reason"):
        warnings.append(f"Resubmitted after review: {reason}")

    return warnings


def _agent_events(ctx: dict[str, Any]) -> list[dict[str, Any]]:
    """The steps that ran here, for the CMS to replay into its own history.

    Sent so the trail a person reads covers the whole chain rather than starting
    at the moment the proposal arrived. Timestamps are UTC with an offset; the
    CMS converts them into its own timezone.
    """
    events = ctx.get("step_times") or {}
    took = ctx.get("step_durations") or {}

    return [
        {
            "event": event,
            "occurred_at": occurred,
            "message": message,
            # How long the step took, so the CMS can answer "what is slow"
            # without querying a database it cannot reach (ADR-0011).
            "duration_ms": took.get(step),
        }
        for event, step, occurred, message in (
            ("ocr", "read", events.get("read"), ctx.get("read", {}).get("engine")),
            ("classified", "classify", events.get("classify"), ctx.get("classify", {}).get("document_type")),
            ("extracted", "extract", events.get("extract"), None),
            ("identified", "identify", events.get("identify"), _match_reason(ctx)),
        )
        if occurred
    ]


def _note(ctx: dict[str, Any]) -> dict[str, Any]:
    """The note a later run will read.

    A summary, not a transcript. The raw OCR text stays out of it: memory is
    embedded and may be sent to a hosted model for reasoning, and the text of a
    bank statement is Level 3 (ADR-0002). What is worth carrying forward is that
    a document of a type arrived and where it went.
    """
    filed = ctx.get("file", {})
    document_type = ctx.get("classify", {}).get("document_type", UNKNOWN)
    client_id = filed.get("client_id") or ctx.get("identify", {}).get("matches", [{}])[0].get("id")

    return {
        "kind": MemoryKind.DOCUMENT,
        "content": f"Received and filed a {document_type.replace('_', ' ')} sent over WhatsApp.",
        "client_id": client_id,
        "document_id": filed.get("document_id"),
        "workflow_id": ctx.get("run_id"),
        "confidence": ctx.get("classify", {}).get("confidence"),
        "metadata": {
            "document_type": document_type,
            "source": "whatsapp",
            # Recorded because it is the question asked later: did a human choose
            # this client, or did the AI?
            "approved": bool(filed.get("approved")),
        },
    }


DOCUMENT_INTAKE = Workflow(
    name=INTAKE,
    steps=(
        Step(name="prepare", tool="prepare_image", optional=True,
             build_input=lambda ctx: {"path": ctx.get("path")},
             skip_when=lambda ctx: not is_photograph(ctx)),
        Step(name="read", tool="ocr_document", build_input=lambda ctx: {"path": image_to_read(ctx)}),
        Step(name="classify", tool="classify_document", build_input=lambda ctx: {"text": _text(ctx)}),
        Step(name="extract", tool="extract_fields", build_input=lambda ctx: {"text": _text(ctx)}),
        Step(name="identify", tool="search_client", build_input=_identify_query),
        # Submits to the Approval Queue and waits. The CMS does the filing once a
        # human approves (ADR-0008), so this step's result arrives from there
        # rather than from running the tool again.
        Step(name="file", tool="submit_filing_proposal", build_input=_propose_filing,
             skip_when=nobody_named_it),
        # Optional on purpose: the document is already filed by this point, and
        # losing the note is a far smaller harm than failing a run whose real
        # work succeeded.
        Step(name="remember", tool="remember", build_input=_note, optional=True),
    ),
)
