"""Hand one document to an intake workflow, by hand.

    python scripts/submit_document.py path/to/scan.jpg
    python scripts/submit_document.py scan.jpg --sender 923001234567
    python scripts/submit_document.py statement.pdf --caption "file 1420"

WHY THIS EXISTS

WhatsApp is the only way a document normally arrives, and until Meta
verification is done nothing can arrive at all. That leaves no way to exercise
the platform by hand — which is exactly what somebody wants when they have just
deployed it and would like to see it work.

So this runs the real workflow: the real OCR engine, the real classifier, the
real field extraction, the real client search over the real Agent API, and a
real proposal submitted to the real Approval Queue. The only thing it replaces
is the WhatsApp message that would normally have carried the file.

WHICH WORKFLOW IT PICKS

The same way the inbox picks: the document is classified from its caption, its
filename and — only if those settle nothing — its text, and the router chooses
the workflow from the result.

That is the real classifier, the real router and the real catalogue, not a
restatement of their rules. A diagnostic tool that decided differently from the
thing it stands in for would be worse than no tool, and this one drifted exactly
that way once: it went on routing by identifier alone after the daemon had moved
to classifying first, so traffic pushed through it exercised a path that no
longer existed.

It does NOT file anything. The proposal waits for a person, exactly as it would
have (ADR-0008), and the running daemon picks the decision up on its next poll.

Configuration comes from the environment, the same as the daemon — so run it
through the deployment's launcher rather than directly, or the CMS URL and the
agent credential will be missing.
"""

from __future__ import annotations

import argparse
import json
import logging
import sys
from pathlib import Path

sys.path.insert(0, str(Path(__file__).resolve().parents[1]))

from app.config.settings import ConfigurationError  # noqa: E402
from app.runtime.container import Container  # noqa: E402
from app.documents import classifier, identifiers, router  # noqa: E402
from app.documents.signals import Signals  # noqa: E402
from app.workflow import catalogue  # noqa: E402
from app.workflow.bank_statement_intake import BANK_STATEMENT_INTAKE  # noqa: E402
from app import __main__ as entry
from app.workflow.state import RunState  # noqa: E402


def main(argv: list[str] | None = None) -> int:
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("document", type=Path, help="the file to read")
    parser.add_argument(
        "--sender",
        default=None,
        help="the number it would have arrived from; used to find the client "
        "when the document itself carries no identifier",
    )
    parser.add_argument(
        "--caption",
        default="",
        help="the message caption it would have arrived with; a client identifier "
        'here ("file 1420", or a CNIC) files it as a bank statement without reading it',
    )
    parser.add_argument("--quiet", action="store_true", help="only print the result")
    arguments = parser.parse_args(argv)

    logging.basicConfig(
        level=logging.WARNING if arguments.quiet else logging.INFO,
        format="%(levelname)-8s %(message)s",
        stream=sys.stdout,
    )

    for noisy in ("paddle", "paddlex", "PIL", "urllib3"):
        logging.getLogger(noisy).setLevel(logging.ERROR)

    if not arguments.document.is_file():
        print(f"No document at {arguments.document}")

        return 1

    try:
        container = Container()
        container.settings
    except ConfigurationError as exc:
        print(f"{exc}\n\nRun this through the deployment launcher so the environment is loaded.")

        return 78

    document = arguments.document.resolve()
    context = {"path": str(document)}
    found = identifiers.read(arguments.caption)

    # Registered FIRST — before classification, not after (ADR-0010).
    #
    # The order matters here in a way it does not on the WhatsApp path. There,
    # _classify_message deliberately leaves read_first_page unset, so nothing is
    # downloaded or read to classify and registration can safely follow. Here the
    # file is already on disk, so the classifier is given a reader — and a
    # document that cannot be settled by its filename is therefore OCR'd to
    # decide what it is.
    #
    # Registering afterwards meant that read happened to a document with no
    # reference. Killing this process during it left exactly nothing: no record,
    # no trace, no way to answer where the document went. Measured on 2026-08-02.
    #
    # The same function the WhatsApp path calls, not a copy of it: a second
    # registration path is a second place for a document to slip through.
    reference = entry.register_arrival(
        container,
        source="manual",
        path=document,
        filename=document.name,
        mime_type=_mime_of(document),
        size_bytes=document.stat().st_size,
    )

    if reference is None:
        # Same rule as every other source: a document the CMS has never heard of
        # does not proceed. It is held in the outbox and retried by the daemon.
        print("\nThe CMS could not be told this document arrived. It is held and "
              "will be registered when the CMS returns; nothing has been read.")

        return 1

    print(f"Registered as {reference}")

    # Classified and routed exactly as the daemon does it — the same pipeline,
    # the same router, the same catalogue. A diagnostic tool that decided
    # differently from the thing it stands in for would be worse than no tool,
    # and this one drifted once already: it kept routing on the identifier alone
    # after the daemon had moved to classifying first.
    #
    # Now safely after registration, so the read this may perform happens to a
    # document the CMS already knows about.
    entry._push(container, reference, "processing", reason="Classifying the document.")

    classification = classifier.classify(
        Signals(
            path=document,
            source="manual",
            caption=arguments.caption,
            filename=document.name,
            mime_type=_mime_of(document),
            size_bytes=document.stat().st_size,
            read_first_page=lambda: _first_page(container, document),
            model=container.model,
        )
    )
    destination = router.route(classification)
    workflow = catalogue.require(destination.workflow)

    print(f"\n{document.name} -> {container.settings.cms_base_url}")
    print(
        f"Recognised as {classification.label} "
        f"({classification.method.value}, {classification.confidence:.0%}) "
        f"-> {workflow.name}"
    )
    print(f"  {classification.reason}\n")

    context |= {
        "workflow_name": workflow.name,
        "document_type": classification.filing_type,
        "file_number": found.file_number,
        "cnic": found.cnic,
        "sender": arguments.sender,
        "intake_reference": reference,
        "source": _stand_in_source(document, arguments.sender),
        "classification": {
            "method": classification.method.value,
            "confidence": classification.confidence,
            "reason": classification.reason,
            "refinement": classification.refinement,
            "label": classification.label,
            "was_read": classification.was_read,
        },
    }

    entry._push(
        container, reference, "processing",
        reason=f"Reading the document for {workflow.name}.",
        workflow_name=workflow.name, current_step="read",
    )

    run = container.engine.start(workflow, context)

    # Reported the same way a forwarded document is, so the lifecycle a person
    # sees in the Intake Center does not depend on how the file got here.
    entry._push_outcome(container, reference, run, workflow.name)

    for step in run.steps:
        marker = {"succeeded": "ok", "failed": "FAILED"}.get(str(step.state.value), str(step.state.value))
        print(f"  {step.name:<10} {marker}")

        # The reason, not just the verdict. A step that failed silently is the
        # whole difficulty of diagnosing this by hand.
        if step.error:
            print(f"             -> {step.error}")

    print()
    _report(run)

    # Anything other than a proposal waiting for a human is worth a non-zero
    # exit: this gets run from scripts, and "it printed something" is not a
    # result.
    return 0 if run.state is RunState.AWAITING_APPROVAL else 1


def _first_page(container, document) -> str:
    """Read page one, for the classification stage that needs text.

    The real OCR engine, through the real Tool, so what this reports is what the
    daemon would have read — including the second-pass fallback into the other
    alphabet.
    """
    result = container.tools.get("ocr_document").execute(path=str(document))

    return str((result.data or {}).get("text") or "")


def _stand_in_source(document, sender: str | None) -> dict:
    """The message that would have carried this file.

    `channel` is whatsapp because that is what this stands in for, and the CMS
    decides how to render and how to deduplicate from it. The message id says
    `manual:` and carries the file's own hash — honest about where it came from
    on the review screen, and still a stable key, so submitting the same file
    twice is refused as a duplicate exactly as a second forward would be.
    """
    import hashlib
    from datetime import UTC, datetime

    digest = hashlib.sha256(document.read_bytes()).hexdigest()[:16]

    return {
        "channel": "whatsapp",
        "message_id": f"manual:{digest}",
        "number": sender or "",
        "received_at": datetime.now(UTC).isoformat(),
        "file_name": document.name,
        "file_size": document.stat().st_size,
        "mime_type": _mime_of(document),
    }


def _mime_of(document) -> str:
    return {
        ".pdf": "application/pdf",
        ".jpg": "image/jpeg",
        ".jpeg": "image/jpeg",
        ".png": "image/png",
    }.get(document.suffix.lower(), "application/octet-stream")


def _report(run) -> None:
    if run.workflow == BANK_STATEMENT_INTAKE.name:
        _report_forward(run)

        return

    read = run.context.get("read") or {}
    classify = run.context.get("classify") or {}
    extract = run.context.get("extract") or {}
    identify = (run.step("identify").output if run.step("identify") else {}) or {}
    filed = (run.step("file").output if run.step("file") else {}) or {}

    text = (read.get("text") or "").strip()
    print(f"  text read     : {len(text)} characters")
    print(f"  document type : {classify.get('document_type') or 'unknown'}")

    fields = extract.get("fields") or {}
    if fields:
        # A field is normally {"value": ...}, but not every extractor returns
        # that shape — printing it defensively beats a traceback in a tool whose
        # whole job is to explain what happened.
        shown = {
            key: (value.get("value") if isinstance(value, dict) else value)
            for key, value in fields.items()
        }
        print(f"  fields        : {json.dumps(shown, ensure_ascii=False, default=str)}")

    matches = identify.get("matches") or []
    print(f"  clients found : {len(matches)}" + (" (ambiguous)" if len(matches) > 1 else ""))

    for match in matches[:3]:
        print(f"                  #{match.get('id')} {match.get('name')}")

    if proposal := filed.get("proposal_id"):
        print()
        print(f"  PROPOSAL #{proposal} is waiting in the Approval Queue.")
        print("  Open /ai/approvals in the CMS to decide it.")
    else:
        print()
        print(f"  No proposal was submitted. The run ended: {run.state.value}")


def _report_forward(run) -> None:
    """What the forwarding path did.

    Deliberately says nothing about text, type or fields — there are none, and
    printing "document type: bank_statement" as though it had been worked out
    would misrepresent the one thing this path is defined by.
    """
    identify = run.context.get("identify") or {}
    filed = (run.step("file").output if run.step("file") else {}) or {}
    identifier = identify.get("identifier") or {}
    matches = identify.get("matches") or []

    print(f"  identifier    : {identifier.get('type')} = {identifier.get('value')}")
    print("  document read : no — never opened (ADR-0002)")

    if identify.get("conflict"):
        print("  clients found : the two identifiers named DIFFERENT clients")
    else:
        print(f"  clients found : {len(matches)}" + (" (ambiguous)" if len(matches) > 1 else ""))

    for match in matches[:3]:
        print(f"                  #{match.get('id')} {match.get('name')}")

    if proposal := filed.get("proposal_id"):
        print()
        print(f"  PROPOSAL #{proposal} is waiting in the Approval Queue.")
        print("  Open /ai/approvals in the CMS to decide it.")
    else:
        print()
        print(f"  No proposal was submitted. The run ended: {run.state.value}")


if __name__ == "__main__":
    raise SystemExit(main())
