"""Document Tools — reading, understanding and filing.

The split by execution policy (ADR-0004) is the important part: reading a
document is AUTOMATIC because being wrong wastes a call, while filing one is
REQUIRES_APPROVAL because being wrong puts a stranger's bank statement on a
client's record.
"""

from __future__ import annotations

from pathlib import Path
from typing import Any

from app.api.client import AgentApiError, CmsClient
from app.ocr.classification import UNKNOWN, classify
from app.ocr.engine import OcrEngine
from app.ocr.extraction import extract_all
from app.runtime.metrics import RATIO_BUCKETS, METRICS
from app.tools.base import ExecutionPolicy, Tool, ToolResult


class PrepareImageTool(Tool):
    """Crop a photograph down to the document in it, before anything reads it.

    Answers with a path either way. When preprocessing cannot safely improve the
    image it returns the original and says why, so the caller has one thing to
    use and never a decision to make — a document is never lost to this stage.
    """

    name = "prepare_image"
    description = "Detect, crop and clean a photographed document. Falls back to the original."
    policy = ExecutionPolicy.AUTOMATIC

    def run(self, **kwargs: Any) -> ToolResult:
        from app.images.prepare import prepare

        path = kwargs.get("path")

        if not path:
            return ToolResult.failure("A document path is required.")

        result = prepare(Path(path), debug=bool(kwargs.get("debug")))

        return ToolResult.success({
            "path": str(result.path),
            "cropped": result.cropped,
            "reason": result.reason,
            "uploaded_version": result.uploaded_version,
            **result.metadata,
        })


class OcrTool(Tool):
    """Read a document's text, locally (ADR-0002)."""

    name = "ocr_document"
    description = "Extract text from a PDF or image. Returns text with a confidence score."
    policy = ExecutionPolicy.AUTOMATIC

    def __init__(self, engine: OcrEngine) -> None:
        self._engine = engine

    def run(self, **kwargs: Any) -> ToolResult:
        path = kwargs.get("path")

        if not path:
            return ToolResult.failure("A document path is required.")

        source = Path(path)

        if not source.exists():
            return ToolResult.failure(f"No document at {source}.")

        with METRICS.time("taxpilot_ocr_seconds", "Time spent reading a document."):
            result = self._engine.read(source)

        # The distribution is what matters, not the average: a bimodal spread of
        # clean scans and unreadable photographs averages to "mediocre" and hides
        # both populations.
        METRICS.observe("taxpilot_ocr_confidence", result.confidence,
                        "Confidence per document read.", buckets=RATIO_BUCKETS)
        METRICS.counter(
            "taxpilot_ocr_documents_total", "Documents read.",
            # Three outcomes, not two. A file the engine could not decode and a
            # page with nothing on it were both counted "unreadable", which made
            # a rise in corrupt uploads indistinguishable from a rise in blank
            # ones on the only graph that would have shown either.
            outcome="failed" if result.failed else ("unreadable" if result.is_empty else "read"),
        )

        if result.is_empty:
            # Still a success, and deliberately so: a document nobody could read
            # must still reach a person, and failing the step here would end the
            # workflow instead of sending it to the Approval Queue.
            #
            # `failure` rides along so the two ways of being empty stay apart. A
            # blank page is finished with; a file that would not decode needs the
            # client asked to send it again, and until this was carried the record
            # could not say which had happened.
            return ToolResult.success(
                {
                    "text": "",
                    "pages": result.pages,
                    "readable": False,
                    "engine": result.engine,
                    "failure": result.failure,
                },
                confidence=0.0,
            )

        return ToolResult.success(
            {
                "text": result.text,
                "pages": result.pages,
                "readable": True,
                "engine": result.engine,
            },
            confidence=result.confidence,
        )


class ClassifyDocumentTool(Tool):
    """Decide what a document is, or admit that it cannot tell."""

    name = "classify_document"
    description = "Identify a document's type from its text. Returns 'unknown' rather than guessing."
    policy = ExecutionPolicy.AUTOMATIC

    def run(self, **kwargs: Any) -> ToolResult:
        text = kwargs.get("text") or ""

        result = classify(text)

        return ToolResult.success(
            {
                "document_type": result.document_type,
                "evidence": result.evidence,
                "runner_up": result.runner_up,
                # Stated so a workflow branches rather than filing on a shrug.
                "needs_human": result.document_type == UNKNOWN,
            },
            confidence=result.confidence,
        )


class ExtractFieldsTool(Tool):
    """Pull structured fields out of recognised text."""

    name = "extract_fields"
    description = "Extract CNIC, IBAN, mobile, email, dates, tax year and amounts from text."
    policy = ExecutionPolicy.AUTOMATIC

    def run(self, **kwargs: Any) -> ToolResult:
        fields = extract_all(kwargs.get("text") or "")

        return ToolResult.success({"fields": fields, "found": sorted(fields)})


class ExtractDocumentFieldsTool(Tool):
    """Pull the fields that matter for this particular kind of document.

    One tool, not one per type. A tool per type would mean every new document
    type needed a tool, a registration *and* a workflow — and the whole point of
    the registry is that it needs one entry. Dispatch is on the classified type,
    in `app/documents/extractors.py`.

    AUTOMATIC: it reads text that has already been read and produces evidence
    for a person. Nothing it finds files anything or chooses a client.
    """

    name = "extract_document_fields"
    description = (
        "Extract the fields that matter for a document of a known type — "
        "employer and net pay from a salary slip, registry number and area from "
        "a deed. Returns only what it found."
    )
    policy = ExecutionPolicy.AUTOMATIC

    def run(self, **kwargs: Any) -> ToolResult:
        from app.documents import extractors

        text = kwargs.get("text") or ""
        document_type = str(kwargs.get("document_type") or "")
        fields = extractors.extract(document_type, text)

        return ToolResult.success(
            {
                "fields": fields,
                "found": sorted(fields),
                # Stated so a reviewer can tell "this type has no fields worth
                # extracting" from "this document had none of them".
                "specific": extractors.has_extractor(document_type),
            }
        )


class SubmitFilingProposalTool(Tool):
    """Ask a human to approve filing a document (ADR-0008).

    AUTOMATIC, which looks wrong for something that leads to a client's record
    changing, and is not: submitting a proposal writes nothing. It puts the
    request in front of a reviewer, and the CMS does the filing itself once they
    approve. The gate is a person, not this policy.

    This is the ordinary path. UploadDocumentTool below is the promoted one,
    which needs a grant no agent holds by default.
    """

    name = "submit_filing_proposal"
    description = (
        "Submit a document to the CMS Approval Queue for a human to review. "
        "Files nothing; returns once the proposal is lodged."
    )
    policy = ExecutionPolicy.AUTOMATIC

    def __init__(self, cms: CmsClient) -> None:
        self._cms = cms

    def run(self, **kwargs: Any) -> ToolResult:
        path = kwargs.get("path")
        document_type = kwargs.get("document_type")

        if not path:
            return ToolResult.failure("A document path is required.")

        warnings = list(kwargs.get("warnings") or [])

        if not document_type or document_type == UNKNOWN:
            """
            An unknown type goes to a human. It used to be refused here.

            The reasoning was that a reviewer can authorise a filing but cannot
            supply a type nobody determined. That is backwards: supplying the
            type is precisely what a person looking at the document can do and
            this cannot, and the CMS already lets a reviewer amend
            `document_type` when they approve.

            What refusing actually did was throw the whole run away. Seen on a
            real document: the CNIC was read, the client was identified as a
            specific person, the image was on disk — and because the classifier
            could not name an old all-Urdu card, none of it reached anybody.
            From the customer's side they put a document in their self-chat and
            nothing happened, with no error to look at.

            The sibling decision in _propose_filing already says this: an
            ambiguous *identification* is carried into the proposal rather than
            failing, because a run that fails throws away the OCR and the
            candidates and makes a human start again. An unrecognised type is
            the same situation and now gets the same answer.
            """
            document_type = None
            warnings.append(
                "The document type could not be determined. Choose one before approving."
            )

            # And in the payload, which carries its own copy — that is the one
            # the CMS reads when a reviewer approves. Left as the string
            # "unknown" it would file a document under a type by that name
            # rather than making somebody choose.
            payload = dict(kwargs.get("payload") or {})

            if payload.get("document_type") in (None, "", UNKNOWN):
                payload["document_type"] = None

            kwargs = {**kwargs, "payload": payload}

        submission = {
            "idempotency_key": kwargs.get("idempotency_key"),
            "workflow_id": kwargs.get("workflow_id"),
            "workflow_name": kwargs.get("workflow_name", "document_intake"),
            "step_name": kwargs.get("step_name", "file"),
            "type": "document.file",
            "tool_name": self.name,
            "tool_version": kwargs.get("tool_version"),
            "agent_version": kwargs.get("agent_version"),
            "confidence": kwargs.get("confidence"),
            "payload": kwargs.get("payload") or {},
            "evidence": kwargs.get("evidence") or {},
            "events": kwargs.get("events") or [],
            "warnings": warnings,
        }

        # The CMS already holds these bytes: they were stored when the document
        # was registered on arrival (ADR-0010). Naming that record lets the
        # proposal adopt the stored file instead of carrying it again — which was
        # a second copy of a Level 3 document on disk, a second copy in every
        # backup, and 11 MB across the network for a 4 MB scan.
        #
        # Sent alongside the file rather than instead of it: a CMS still on
        # contract 1.0 ignores the field and reads the upload, so this works
        # against an installation that has not been updated yet.
        if kwargs.get("intake_reference"):
            submission["intake_reference"] = kwargs["intake_reference"]

        try:
            proposal = self._cms.submit_proposal(submission, path=path)
        except AgentApiError as exc:
            return ToolResult.failure(str(exc))

        # Pending, not success: the document has not been filed, and a workflow
        # that treated this as done would write a memory note saying it had been.
        return ToolResult.pending(
            {
                "proposal_id": proposal.get("id"),
                # The DOC- handle, kept so the intake record can be linked to
                # this proposal when the run reports its outcome. Without it the
                # two records stayed unlinked, the self-healer had nothing to
                # match, and every document that filed successfully sat at
                # "awaiting a reviewer" for ever while its file was on the
                # client's record.
                "proposal_reference": proposal.get("reference"),
                "status": proposal.get("status"),
                "risk_level": proposal.get("risk_level"),
                # Present only on the FIRST response for a document nobody could
                # be matched to, and never again — the CMS stamps it as it hands
                # it out. Carried through so the sender can be asked for an
                # identifier without this side deciding whether it already has.
                "notify": proposal.get("notify"),
            }
        )


class UploadDocumentTool(Tool):
    """File a document against a client.

    REQUIRES_APPROVAL: this writes to a real client's record. The base class
    stops it before ``run`` and returns a proposal, so the code below only ever
    executes once a human has approved — which is why it can be written as a
    straightforward write with no autonomy checks of its own.
    """

    name = "upload_document"
    description = "File a document against a client. Requires human approval before it executes."
    policy = ExecutionPolicy.REQUIRES_APPROVAL

    def __init__(self, cms: CmsClient) -> None:
        self._cms = cms

    def run(self, **kwargs: Any) -> ToolResult:
        client_id = kwargs.get("client_id")
        path = kwargs.get("path")
        document_type = kwargs.get("document_type")

        if not isinstance(client_id, int) or client_id <= 0:
            # Reached when identification was ambiguous and the reviewer approved
            # without choosing a client. Refusing is the only safe answer: there
            # is no default client, and picking the first candidate would file a
            # stranger's bank statement onto someone's record.
            return ToolResult.failure(
                "No client was identified for this document. "
                "Approve again with the correct client_id."
            )

        if not path or not document_type:
            return ToolResult.failure("A path and a document_type are required.")

        if document_type == UNKNOWN:
            # An approval can authorise filing a document; it cannot supply a
            # type nobody determined. Filing under "unknown" hides it.
            return ToolResult.failure("Refusing to file a document whose type is unknown.")

        try:
            filed = self._cms.upload_document(
                client_id=client_id,
                path=path,
                document_type=document_type,
                title=kwargs.get("title"),
                tax_year=kwargs.get("tax_year"),
                notes=kwargs.get("notes"),
                source=kwargs.get("source", "WhatsApp"),
            )
        except AgentApiError as exc:
            return ToolResult.failure(str(exc))

        return ToolResult.success(
            {
                "document_id": filed.get("document_id"),
                "client_id": filed.get("client_id"),
                "document_type": filed.get("document_type"),
                "title": filed.get("title"),
            }
        )
