"""Deciding what a document is, as cheaply as it can be decided.

Four stages, each free to abstain, ordered by cost:

    caption   → what the sender typed          free, and the most reliable
    filename  → what the file is called        free, and a strong prior
    rules     → what the text says             expensive: this is where OCR runs
    model     → not built (see below)

The ordering is the design. Reading a document is the most expensive thing this
system does — about eighteen seconds warm for a page — and it is also the only
stage that touches the document's contents. So the cheap stages are not an
optimisation bolted on afterwards; they are how most documents get classified
without ever being opened.

## Abstaining is not failing

A stage returns None when it has nothing to say, and the pipeline moves on. Only
the last word matters, and "unknown" is a legitimate last word: it routes to a
human, which is the correct outcome for a document nothing recognised. Nothing
here guesses to avoid returning unknown.

## Where the model stage would go

Deliberately absent. A hosted model receiving OCR text would put a client's bank
statement into somebody else's infrastructure, and ADR-0002 classifies raw OCR
output as Level 3 — it does not leave the installation. A local model does not
cross that boundary and is where this is going, but it is not built, and the
measured unknown rate should decide whether it is worth building at all.

`Method.MODEL` exists so the metric that answers that question can be collected
before anything is added.
"""

from __future__ import annotations

import logging
import time
from dataclasses import dataclass
from enum import StrEnum

from app.documents import registry
from app.documents.signals import Signals
from app.documents.statistics import STATS
from app.ocr import classification as rules
from app.runtime.metrics import METRICS, RATIO_BUCKETS

logger = logging.getLogger(__name__)

UNKNOWN = rules.UNKNOWN


class Method(StrEnum):
    """How a document was recognised. Level 1 data — safe to report anywhere."""

    CAPTION = "caption"
    FILENAME = "filename"
    RULES = "rules"
    MODEL = "model"
    NONE = "none"


#: Confidence at or above which a stage's answer is taken and the rest skipped.
#:
#: Set so a caption always short-circuits and a filename never quite does. A
#: filename is a strong prior and a bad one is silently wrong — `scan001.pdf`
#: renamed by a phone, a template reused from last year's client — so it wins
#: only when nothing better follows it, never before the text has had a chance.
ACCEPT = 0.90

#: What a caption is worth.
#:
#: Not 1.0. A person naming the document is the best evidence available and is
#: still a person typing on a phone, and a claim of certainty would invite a
#: caller to skip the reviewer. It clears ACCEPT, which is what matters.
CAPTION_CONFIDENCE = 0.95

#: What a filename is worth. Below ACCEPT on purpose — see above.
FILENAME_CONFIDENCE = 0.70


@dataclass(frozen=True, slots=True)
class Classification:
    """What a document is, how sure, on what basis, and by which stage."""

    filing_type: str
    """What the CMS will store. `unknown` when nothing recognised it."""

    refinement: str | None = None
    """A more precise name, when one could be given. `sale_deed` inside
    `property_document`."""

    confidence: float = 0.0
    reason: str = ""
    method: Method = Method.NONE

    was_read: bool = False
    """Whether anything opened the document. Distinct from "could not be read",
    which the CMS scores as a risk and this does not."""

    @property
    def is_known(self) -> bool:
        return self.filing_type != UNKNOWN

    @property
    def label(self) -> str:
        """The most specific human name available."""
        return registry.label_for(self.refinement or self.filing_type)

    @property
    def type(self) -> registry.FilingType | None:
        return registry.get(self.filing_type)


def classify(signals: Signals) -> Classification:
    """Run the stages in order and return the first confident answer.

    A stage below ACCEPT is remembered rather than returned: it is better than
    nothing, and if every later stage abstains it becomes the answer. That is
    what lets a filename classify a document the reader could not make sense of.

    Measured here rather than at each call site, so every intake — WhatsApp
    today, email and uploads later — is counted the same way without any of them
    remembering to.
    """
    started = time.perf_counter()
    verdict = _classify(signals)
    elapsed = time.perf_counter() - started

    STATS.record(
        verdict.method.value,
        known=verdict.is_known,
        was_read=verdict.was_read,
        seconds=elapsed,
    )

    METRICS.counter(
        "taxpilot_classifications_total",
        "Documents classified, by the stage that decided and whether it did.",
        method=verdict.method.value,
        outcome="known" if verdict.is_known else "unknown",
    )
    # The distribution, not the mean: a bimodal spread of captioned documents at
    # 0.95 and rule matches at 0.4 averages to something neither population
    # resembles.
    METRICS.observe(
        "taxpilot_classification_confidence",
        verdict.confidence,
        "Confidence per classification.",
        buckets=RATIO_BUCKETS,
    )
    METRICS.observe(
        "taxpilot_classification_seconds",
        elapsed,
        "Time spent deciding what a document is, reading included.",
    )

    return verdict


def _classify(signals: Signals) -> Classification:
    best = Classification(UNKNOWN)
    refusal: Classification | None = None

    for stage in (_from_caption, _from_filename, _from_rules, _from_model):
        verdict = stage(signals)

        if verdict is None:
            continue

        if verdict.confidence >= ACCEPT:
            return _with_read_flag(verdict, signals)

        if not verdict.is_known:
            # A stage that looked and decided against is worth more than one
            # that had nothing to look at. "'invoice' rests on a single marker"
            # tells a reviewer what happened; "nothing identified this" does not.
            refusal = verdict

            continue

        if verdict.confidence > best.confidence:
            best = verdict

    if not best.is_known:
        return _with_read_flag(
            refusal
            or Classification(
                UNKNOWN,
                reason="Nothing in the caption, the filename or the text identified this document.",
                method=Method.NONE,
            ),
            signals,
        )

    return _with_read_flag(best, signals)


def _with_read_flag(verdict: Classification, signals: Signals) -> Classification:
    """Stamp whether the document was actually opened.

    Set here rather than in each stage because it is a property of the run, not
    of the stage that happened to win: a caption can decide a document *after*
    an earlier attempt read it, and the CMS needs to know it was read.
    """
    from dataclasses import replace

    return replace(verdict, was_read=signals.was_read)


# ── Stage 1 — what the sender said ───────────────────────────────────────


def _from_caption(signals: Signals) -> Classification | None:
    """The sender named the document.

    Trusted above everything else, and it is the only stage that can settle a
    document without the file being opened at all. That is not a shortcut — for
    a bank statement it is the whole design, because its contents are Level 3
    and the cheapest way to keep them inside the installation is never to read
    them (ADR-0002).

    Longest match wins. "Income Tax Return" contains "Tax Return", and a caption
    saying the first means the first.
    """
    caption = (signals.caption or "").strip()

    if not caption:
        return None

    matches: list[tuple[int, int, str, str | None]] = []


    for filing_slug, refinement_slug, pattern in registry.caption_patterns():
        if found := pattern.search(caption):
            matches.append((found.start(), found.end(), filing_slug, refinement_slug))

    if not matches:
        return _identifier_convention(caption)

    matches.sort(key=lambda m: m[1] - m[0], reverse=True)
    start, end, filing_slug, refinement_slug = matches[0]

    # Overlapping matches are one mention read at two precisions — "CNIC Back"
    # contains "CNIC", and the longer reading is the right one. Ambiguity is a
    # *separate* mention of a different type, as in "salary slip and utility
    # bill", where picking either would be a guess dressed as a rule.
    for other_start, other_end, other_filing, _ in matches[1:]:
        overlaps = other_start < end and other_end > start

        if other_filing != filing_slug and not overlaps:
            return None

    label = registry.label_for(refinement_slug or filing_slug)

    return Classification(
        filing_type=filing_slug,
        refinement=refinement_slug,
        confidence=CAPTION_CONFIDENCE,
        reason=f"The message described this as a {label.lower()}.",
        method=Method.CAPTION,
    )


def _identifier_convention(caption: str) -> Classification | None:
    """A caption that names a client and nothing else means a bank statement.

    This is not an inference about the document — it is a convention this
    installation already runs on, shipped and in use: a firm forwards a
    statement to their own self-chat and captions it `file 1420`, and that is
    the whole instruction.

    It is preserved here rather than left in the inbox because the router now
    decides from the classification, and a rule that used to sit in an `if`
    would otherwise be silently dropped — turning every captioned forward into
    an OCR run against a document the design says must not be read.

    Narrow on purpose. It applies only when the caption names *no* type at all;
    `Salary Slip, file 1420` is a salary slip, and the type wins.
    """
    from app.documents import identifiers

    if not identifiers.read(caption).any:
        return None

    return Classification(
        filing_type="bank_statement",
        confidence=CAPTION_CONFIDENCE,
        reason="The message named a client and no document type, "
        "which this installation treats as a forwarded bank statement.",
        method=Method.CAPTION,
    )


# ── Stage 2 — what the file is called ────────────────────────────────────


def _from_filename(signals: Signals) -> Classification | None:
    """`Meezan_Bank_Statement_July2026.pdf` is not proof, but it is evidence.

    Below ACCEPT deliberately, so it never prevents the text being read. A
    filename is wrong in ways that leave no trace — a template reused from
    another client, a phone renaming a scan — and the failure is silent.
    """
    name = signals.filename or signals.path.name

    if not name:
        return None

    for filing in registry.filing_types():
        for pattern in filing.filename:
            if pattern.search(name):
                refinement = _refine_filename(filing.slug, name)

                return Classification(
                    filing_type=filing.slug,
                    refinement=refinement,
                    confidence=FILENAME_CONFIDENCE,
                    reason=f"The filename looks like a {registry.label_for(refinement or filing.slug).lower()}.",
                    method=Method.FILENAME,
                )

    return None


def _refine_filename(filing_slug: str, name: str) -> str | None:
    for refinement in registry.refinements_for(filing_slug):
        for pattern in refinement.filename:
            if pattern.search(name):
                return refinement.slug

    return None


# ── Stage 3/4 — what the text says ───────────────────────────────────────


def _from_rules(signals: Signals) -> Classification | None:
    """Read the document and score it against the registry's markers.

    The expensive stage, and the only one that opens the file. It is skipped
    entirely for a type that must not be read — but by the time that is known
    the caption has already settled it, so in practice this is not reached for
    those documents at all.
    """
    text = signals.text()

    if not text.strip():
        return None

    verdict = rules.classify(text)

    if verdict.document_type == UNKNOWN:
        # The rule engine's refusals are informative — "scored too close",
        # "rests on a single marker" — and are worth carrying to a reviewer
        # rather than flattening to "unknown".
        return Classification(
            UNKNOWN,
            reason="; ".join(verdict.evidence[:2]),
            method=Method.RULES,
        )

    refinement = registry.refine(verdict.document_type, text)
    label = registry.label_for(refinement.slug if refinement else verdict.document_type)

    return Classification(
        filing_type=verdict.document_type,
        refinement=refinement.slug if refinement else None,
        confidence=verdict.confidence,
        reason=f"Recognised as a {label.lower()} from "
        + ", ".join(verdict.evidence[:3]),
        method=Method.RULES,
    )


# ── Stage 4 — a model on this host ───────────────────────────────────────


def _from_model(signals: Signals) -> Classification | None:
    """The residue, and only the residue.

    Reached last, after the rule engine has already read the text and failed to
    name it. Off unless a deployment has configured a model, and refused
    outright unless that model runs on this host — a document's text is Level 3
    and does not leave the installation (ADR-0002, ADR-0009).

    The model is given the text and nothing else: not the filename, not the
    caption, not a client. It is asked one question.
    """
    model = signals.model

    if model is None:
        return None

    text = signals.text()

    if not text.strip():
        return None

    verdict = model.classify(text)

    if verdict is None:
        # Down, slow, or answering something the registry does not hold. Any of
        # those must leave the document where the earlier stages left it.
        METRICS.counter(
            "taxpilot_model_answers_total", "What became of a model's answer.",
            outcome="abstained",
        )

        return None

    corroboration = _corroborates(verdict.document_type, text)

    if corroboration is None:
        METRICS.counter(
            "taxpilot_model_answers_total", "What became of a model's answer.",
            outcome="uncorroborated",
        )
        logger.info(
            "Discarding the model's answer of '%s': nothing in the text supports it.",
            verdict.document_type,
        )

        return None

    METRICS.counter(
        "taxpilot_model_answers_total", "What became of a model's answer.",
        outcome="accepted",
    )

    refinement = registry.refine(verdict.document_type, text)
    label = registry.label_for(refinement.slug if refinement else verdict.document_type)

    return Classification(
        filing_type=verdict.document_type,
        refinement=refinement.slug if refinement else None,
        confidence=verdict.confidence,
        # The corroborating text is named, because it is the difference between
        # a reviewer trusting this and a reviewer having only a model's word.
        reason=f"A local model recognised this as a {label.lower()}, "
        f"and the text contains “{corroboration}”: {verdict.reason}",
        method=Method.MODEL,
    )


def _corroborates(filing_slug: str, text: str) -> str | None:
    """Whether anything in the text supports what the model said.

    ## Why a model's answer is not enough on its own

    Measured on staging with a 0.5B model, and it was not close. Asked about a
    covering note reading "Dear Sir, please find attached the file we
    discussed", it answered **invoice, 0.85, "clearly an invoice"**. Asked about
    a payslip worded "Remuneration Advice / Emoluments", it answered **tax
    return**. Both had been `unknown` before this stage existed — which routes
    to a person, and is the correct answer.

    The residue this stage sees has nothing better to beat, so a confident wrong
    answer simply becomes the classification, and the document goes to the wrong
    workflow and the wrong extractors. A model that cannot say "I don't know" is
    worse here than no model at all.

    ## What this rule is

    The model may break a tie the rule engine refused to break — but only in a
    direction the evidence already pointed. One marker for the type it named has
    to appear in the text.

    That is deliberately below the rule engine's own bar: it needs `MIN_HITS`
    corroborating markers, a minimum score and a margin over the runner-up, and
    it returned `unknown` precisely because it could not reach them. A single
    hit is a hint the rules were right to distrust on their own, and exactly the
    thing a model's opinion can legitimately tip.

    Returns the matched text so the reason can name it. `other` has no markers
    and can never be corroborated, which is correct — it is a filing decision a
    person makes, not something recognised.
    """
    filing = registry.get(filing_slug)

    if filing is None or not filing.markers:
        return None

    for pattern, _weight in filing.markers:
        if found := pattern.search(text):
            return found.group(0)[:40].strip()

    return None
