"""Deciding what a document is.

Evidence-based and deliberately unwilling to guess. Every type carries markers;
a document scores against each, and the winner must clear a threshold *and* beat
the runner-up by a margin. Anything else is ``unknown`` — which is a useful
answer, because it routes to a human instead of filing a bank statement as a
CNIC.

Confidence here is the classifier's own, and is separate from the OCR
confidence that produced the text. Both matter: a confident classification of
badly-read text is not trustworthy, and the caller is given each so it can say so.
"""

from __future__ import annotations

import re
import unicodedata
from dataclasses import dataclass

from app.documents import registry

UNKNOWN = "unknown"

# Types the CMS can file under.
#
# Read from the registry rather than restated here. This was a hand-maintained
# tuple and it had already drifted — it named twelve types while the markers
# below scored a different twelve, and nothing failed because no caller used it.
# A second list of the same thing is a second thing to forget to update.
DOCUMENT_TYPES = registry.filing_slugs()

# Arabic-script letters the recogniser mixes freely, folded to one form before
# anything is matched.
#
# PaddleOCR's Arabic model is trained on Arabic, and Urdu uses letters Arabic
# does not: it returns ك for ک, ي or ى for ی, ه for ہ. Matching Urdu written
# correctly against that output finds nothing at all — the words are *there*,
# spelled with the neighbouring letter.
#
# Diacritics go too. They are optional in printed Urdu and the recogniser emits
# them inconsistently, so leaving them in makes a match depend on a mark the
# document may not even carry.
_ARABIC_FOLD = str.maketrans({
    "ك": "ک", "ي": "ی", "ى": "ی", "ئ": "ی",
    "ة": "ہ", "ه": "ہ", "ھ": "ہ", "ۀ": "ہ",
    "أ": "ا", "إ": "ا", "آ": "ا", "ٱ": "ا",
    "ؤ": "و", "ڈ": "د", "ٹ": "ت", "ڑ": "ر", "ں": "ن",
})

_DIACRITICS = re.compile(r"[ً-ْٰ]")


def fold_arabic(text: str) -> str:
    """One spelling per word, whichever the recogniser chose."""
    return _DIACRITICS.sub("", unicodedata.normalize("NFKC", text).translate(_ARABIC_FOLD))


# Markers per type: (pattern, weight). Weights are rough and deliberately so —
# they order evidence, they do not model probability, and presenting them as
# probability would be false precision.
#
# Urdu markers are written folded, and are single words rather than the phrases
# actually printed on a card. That is not laziness — measured on a real old
# all-Urdu CNIC, "قومی شناختی کارڈ" came back as "قويى شنايى كارد", with letters
# *missing* rather than merely substituted. A phrase pattern matches nothing on
# real output; the words that survive intact are the ones worth asking for.
# Evidence per type now lives in the document registry, which is also where a
# new type is added. Moved rather than rewritten: every weight below was arrived
# at against a real document, and the registry carries them and their reasons
# unchanged.
#
# Read once at import, like the dict it replaces. The registry is static data.
_MARKERS = registry.markers()


# A winner must reach this score, beat the runner-up by the margin, AND rest on
# more than one marker.
#
# The hit requirement was added after a test caught the classifier calling
# "please see the invoice discussion in our last meeting" an invoice: one
# keyword scored exactly MIN_SCORE with nothing corroborating it. A single
# mention in prose is a mention; two markers is evidence. Score alone cannot
# express that, because the strongest single marker is worth MIN_SCORE by design.
MIN_SCORE = 3.0
MIN_MARGIN = 1.5
MIN_HITS = 2


@dataclass(frozen=True, slots=True)
class Classification:
    """What the document is, how sure, and on what basis."""

    document_type: str
    confidence: float
    evidence: list[str]
    runner_up: str | None = None

    @property
    def is_confident(self) -> bool:
        return self.document_type != UNKNOWN


def _score(text: str, markers: list[tuple[re.Pattern[str], float]]) -> tuple[float, list[str]]:
    total = 0.0
    hits: list[str] = []

    for pattern, weight in markers:
        match = pattern.search(text)
        if match:
            total += weight
            hits.append(match.group(0)[:40].strip())

    return total, hits


def classify(text: str) -> Classification:
    """Identify a document from its text.

    Returns ``unknown`` rather than a low-confidence guess. A wrong type files a
    document where nobody will look for it, which is worse than an unfiled one
    sitting in a review queue.
    """
    if not text.strip():
        return Classification(UNKNOWN, 0.0, ["no text was recognised"])

    # Folded once, and every marker is matched against the folded form. Latin is
    # untouched by the table, so the English markers behave exactly as before.
    text = fold_arabic(text)

    scored = sorted(
        ((name, *_score(text, markers)) for name, markers in _MARKERS.items()),
        key=lambda row: row[1],
        reverse=True,
    )

    best_name, best_score, best_hits = scored[0]
    runner_name, runner_score, _ = scored[1] if len(scored) > 1 else (None, 0.0, [])

    if best_score < MIN_SCORE:
        return Classification(
            UNKNOWN,
            0.0,
            [f"strongest match '{best_name}' scored {best_score:.1f}, below {MIN_SCORE}"],
            runner_up=best_name,
        )

    if len(best_hits) < MIN_HITS:
        return Classification(
            UNKNOWN,
            0.0,
            [
                f"'{best_name}' rests on a single marker ({best_hits[0] if best_hits else '?'}) "
                f"— one mention is not evidence"
            ],
            runner_up=best_name,
        )

    if best_score - runner_score < MIN_MARGIN:
        return Classification(
            UNKNOWN,
            0.0,
            [f"'{best_name}' and '{runner_name}' scored too close ({best_score:.1f} vs {runner_score:.1f})"],
            runner_up=runner_name,
        )

    # Scaled against a score that represents strong agreement, and capped: a
    # document cannot be more than 99% certain from keyword evidence alone, and
    # claiming 100% would invite a caller to skip the human.
    confidence = min(0.99, best_score / 8.0)

    return Classification(best_name, round(confidence, 2), best_hits, runner_up=runner_name)
