"""Score the reader against real documents, without the text ever leaving.

`ocr_accuracy.py` measures the *engine* on authored text. This measures the
*deployment* on its own documents — photographs of laminated cards, creases,
glare, real handwriting in the margins — which is the number that actually
describes what a firm will experience.

WHO TRANSCRIBES, AND WHY IT IS NOT AUTOMATED

A person inside the installation, by hand. That is not an oversight; it is the
whole reason this script can exist.

Client documents are Level 3 under ADR-0002 and do not leave the installation.
Any automated transcription means sending the document to something — a hosted
OCR, an assistant, a model API — and the moment that happens the CNIC has left,
which is precisely what the classification forbids. So the ground truth is
produced by somebody who is already permitted to read the document, on the
machine that already holds it.

    documents/  scan-01.jpg      truth/  scan-01.txt
                scan-02.pdf              scan-02.txt

Type what the document says. Reading order, one line per line. Do not correct
the document's own spelling — the measurement is whether the reader saw what is
there, not whether what is there is right.

WHAT THIS PRINTS

Counts and percentages. Never a line of either the document or the transcription,
not in output, not in a log, not on a failure. A benchmark that spills the text
it was scoring has re-created the problem the boundary exists for — so documents
are identified by position, never by filename, because a filename is routinely
somebody's name.

    python scripts/ocr_accuracy_real.py --documents ./documents --truth ./truth
"""

from __future__ import annotations

import argparse
import statistics
import sys
from pathlib import Path

sys.path.insert(0, str(Path(__file__).resolve().parent.parent))

from scripts.ocr_accuracy import edit_distance, normalise  # noqa: E402

READABLE = {".jpg", ".jpeg", ".png", ".pdf", ".webp", ".bmp", ".tif", ".tiff"}

#: Below this, the sample is too small to say anything with.
#:
#: Not a hard refusal — a run of three is still worth looking at while the set
#: is being built — but a figure from four documents quoted as "the accuracy of
#: the system" is the kind of number that ends up in front of a customer.
ADVISED_MINIMUM = 10


def main() -> int:
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--documents", required=True, type=Path)
    parser.add_argument("--truth", required=True, type=Path)
    args = parser.parse_args()

    if not args.documents.is_dir():
        print(f"  No such directory: {args.documents}")

        return 2

    documents = sorted(p for p in args.documents.iterdir() if p.suffix.lower() in READABLE)

    if not documents:
        print(f"  Nothing readable in {args.documents}")

        return 2

    pairs, untranscribed = [], []

    for path in documents:
        truth_file = args.truth / f"{path.stem}.txt"

        if truth_file.is_file() and truth_file.read_text(encoding="utf-8").strip():
            pairs.append((path, truth_file))
        else:
            untranscribed.append(path)

    if untranscribed:
        # By count. Naming them would print filenames, and a filename is
        # routinely a client's name.
        print(f"\n  {len(untranscribed)} document(s) have no transcription and were skipped.")
        print(f"  Write one per document as {args.truth}\\<same name>.txt")

    if not pairs:
        print("\n  Nothing to score.\n")

        return 1

    return _score(pairs)


def _score(pairs) -> int:
    from app.ocr.config import OcrConfig
    from app.ocr.paddle import PaddleOcrEngine

    engine = PaddleOcrEngine(OcrConfig())
    rows = []

    print(f"\n  Scoring {len(pairs)} document(s). No text is printed.\n")
    print(f"  {'document':<12}{'chars':>8}{'accuracy':>11}{'digits':>9}{'claimed':>10}")
    print(f"  {'-' * 50}")

    for index, (path, truth_file) in enumerate(pairs, 1):
        truth = normalise(truth_file.read_text(encoding="utf-8"))
        result = engine.read(path)

        if result.failed:
            # A read that failed is not a read that scored zero, and averaging
            # it in as 0% would blame the engine's accuracy for a timeout.
            print(f"  {index:<12}{len(truth):>8}{'unread':>11}{'-':>9}{'-':>10}")
            continue

        read = normalise(" ".join(block.text for block in result.blocks))
        accuracy = max(0.0, 1 - edit_distance(read, truth) / max(len(truth), 1))
        digits = _digit_accuracy(read, truth)
        claimed = result.confidence if result.blocks else 0.0

        rows.append((accuracy, digits, claimed))
        print(f"  {index:<12}{len(truth):>8}{accuracy:>11.1%}{digits:>9.1%}{claimed:>10.1%}")

    if not rows:
        print("\n  Every document failed to read. That is a reader problem, not an accuracy figure.\n")

        return 1

    print(f"  {'-' * 50}")
    print(f"  {'median':<12}{'':>8}{statistics.median(r[0] for r in rows):>11.1%}"
          f"{statistics.median(r[1] for r in rows):>9.1%}"
          f"{statistics.median(r[2] for r in rows):>10.1%}")
    print(f"  {'worst':<12}{'':>8}{min(r[0] for r in rows):>11.1%}"
          f"{min(r[1] for r in rows):>9.1%}{min(r[2] for r in rows):>10.1%}")

    _comment_on_confidence(rows)

    if len(rows) < ADVISED_MINIMUM:
        print(f"  {len(rows)} document(s) is below the {ADVISED_MINIMUM} this was meant to be run "
              "with.\n  Worth looking at; not worth quoting.")

    print()

    return 0


def _comment_on_confidence(rows) -> None:
    """Whether the engine's own score can be trusted on real input.

    The synthetic run found it overstating correctness by 21 points at the
    failing end while still halving — a usable signal, not a usable gate. Real
    documents are where that either holds or does not, and it matters because
    the system already routes on confidence.
    """
    gaps = [claimed - accuracy for accuracy, _digits, claimed in rows]
    median_gap = statistics.median(gaps)

    print(f"\n  Confidence runs {median_gap:+.1%} against measured accuracy (median).")

    if median_gap > 0.25:
        print("  It overstates badly enough that no threshold keyed to it is meaningful.")
    elif median_gap > 0.05:
        print("  It overstates. Usable to rank documents, not to certify one.")
    elif median_gap < -0.05:
        print("  It understates — documents are being flagged that read correctly.")
    else:
        print("  It tracks measured accuracy on this set.")


def _digit_accuracy(read: str, truth: str) -> float:
    read_digits = "".join(c for c in read if c.isdigit())
    truth_digits = "".join(c for c in truth if c.isdigit())

    if not truth_digits:
        return 1.0

    return max(0.0, 1 - edit_distance(read_digits, truth_digits) / len(truth_digits))


if __name__ == "__main__":
    raise SystemExit(main())
