"""How often does the reader get the characters right?

Every OCR number in this project until now has been throughput or the engine's
own confidence score. Confidence is the engine grading its own work; it says
nothing about whether the text is correct. This measures correctness against
text that is known because we wrote it.

WHY THE DOCUMENTS ARE SYNTHETIC

Ground truth has to come from somewhere. The obvious source — real documents
from the installation, transcribed by hand — is closed off: those are Level 3
under ADR-0002, and transcribing one means moving a client's CNIC number out of
the installation and into whatever tool did the transcribing. A measurement that
requires breaking the data boundary is not a measurement worth having.

So the text here is authored. Every identifier is invented, follows the shape of
a real one and belongs to nobody.

WHAT THIS DOES AND DOES NOT TELL YOU

It measures the engine's character accuracy on rendered text at several
qualities, from a clean render down to a poor phone photograph. That covers the
fields the system actually acts on — CNIC numbers, NTNs, dates, amounts — which
are Latin and numeric.

It does NOT measure:

  * **Urdu.** Rendering Urdu correctly needs text shaping and bidirectional
    layout that this environment cannot do faithfully, so generated Urdu would
    come out as disconnected letterforms. Scoring the engine against a broken
    render would produce a low number that says nothing about the engine.
  * **Real photographs.** Synthetic degradation approximates a bad scan; it does
    not reproduce creases, glare, a thumb over the corner, or a laminated card
    reflecting a ceiling light.
  * **Layout.** Whether the right value landed in the right field is a different
    question from whether the characters were read correctly.

Treat the clean-render figure as a ceiling — the engine will not do better on a
real document than on a perfect render of the same text.

    python scripts/ocr_accuracy.py
"""

from __future__ import annotations

import statistics
import sys
import unicodedata
from dataclasses import dataclass
from pathlib import Path
from tempfile import TemporaryDirectory

sys.path.insert(0, str(Path(__file__).resolve().parent.parent))

from PIL import Image, ImageDraw, ImageFilter, ImageFont  # noqa: E402

#: Lines resembling the fields this system reads, all of them invented.
#:
#: Identifier shapes are real — a CNIC is 13 digits as 5-7-1, an NTN is 7 — so
#: the engine faces the same digit runs and separators it would in the field.
#: The numbers themselves are not anybody's.
DOCUMENT = [
    "GOVERNMENT OF PAKISTAN",
    "NATIONAL IDENTITY CARD",
    "Name: Muhammad Test Sample",
    "Father Name: Abdul Example",
    "Identity Number: 35202-7418639-5",
    "Date of Birth: 14.03.1985",
    "Date of Issue: 22.11.2019",
    "Date of Expiry: 22.11.2029",
    "NTN: 4471903",
    "Tax Year: 2025",
    "Taxable Income: 2,847,500",
    "Tax Deducted: 284,750",
]


@dataclass(frozen=True)
class Condition:
    """One rendering quality, and what it is meant to stand in for."""

    name: str
    stands_for: str
    blur: float = 0.0
    rotate: float = 0.0
    scale: float = 1.0
    noise: int = 0


#: A ladder, not a sample.
#:
#: The first five conditions were written as "realistic" and the engine scored
#: 100% on every one of them, which measured the difficulty of the test rather
#: than the accuracy of the reader. Everything below "poor" exists to find where
#: it actually breaks — a threshold is a usable fact, and five perfect scores
#: are not.
CONDITIONS = [
    Condition("clean", "a PDF or a flatbed scan", scale=1.0),
    Condition("slight blur", "a steady phone photograph", blur=1.0),
    Condition("rotated", "a page photographed off-square", rotate=1.5, blur=0.5),
    Condition("small", "a photograph taken too far away", scale=0.45, blur=0.4),
    Condition("poor", "a hurried photograph in bad light", blur=1.8, rotate=2.5, scale=0.7, noise=18),
    Condition("very blurred", "a photograph taken while moving", blur=3.5),
    Condition("tiny", "a page filling a fifth of the frame", scale=0.22, blur=0.5),
    Condition("noisy", "a dark room, sensor pushed hard", noise=70, blur=1.0),
    Condition("bad", "a phone photograph nobody checked", blur=3.0, rotate=4.0, scale=0.35, noise=45),
    Condition("worst", "past anything worth filing", blur=5.0, rotate=6.0, scale=0.2, noise=80),
]


def render(lines: list[str], condition: Condition, into: Path) -> Path:
    """Draw the text, then degrade it the way a camera would."""
    from PIL import ImageOps

    width, height, margin, spacing = 1240, 780, 60, 58
    image = Image.new("L", (width, height), 255)
    draw = ImageDraw.Draw(image)
    font = ImageFont.truetype("C:/Windows/Fonts/arial.ttf", 34)

    for index, line in enumerate(lines):
        draw.text((margin, margin + index * spacing), line, font=font, fill=25)

    if condition.rotate:
        image = image.rotate(condition.rotate, expand=True, fillcolor=255, resample=Image.BICUBIC)

    if condition.scale != 1.0:
        # Downscale and back up: the information a distant photograph loses is
        # gone, and enlarging it afterwards does not bring it back.
        small = image.resize(
            (int(image.width * condition.scale), int(image.height * condition.scale)),
            Image.LANCZOS,
        )
        image = small.resize(image.size, Image.LANCZOS)

    if condition.blur:
        image = image.filter(ImageFilter.GaussianBlur(condition.blur))

    if condition.noise:
        import random

        rng = random.Random(20260803)
        pixels = image.load()

        for y in range(0, image.height, 2):
            for x in range(0, image.width, 2):
                pixels[x, y] = max(0, min(255, pixels[x, y] + rng.randint(-condition.noise, condition.noise)))

    image = ImageOps.autocontrast(image)

    path = into / f"{condition.name.replace(' ', '_')}.png"
    image.convert("RGB").save(path)

    return path


def normalise(text: str) -> str:
    """Fold away differences that are not reading errors.

    Case, accents and runs of whitespace are not what this is measuring — a
    reader that returns the right digits in the right order has read the CNIC
    correctly whether or not it capitalised the label above it.
    """
    text = unicodedata.normalize("NFKD", text)
    text = "".join(c for c in text if not unicodedata.combining(c))

    return " ".join(text.lower().split())


def edit_distance(a: str, b: str) -> int:
    """Levenshtein, two rows at a time."""
    if len(a) < len(b):
        a, b = b, a

    previous = list(range(len(b) + 1))

    for i, ca in enumerate(a, 1):
        current = [i]

        for j, cb in enumerate(b, 1):
            current.append(min(
                previous[j] + 1,
                current[j - 1] + 1,
                previous[j - 1] + (ca != cb),
            ))

        previous = current

    return previous[-1]


def main() -> int:
    from app.ocr.config import OcrConfig
    from app.ocr.paddle import PaddleOcrEngine

    truth = normalise(" ".join(DOCUMENT))
    engine = PaddleOcrEngine(OcrConfig())

    print(f"\n  Ground truth: {len(truth)} characters, {len(DOCUMENT)} lines\n")
    print(f"  {'condition':<14}{'CER':>8}{'accuracy':>11}{'digits':>9}{'claimed':>10}   stands for")
    print(f"  {'-' * 84}")

    results = []

    with TemporaryDirectory() as tmp:
        for condition in CONDITIONS:
            path = render(DOCUMENT, condition, Path(tmp))
            result = engine.read(path)

            read = normalise(" ".join(block.text for block in result.blocks))
            errors = edit_distance(read, truth)
            cer = errors / len(truth)

            # Digits scored separately: an identifier is the field where a
            # single wrong character is the whole error, and a document whose
            # prose reads perfectly while a CNIC digit flipped is worse than the
            # overall figure suggests.
            digits_ok = _digit_accuracy(read, truth)

            # The engine's own score, beside the truth. Everything in this
            # project has leant on confidence as a proxy for correctness, and
            # nothing had ever checked whether it is one.
            claimed = result.confidence if result.blocks else 0.0

            results.append((condition, cer, digits_ok, claimed))
            print(
                f"  {condition.name:<14}{cer:>7.1%}{1 - cer:>11.1%}{digits_ok:>9.1%}{claimed:>10.1%}"
                f"   {condition.stands_for}"
            )

    print(f"  {'-' * 84}")
    print(f"  median accuracy across conditions: {statistics.median(1 - c for _, c, _, _ in results):.1%}")
    _report_on_confidence(results)

    return 0


def _report_on_confidence(results) -> None:
    """Is the engine's confidence a usable stand-in for being right?

    It matters because it is what the system already acts on. If confidence
    stays high while accuracy collapses, then every threshold keyed off it is
    letting bad reads through wearing a good score — and a reviewer told a
    document was read with 95% confidence has been told nothing.
    """
    worst = max(results, key=lambda r: r[1])
    condition, cer, _digits, claimed = worst

    print(
        f"\n  Worst read was '{condition.name}': {1 - cer:.1%} actually correct, "
        f"{claimed:.1%} claimed."
    )

    gap = claimed - (1 - cer)

    if gap > 0.25:
        print(
            "  Confidence overstates correctness badly at the failing end, so it "
            "cannot\n  be used alone to decide whether a read is trustworthy."
        )
    elif gap > 0.05:
        print("  Confidence runs ahead of correctness; useful as a signal, not as a gate.")
    else:
        print("  Confidence tracked correctness here.")

    print()


def _digit_accuracy(read: str, truth: str) -> float:
    """How much of the numeric content survived, in order.

    Compared as one string of digits rather than field by field: this script has
    no layout model, and pairing a read value to the field it belongs in is the
    question it explicitly does not answer.
    """
    read_digits = "".join(c for c in read if c.isdigit())
    truth_digits = "".join(c for c in truth if c.isdigit())

    if not truth_digits:
        return 1.0

    return max(0.0, 1 - edit_distance(read_digits, truth_digits) / len(truth_digits))


if __name__ == "__main__":
    raise SystemExit(main())
