"""Pulling structured fields out of recognised text.

Everything here is Pakistan-shaped, because that is what this practice files:
CNICs, PK IBANs, 03xx mobiles, FBR tax years. A generic extractor would match
loosely and be wrong quietly, which on a tax record is worse than not matching
at all.

Nothing in this module guesses. A field that cannot be found is absent, never
approximated — an invented CNIC that looks plausible is far more damaging than a
missing one, because nobody checks a field that is already filled in.
"""

from __future__ import annotations

import re
from dataclasses import dataclass
from datetime import date

# ── Patterns ──────────────────────────────────────────────────────────────
#
# Anchored on word boundaries so a 13-digit run inside a longer number is not
# mistaken for a CNIC.

# 5 digits - 7 digits - 1 check digit.
#
# The separator is whatever the recogniser thought it saw. A dash printed on a
# card comes back as a dash, a dot, a middle dot, a space or nothing — and the
# same card photographed twice does not answer the same way each time.
#
# Not hypothetical. One person's card arrived front and back together: the front
# read `35202-7654321-3`, the back read `35202-7654321.3`. The back was refused,
# so it matched no client — and because the number is also the second marker
# `cnic_back` needs, it failed to classify as well. One misread character cost
# both, on a card whose front had already named the person.
#
# Every accepted form normalises to the dashed one, so nothing downstream sees
# which character it was.
_CNIC_SEPARATOR = r"[\s.·•-]?"

CNIC = re.compile(rf"\b(\d{{5}}){_CNIC_SEPARATOR}(\d{{7}}){_CNIC_SEPARATOR}(\d)\b")

# Pakistan IBAN: PK, 2 check digits, 4-letter bank code, 16 alphanumerics.
IBAN_PK = re.compile(r"\bPK\d{2}[A-Z]{4}[A-Z0-9]{16}\b", re.IGNORECASE)

# Mobile: 03xx-xxxxxxx locally, or +92 3xx internationally.
MOBILE = re.compile(r"\b(?:\+?92[\s-]?|0)(3\d{2})[\s-]?(\d{7})\b")

EMAIL = re.compile(r"\b[\w.+-]+@[\w-]+\.[\w.-]+\b")

# Tax years as FBR writes them: a bare year, or 2023-24 / 2023-2024.
TAX_YEAR = re.compile(r"\b(20\d{2})(?:\s*[-/]\s*(\d{2,4}))?\b")

PASSPORT = re.compile(r"\b([A-Z]{2}\d{7})\b")

# Dates: dd/mm/yyyy, dd-mm-yyyy, dd.mm.yyyy. Day-first because that is the
# convention on every Pakistani document this will read; assuming month-first
# would silently swap the two for any day under 13.
DATE = re.compile(r"\b(\d{1,2})[/.-](\d{1,2})[/.-](\d{4})\b")

AMOUNT = re.compile(r"(?:Rs\.?|PKR)\s*([\d,]+(?:\.\d{1,2})?)", re.IGNORECASE)


@dataclass(frozen=True, slots=True)
class ExtractedField:
    """One value, with where it came from.

    ``raw`` is kept so a human reviewing a proposal can see what the extractor
    actually read, not just what it decided — an approval screen showing only
    the cleaned value hides the mistake it is meant to catch.
    """

    value: str
    raw: str
    confidence: float


def _valid_cnic(digits: str) -> bool:
    """Reject impossible CNICs.

    Only structural checks: the first digit is a province code (1-8), and a
    CNIC is never all the same digit. There is no public checksum, so anything
    stronger would be invented certainty.
    """
    if len(digits) != 13:
        return False

    if digits[0] not in "12345678":
        return False

    return len(set(digits)) > 1


def extract_cnic(text: str) -> ExtractedField | None:
    """The identity number, preferring one written the way a card writes it.

    Accepting a dot as a separator is what makes the back of a real card
    readable, and it also lets `12345.6789012 3` in a column of figures look
    like an identity number. A wrong one is worse than none: it files a
    document against a real person who has nothing to do with it.

    So a match punctuated with a dash wins over one that is not. Cards print
    dashes; totals and reference numbers rarely do. It is a preference rather
    than a requirement because a card sometimes comes back with no separators
    at all, and refusing those would lose more than it protects.
    """
    candidates = []

    for match in CNIC.finditer(text):
        digits = f"{match.group(1)}{match.group(2)}{match.group(3)}"

        if not _valid_cnic(digits):
            continue

        candidates.append(match)

    if not candidates:
        return None

    match = next((m for m in candidates if "-" in m.group(0)), candidates[0])

    return ExtractedField(
        value=f"{match.group(1)}-{match.group(2)}-{match.group(3)}",
        raw=match.group(0),
        # High but not certain: the shape is right and the province code is
        # plausible, which is all that can honestly be claimed.
        confidence=0.95,
    )


def extract_iban(text: str) -> ExtractedField | None:
    match = IBAN_PK.search(text)

    if not match:
        return None

    return ExtractedField(value=match.group(0).upper(), raw=match.group(0), confidence=0.95)


def extract_mobile(text: str) -> ExtractedField | None:
    match = MOBILE.search(text)

    if not match:
        return None

    # Normalised to the local 03xxxxxxxxx form the CMS stores.
    return ExtractedField(
        value=f"0{match.group(1)}{match.group(2)}",
        raw=match.group(0),
        confidence=0.9,
    )


def extract_email(text: str) -> ExtractedField | None:
    match = EMAIL.search(text)

    if not match:
        return None

    return ExtractedField(value=match.group(0).lower(), raw=match.group(0), confidence=0.95)


def extract_passport(text: str) -> ExtractedField | None:
    match = PASSPORT.search(text)

    if not match:
        return None

    # Lower confidence than a CNIC: two letters and seven digits is a common
    # shape, so a reference number can match it by coincidence.
    return ExtractedField(value=match.group(0), raw=match.group(0), confidence=0.75)


def extract_dates(text: str) -> list[ExtractedField]:
    """Every plausible date, day-first.

    Returns all of them rather than picking: which one is the issue date and
    which the expiry depends on the document type, and that decision belongs
    with something that knows what it is looking at.
    """
    found: list[ExtractedField] = []

    for match in DATE.finditer(text):
        day, month, year = (int(g) for g in match.groups())

        try:
            parsed = date(year, month, day)
        except ValueError:
            # 31/02/2024 and friends. A real document does not contain one, so
            # this is a misread rather than a date to salvage.
            continue

        found.append(
            ExtractedField(value=parsed.isoformat(), raw=match.group(0), confidence=0.85)
        )

    return found


def extract_tax_year(text: str) -> ExtractedField | None:
    """The tax year, if the document states one.

    Bounded to a sane window: a "20xx" match is otherwise satisfied by any
    four-digit number on the page, including an amount or a reference.
    """
    current = date.today().year

    for match in TAX_YEAR.finditer(text):
        year = int(match.group(1))

        if 2000 <= year <= current + 1:
            return ExtractedField(value=str(year), raw=match.group(0), confidence=0.8)

    return None


def extract_amounts(text: str) -> list[ExtractedField]:
    """Currency amounts, most significant first.

    Only matches figures explicitly marked Rs or PKR. An unmarked number on a
    tax document is as likely to be a year, a reference or a page number, and
    treating it as money is how a wrong figure reaches an invoice.
    """
    found = [
        ExtractedField(value=m.group(1).replace(",", ""), raw=m.group(0), confidence=0.85)
        for m in AMOUNT.finditer(text)
    ]

    return sorted(found, key=lambda f: float(f.value), reverse=True)


def extract_all(text: str) -> dict[str, object]:
    """Everything findable, with absent fields omitted rather than nulled.

    An omitted key says "not found". A key with None invites a caller to write
    that None over a value the CMS already holds.
    """
    result: dict[str, object] = {}

    for name, extractor in (
        ("cnic", extract_cnic),
        ("iban", extract_iban),
        ("mobile", extract_mobile),
        ("email", extract_email),
        ("passport", extract_passport),
        ("tax_year", extract_tax_year),
    ):
        found = extractor(text)
        if found is not None:
            result[name] = {"value": found.value, "raw": found.raw, "confidence": found.confidence}

    if dates := extract_dates(text):
        result["dates"] = [{"value": d.value, "raw": d.raw} for d in dates]

    if amounts := extract_amounts(text):
        result["amounts"] = [{"value": a.value, "raw": a.raw} for a in amounts]

    return result
