"""Pulling out the fields that matter for a particular kind of document.

`app/ocr/extraction.py` finds things that look like themselves anywhere in any
text — a CNIC, an IBAN, a mobile number. This module finds things that only mean
something once you know what the document is: "Gross Salary 150,000" is a gross
salary on a salary slip and a coincidence on a bank statement.

So everything here is keyed by filing type, and a type with no extractor is not
an omission — most documents are attachments, and inventing fields for them
would put guesses in front of a reviewer.

## Labelled values, and why the label is required

Every pattern below anchors on the label printed on the document, never on
position or on the shape of the value alone. A number near the bottom of a
salary slip is not a net salary; a number after the words "Net Pay" is.

That makes these extractors miss things — a slip using wording nobody
anticipated returns nothing for that field. Missing is the correct failure.
An extracted field is shown to a reviewer as fact, and a plausible wrong number
is worse than a blank, because nobody checks a field that is already filled in.

## Nothing here decides anything

Extraction produces evidence for a person. No field routes a document, chooses a
client on its own, or changes what is filed — those decisions belong to the
classifier, the client lookup and the reviewer respectively.
"""

from __future__ import annotations

import re
from typing import Callable

from app.ocr.extraction import (
    ExtractedField,
    extract_amounts,
    extract_cnic,
    extract_dates,
    extract_mobile,
    extract_tax_year,
)

#: Money as these documents print it: `Rs. 150,000`, `150,000/-`, `PKR 1,20,000`.
_MONEY = r"(?:Rs\.?|PKR)?\s*([\d,]+(?:\.\d{1,2})?)\s*/?-?"

#: A person or company name on a labelled line. Deliberately not a name
#: *recogniser* — it takes whatever follows the label to the end of the line,
#: because a matcher that decided what looked like a name would drop every name
#: it had not seen before, which in Pakistan is most of them.
_NAME = r"([^\n\r]{2,60})"


def _labelled(label: str, value: str = _NAME) -> re.Pattern[str]:
    """A pattern for `Label: value`, however the document punctuates it."""
    return re.compile(rf"{label}\s*[:\-–]?\s*{value}", re.I)


def _first(text: str, pattern: re.Pattern[str], confidence: float = 0.8) -> ExtractedField | None:
    match = pattern.search(text)

    if not match:
        return None

    value = (match.group(1) or "").strip(" .:-\t")

    if not value:
        return None

    return ExtractedField(value=value, raw=match.group(0).strip()[:80], confidence=confidence)


# ── Salary slip ───────────────────────────────────────────────────────────

_EMPLOYER = _labelled(r"\b(?:employer|company|organi[sz]ation)\b")
_EMPLOYEE = _labelled(r"\b(?:employee\s*name|employee|name of employee)\b")
_DESIGNATION = _labelled(r"\b(?:designation|position|job title)\b")
_PERIOD = _labelled(r"\b(?:pay period|salary (?:for the )?month|month|period|for the month of)\b")
_GROSS = _labelled(r"\b(?:gross salary|gross pay|total earnings|gross)\b", _MONEY)
_NET = _labelled(r"\b(?:net (?:salary|pay|amount)|take[\- ]?home|net)\b", _MONEY)
_DEDUCTIONS = _labelled(r"\b(?:total deductions?|deductions?)\b", _MONEY)


def salary_slip(text: str) -> dict[str, object]:
    """Employer, employee, period, gross and net.

    `gross` and `net` are both taken from labels rather than by picking the two
    largest numbers, which is the tempting shortcut and is wrong on any slip
    listing a year-to-date total.
    """
    return _collect(
        text,
        employer=_EMPLOYER,
        employee=_EMPLOYEE,
        designation=_DESIGNATION,
        salary_period=_PERIOD,
        gross_salary=_GROSS,
        net_salary=_NET,
        deductions=_DEDUCTIONS,
    )


# ── Identity ──────────────────────────────────────────────────────────────

_HOLDER = _labelled(r"\bname\b")
_FATHER = _labelled(r"\b(?:father(?:'s)?\s*name|father)\b")
_DOB = _labelled(r"\b(?:date of birth|d\.?o\.?b\.?)\b", r"([\d]{1,2}[./-][\d]{1,2}[./-][\d]{4})")
_ISSUE = _labelled(r"\b(?:date of issue|issue date)\b", r"([\d]{1,2}[./-][\d]{1,2}[./-][\d]{4})")
_EXPIRY = _labelled(r"\b(?:date of expiry|expiry date|valid until)\b",
                    r"([\d]{1,2}[./-][\d]{1,2}[./-][\d]{4})")


def identity(text: str) -> dict[str, object]:
    """CNIC, name, father's name, and the three dates a card carries.

    The identity number comes from the shared extractor rather than a label,
    because it is the one field on the card that identifies itself — and because
    that extractor has already been corrected twice against real cards whose
    separators the recogniser invented.
    """
    fields = _collect(
        text,
        name=_HOLDER,
        father_name=_FATHER,
        date_of_birth=_DOB,
        date_of_issue=_ISSUE,
        date_of_expiry=_EXPIRY,
    )

    if cnic := extract_cnic(text):
        fields["cnic"] = _as_dict(cnic)

    return fields


# ── Property ──────────────────────────────────────────────────────────────

_REGISTRY_NO = _labelled(
    r"\b(?:registry\s*(?:no\.?|number)|deed\s*(?:no\.?|number)|document\s*(?:no\.?|number))\b",
    r"([A-Za-z0-9\-/]{1,30})",
)
_OWNER = _labelled(r"\b(?:owner|vendee|purchaser|buyer|allottee|in favour of)\b")
_SELLER = _labelled(r"\b(?:vendor|seller|transferor)\b")
_AREA = _labelled(
    r"\b(?:area|measuring|total area)\b",
    r"([\d,.]+\s*(?:marla|kanal|sq\.?\s*(?:ft|feet|yards?|yds?)|acres?))",
)
_PLOT = _labelled(r"\b(?:plot\s*(?:no\.?|number)|khasra|khewat)\b", r"([A-Za-z0-9\-/]{1,30})")
_LOCATION = _labelled(r"\b(?:situated at|location|address)\b")


def property_document(text: str) -> dict[str, object]:
    """Registry number, owner, area — and who it came from.

    The seller is extracted alongside the owner on purpose. A deed names both,
    and a reviewer matching a document to a client needs to know which name is
    which — showing only "owner" when the extractor picked up the wrong line is
    how a document lands on the other party's file.
    """
    return _collect(
        text,
        registry_number=_REGISTRY_NO,
        owner=_OWNER,
        seller=_SELLER,
        area=_AREA,
        plot_number=_PLOT,
        location=_LOCATION,
    )


# ── Tax documents ─────────────────────────────────────────────────────────

_NTN = _labelled(r"\b(?:ntn|national tax number)\b", r"([\d]{7,8}-?[\d]?)")
_STRN = _labelled(r"\b(?:strn|sales tax registration(?:\s*number)?)\b", r"([\d\-]{7,20})")
_RETURN_TYPE = _labelled(
    r"\b(?:return type|type of return|nature of return)\b", r"([A-Za-z ]{3,40})"
)
_CERTIFICATE_TYPE = _labelled(
    r"\b(?:certificate (?:type|of)|nature of certificate)\b", r"([A-Za-z ]{3,40})"
)
_TAXABLE = _labelled(r"\b(?:taxable income|total income)\b", _MONEY)
_TAX_PAID = _labelled(r"\b(?:tax (?:paid|deducted|payable)|amount of tax)\b", _MONEY)


def tax_document(text: str) -> dict[str, object]:
    """Tax year, NTN, return or certificate type, and the headline figures.

    `tax_year` comes from the shared extractor, which understands the several
    ways FBR writes one — `2025`, `2023-24`, `2023-2024` — and normalises them.
    """
    fields = _collect(
        text,
        ntn=_NTN,
        strn=_STRN,
        return_type=_RETURN_TYPE,
        certificate_type=_CERTIFICATE_TYPE,
        taxable_income=_TAXABLE,
        tax_paid=_TAX_PAID,
    )

    if year := extract_tax_year(text):
        fields["tax_year"] = _as_dict(year)

    return fields


# ── Dispatch ──────────────────────────────────────────────────────────────
#
# Keyed by filing type. A type absent here has no specific fields worth
# extracting — a utility bill or a receipt is an attachment, and inventing
# fields for it would put guesses in front of a reviewer.

_BY_TYPE: dict[str, Callable[[str], dict[str, object]]] = {
    "salary_slip": salary_slip,
    "cnic_front": identity,
    "cnic_back": identity,
    "passport": identity,
    "driving_licence": identity,
    "property_document": property_document,
    "tax_return": tax_document,
    "wealth_statement": tax_document,
    "tax_certificate": tax_document,
    "fbr_notice": tax_document,
    "ntn_certificate": tax_document,
    "strn_certificate": tax_document,
}


def has_extractor(filing_type: str) -> bool:
    return filing_type in _BY_TYPE


def extract(filing_type: str, text: str) -> dict[str, object]:
    """Fields for this kind of document, plus the ones worth having on any.

    The general extractor runs for every type, because a mobile number or a date
    on a document is useful whatever the document is — and because the client
    lookup falls back to a mobile when no exact identifier was supplied.

    Type-specific fields win on a name clash. A CNIC extracted by the identity
    extractor and by the general one are the same value; the specific one
    carries the better context.
    """
    fields: dict[str, object] = {}

    if mobile := extract_mobile(text):
        fields["mobile"] = _as_dict(mobile)

    if cnic := extract_cnic(text):
        fields["cnic"] = _as_dict(cnic)

    specific = _BY_TYPE.get(filing_type)

    if specific is not None:
        fields.update(specific(text))

    return fields


# ── Internals ─────────────────────────────────────────────────────────────


def _collect(text: str, **patterns: re.Pattern[str]) -> dict[str, object]:
    """Run every pattern, keeping only what matched.

    An absent key means "not found". A key holding None invites a caller to
    write that None over a value the CMS already has.
    """
    found: dict[str, object] = {}

    for name, pattern in patterns.items():
        if field := _first(text, pattern):
            found[name] = _as_dict(field)

    return found


def _as_dict(field: ExtractedField) -> dict[str, object]:
    return {"value": field.value, "raw": field.raw, "confidence": field.confidence}


def all_amounts(text: str) -> list[dict[str, object]]:
    """Every money figure, for a reviewer who wants to check one that was missed."""
    return [{"value": a.value, "raw": a.raw} for a in extract_amounts(text)]


def all_dates(text: str) -> list[dict[str, object]]:
    return [{"value": d.value, "raw": d.raw} for d in extract_dates(text)]
