"""Reading a client identifier out of what somebody typed.

When a firm forwards a bank statement to its own self-chat, the caption is the
only thing that says whose it is. Nothing opens the document — a bank statement's
contents are Level 3 and stay inside the installation unread (ADR-0002) — so
this module is the entire basis for the client on that filing.

That makes being wrong here expensive in a specific way: a misread identifier
does not fail, it succeeds against somebody else. So nothing here guesses.

## Why a CNIC needs no keyword and a file number does

A CNIC states what it is by its shape: five digits, seven digits, one digit, in
that order. A run of thirteen digits in that arrangement is not something a
person types by accident.

A file number is a bare integer — this practice's run from single digits into the
thousands, some with a letter suffix like "16 A". There is no shape to recognise.
"Sent on 12" and "3 pages" and "2024" all contain one, and a matcher that took
any number would file a document against client 12 because somebody mentioned a
date. So a file number is read only when it is labelled:

    file 1420 · file no 1420 · file no. 1420 · file# 1420 · f-1420 · file: 1420

That is a real constraint on the customer, and it is the right trade. The cost of
requiring the word "file" is one word. The cost of not requiring it is a bank
statement on a stranger's record.
"""

from __future__ import annotations

import re
from dataclasses import dataclass

# The CNIC pattern is the OCR extractor's, imported rather than rewritten.
#
# It has been through two real corrections — separators the recogniser invents,
# and impossible province codes — and a second copy would be a second thing to
# fix the next time somebody's card reads oddly. A caption is typed rather than
# recognised, so the separator tolerance is unnecessary here and harmless.
from app.ocr.extraction import extract_cnic

#: A file number, and only when it says so.
#:
#: The label is required (see the module docstring). The number itself is one to
#: six digits with an optional single-letter suffix, which is what this
#: practice's numbering actually produces — "16 A" exists, "16 ABC" does not,
#: and matching the second would turn "file 16 August" into client 16A.
#: Two shapes, because two numbering schemes are in use. The plain one is this
#: practice's own ("16", "16 A"). The structured one is a reference like
#: TP-2026-00125, which the intake rule names explicitly — letters, then
#: hyphen- or slash-separated parts.
#:
#: The structured form is tried first and is only ever reached after the label,
#: which is what keeps it from being greedy: "file number for August" cannot
#: match it, because "for August" has a space where it needs a separator.
FILE_NUMBER = re.compile(
    r"""
    \b
    (?: file \s* (?: number | num | no\.? )? | f )   # file, file no., file num, f
    \s* [:#\-]? \s*
    (?:
        (?P<reference>                               # TP-2026-00125, TP/2026/00125
            [A-Za-z]{2,6} (?: [-/] [A-Za-z0-9]{1,8} ){1,3}
        )
        |
        (?P<digits>\d{1,6})                          # the number
        \s* (?P<suffix>[A-Za-z])?                    # an optional single-letter suffix
    )
    \b
    """,
    re.IGNORECASE | re.VERBOSE,
)

#: What the caller gets told the identifier was.
FILE_NUMBER_TYPE = "file_number"
CNIC_TYPE = "cnic"


@dataclass(frozen=True, slots=True)
class Identifiers:
    """What a caption named, if anything.

    Both may be present. Neither is preferred here — precedence is the client
    lookup's decision, because preferring one means nothing until you know
    whether they agree, and only the CMS can answer that.
    """

    file_number: str | None = None
    cnic: str | None = None

    @property
    def any(self) -> bool:
        """Whether this caption identifies a client at all.

        The routing question. A message with an identifier is a filing
        instruction; one without is somebody making a note to themselves, and
        goes to the reading workflow as it always has.
        """
        return bool(self.file_number or self.cnic)

    @property
    def both(self) -> bool:
        return bool(self.file_number and self.cnic)

    def describe(self) -> str:
        """For a log line. Masked, because a CNIC is Level 3 (ADR-0002).

        The self-chat is the firm's own, but this goes to a file that is
        collected and kept, and the number belongs to their client rather than
        to them.
        """
        parts = []

        if self.file_number:
            parts.append(f"file number {self.file_number}")

        if self.cnic:
            parts.append(f"CNIC …{self.cnic[-4:]}")

        return " and ".join(parts) or "nothing"


def read(caption: str | None) -> Identifiers:
    """Pull whatever identifiers a caption carries.

    Never raises and never guesses: an unreadable caption produces an
    ``Identifiers`` with nothing in it, which routes the message to the reading
    workflow rather than filing it against a number nobody typed.
    """
    text = (caption or "").strip()

    if not text:
        return Identifiers()

    return Identifiers(file_number=_file_number(text), cnic=_cnic(text))


def _file_number(text: str) -> str | None:
    """The first labelled file number, normalised.

    First rather than best. A caption naming two file numbers is a person being
    unclear, and picking between them is exactly the guess this module does not
    make — the CMS lookup will find one client for the first and the reviewer
    sees the caption on screen either way.
    """
    match = FILE_NUMBER.search(text)

    if not match:
        return None

    if reference := match.group("reference"):
        # Left as written, upper-cased. There are no leading zeros to strip
        # from a reference, and its parts carry meaning — normalising them
        # would be inventing a scheme rather than reading one.
        return reference.upper()

    digits, suffix = match.group("digits"), match.group("suffix")

    # Leading zeros stripped: "file 0016" and "file 16" are one client, and the
    # CMS stores the number without them. Guarded so "file 0" survives as "0"
    # rather than becoming empty.
    digits = digits.lstrip("0") or "0"

    return f"{digits} {suffix.upper()}" if suffix else digits


def _cnic(text: str) -> str | None:
    """The CNIC a caption contains, as thirteen digits.

    Digits only, no dashes. The CMS matches punctuation-insensitively either
    way, and sending one canonical form means a caption written three different
    ways cannot produce three different-looking proposals for one client.
    """
    field = extract_cnic(text)

    if field is None:
        return None

    return re.sub(r"\D", "", field.value)
