"""What documents this system knows about, and what it does with each.

One table, and adding a type is one entry in it. That is the whole point: the
vocabulary of a tax practice grows — withholding certificates, ATL certificates,
partnership deeds — and a system where each addition means editing a classifier,
a router and a schema is one where the list stays short out of inertia rather
than out of judgement.

## Two levels, and why

**Filing types** are what the CMS stores in `documents.document_type`. They are
deliberately coarse, because a client's file is browsed and filtered by people
and thirty filter options are not thirty times more useful than twenty.

**Refinements** are what the document actually is. A sale deed, a purchase deed
and a registry are all filed as `property_document`, but they are not the same
thing, and telling a reviewer "Property Document" when the classifier knows it
saw a sale deed throws away the most useful thing it worked out.

So a classification carries both: the filing type the CMS will store, and the
refinement it was recognised as. Nothing is lost and the filter list stays
readable.

## The filing list is not authoritative here

The CMS owns it — `Document::$typeLabels`, served at `GET agent/v1/document-types`.
This module must agree with that list, and `preflight` checks that it does at
startup rather than discovering a mismatch when a filing is refused. A type here
that the installation does not have would produce a proposal no reviewer can
approve.

## Refinements never compete with filing types

They are resolved in a second pass, *within* the filing type the rule engine
already chose, and never against it. That ordering is not a detail. `sale deed`
is a marker for `property_document`; if a `sale_deed` type scored against the
same text at the same time, the two would tie, the margin rule would refuse to
pick either, and a document that used to classify correctly would come back
`unknown`. Refining after the decision keeps every existing classification
exactly as it was.
"""

from __future__ import annotations

import re
from dataclasses import dataclass, field
from enum import StrEnum
from typing import Pattern

from app.ocr.extraction import CNIC, IBAN_PK


class ProcessingStrategy(StrEnum):
    """How a type's client is worked out, decided by the type rather than the file.

    The distinction that matters is not PDF-versus-image. It is whether the
    document can be expected to name its own client. A photograph of a CNIC
    carries the number on its face; a bank statement carries an account number
    and a balance and, very often, no client identity this system can use. So
    the first is read and the second is asked about, and the file format is a
    coincidence in both cases.
    """

    OCR_REQUIRED = "ocr_required"
    """Read it now. The document is expected to identify its own client, and
    making somebody type a file number for every photograph they send would be
    a worse workflow than the one already working."""

    USER_METADATA_REQUIRED = "user_metadata_required"
    """Never read it; wait to be told whose it is.

    For a bank statement this is a data-protection decision, not a workflow
    preference: its contents are Level 3 under ADR-0002 and the cheapest way to
    guarantee they never leave the installation is never to hold them.
    """

    OCR_WITH_METADATA_FALLBACK = "ocr_with_metadata_fallback"
    """Read it, and ask only if that failed to name a client.

    For the documents that usually identify their client and sometimes do not —
    a salary slip with the name in a logo, a certificate scanned at an angle.
    The fallback costs a message only in the cases that would otherwise have
    reached a reviewer with nothing attached.
    """

    @property
    def reads_document(self) -> bool:
        return self is not ProcessingStrategy.USER_METADATA_REQUIRED

    @property
    def may_wait(self) -> bool:
        return self is not ProcessingStrategy.OCR_REQUIRED


#: What may be extracted, in the order it is trusted.
#:
#: A file number is a person's assertion about their own filing system and beats
#: anything printed on a page. A CNIC is the strongest printed identifier.
#: Beyond those the list is declared but not yet implemented — registering the
#: order now is what lets a new identifier be added without the engine changing,
#: and claiming extraction that does not exist would be worse than saying so.
IDENTIFIER_PRIORITY: tuple[str, ...] = ("file_number", "cnic")

#: Declared for the types that will use them, unimplemented until there are real
#: samples to write patterns against. The CNIC separator bugs came from guessing
#: at a format; these will not repeat that.
IDENTIFIERS_NOT_YET_EXTRACTED: tuple[str, ...] = ("ntn", "passport_number", "licence_number")

#: The workflow a document goes to when its type has no workflow of its own.
#:
#: The reading workflow — OCR, classify, extract, identify, propose. It handles
#: any document adequately and none of them specially, which is exactly what a
#: default should do. A type registered without a workflow therefore still
#: works, which is what makes "register a type" genuinely one step.
#:
#: **Only workflows that exist are named in this table.** The obvious
#: alternative — naming `salary_slip_intake` now and building it later — puts a
#: dead name in the registry, and the failure is a document routed into nothing
#: at the moment a real one arrives. Each per-type workflow claims its types
#: when it is built.
GENERIC_WORKFLOW = "document_intake"


@dataclass(frozen=True, slots=True)
class FilingType:
    """A type the CMS can store, and how this system handles it."""

    slug: str
    label: str

    markers: tuple[tuple[Pattern[str], float], ...] = ()
    """Weighted evidence for the rule engine. Weights order evidence; they do
    not model probability, and presenting them as probability would be false
    precision."""

    filename: tuple[Pattern[str], ...] = ()
    """Patterns that identify this type from a filename alone, before anything
    is opened. `Meezan_Bank_Statement.pdf` is not proof, but it is a strong
    prior and it costs nothing."""

    aliases: tuple[Pattern[str], ...] = ()
    """Extra ways a person might name this type in a caption.

    The label itself is always matched and is not repeated here. These are for
    what people actually type — "paysalip", "shanakhti card", "ATL" — and for
    the cases where the obvious word is dangerous on its own. See
    `caption_patterns` for the one that matters."""

    workflow: str = GENERIC_WORKFLOW

    strategy: ProcessingStrategy = ProcessingStrategy.OCR_WITH_METADATA_FALLBACK
    """How this type's client is worked out.

    The default reads the document and asks only if that named nobody, which is
    the right behaviour for most of a tax practice's paperwork: it usually
    identifies its own client and occasionally does not.

    A type that must never be opened says so here, once, and every part of the
    pipeline reads it from this one place rather than remembering."""

    allowed_file_types: tuple[str, ...] = ("pdf", "jpg", "jpeg", "png")
    """Which formats may be filed as this type.

    Declared per type rather than globally because it is a property of the
    document: a CNIC arrives as a photograph and a tax return as a PDF, and a
    reviewer has to be able to look at whatever is attached."""

    preprocess: bool = True
    """Whether a photograph of this type is cropped and cleaned before reading.

    On by default: a client's file should hold the document, not a snap of a
    desk with a hand in it. Off is for a type where the surroundings carry
    meaning — a photograph taken to show where something was, rather than what
    it says.

    Ignored for PDFs, which are not photographs of anything."""

    identifier_priority: tuple[str, ...] = IDENTIFIER_PRIORITY
    """Which identifiers to trust, in order, when this type is read.

    Per type because the order genuinely differs: a salary slip carries a CNIC,
    a tax document carries an NTN, and trying them in the wrong order finds the
    employer instead of the employee."""

    sensitive: bool = False
    """Whether misfiling this is expensive. Feeds the CMS's risk scoring, which
    already treats a misfiled bank statement as worse than a misfiled receipt."""

    @property
    def reads_document(self) -> bool:
        """Whether this type may be OCR'd at all.

        Derived from the strategy rather than stored beside it. The two were
        separate fields for one release and could disagree — a type could
        declare "never read me" and a strategy that reads, and nothing would
        catch it. One of them has to be the truth, and it is the strategy.
        """
        return self.strategy.reads_document


@dataclass(frozen=True, slots=True)
class Refinement:
    """A more precise name for something already classified.

    Resolved only within its own filing type, and only after that type has been
    decided — see the module docstring for why the ordering matters.
    """

    slug: str
    label: str
    files_as: str
    markers: tuple[tuple[Pattern[str], float], ...] = field(default=())
    filename: tuple[Pattern[str], ...] = ()
    aliases: tuple[Pattern[str], ...] = ()


def _p(pattern: str) -> Pattern[str]:
    return re.compile(pattern, re.I)


# ── Filing types ──────────────────────────────────────────────────────────
#
# The markers below were moved here from app/ocr/classification.py unchanged,
# including their weights and their comments, because each one was arrived at
# against a real document and rewriting them would quietly discard that. New
# types are added at the end of each group.

FILING_TYPES: tuple[FilingType, ...] = (
    # ── Identity ─────────────────────────────────────────────────────────
    FilingType(
        slug="cnic_front",
        # Carries its client's identity on the face of it, so it is read
        # rather than asked about.
        strategy=ProcessingStrategy.OCR_REQUIRED,
        label="CNIC Front",
        workflow="identity_intake",
        sensitive=True,
        filename=(_p(r"\bcnic\b"), _p(r"identity[_\- ]?card")),
        aliases=(
            # The trap this whole field exists for.
            #
            # "Bank statement — CNIC 35202-1234567-1" is a bank statement whose
            # caption names the client by CNIC. A bare `cnic` alias would read
            # that as somebody saying the document IS a CNIC, and the two types
            # would fight over one caption.
            #
            # So `cnic` counts as naming the type only when it is not being used
            # as a label for a number. Followed by digits, it is an identifier
            # and belongs to the client lookup, not to classification.
            _p(r"\bcnic\b(?!\s*(no\.?|number|#|:)?\s*\d)"),
            _p(r"\bshanakht[iy]\b"),
            _p(r"identity card"),
        ),
        markers=(
            (_p(r"national identity card"), 3.0),
            (_p(r"\bnadra\b"), 2.0),
            (_p(r"identity number"), 2.0),
            (_p(r"date of birth"), 1.0),
            (CNIC, 2.0),
            # An older card carries no English at all. These six words were
            # tested against a real one and are the ones that came back intact:
            # the header and the field labels. Nothing here is a phrase — on
            # real recogniser output letters go missing, not just substituted.
            (re.compile(r"حکومت"), 2.5),   # government — the card's header
            (re.compile(r"جنس"), 2.0),     # gender — a field label
            (re.compile(r"والد"), 2.0),    # father — a field label
            (re.compile(r"کارد"), 1.5),    # card, folded from کارڈ
        ),
    ),
    FilingType(
        slug="cnic_back",
        # Carries its client's identity on the face of it, so it is read
        # rather than asked about.
        strategy=ProcessingStrategy.OCR_REQUIRED,
        label="CNIC Back",
        workflow="identity_intake",
        sensitive=True,
        markers=(
            # `registrar general` rather than `registrar` alone — a property
            # document mentions a registrar too, and that word belongs to it.
            (_p(r"registrar general"), 3.0),
            (_p(r"\bnadra\b"), 2.0),
            (_p(r"permanent address"), 3.0),
            (_p(r"present address"), 2.0),
            (_p(r"signature.*holder"), 2.0),
            (CNIC, 2.0),
        ),
    ),
    FilingType(
        slug="passport",
        # Carries its client's identity on the face of it, so it is read
        # rather than asked about.
        strategy=ProcessingStrategy.OCR_REQUIRED,
        label="Passport",
        workflow="identity_intake",
        sensitive=True,
        filename=(_p(r"\bpassport\b"),),
        markers=(
            (_p(r"\bpassport\b"), 3.0),
            (_p(r"islamic republic of pakistan"), 1.5),
            (_p(r"place of issue"), 2.0),
            (_p(r"\bP<PAK"), 3.0),
        ),
    ),
    FilingType(
        slug="driving_licence",
        # Carries its client's identity on the face of it, so it is read
        # rather than asked about.
        strategy=ProcessingStrategy.OCR_REQUIRED,
        label="Driving Licence",
        workflow="identity_intake",
        filename=(_p(r"driving[_\- ]?lic"),),
        markers=(
            (_p(r"driving licen[cs]e"), 3.0),
            (_p(r"licence number|license number"), 2.0),
            (_p(r"vehicle class"), 2.0),
        ),
    ),

    # ── Income and banking ───────────────────────────────────────────────
    FilingType(
        slug="bank_statement",
        label="Bank Statement",
        # Never opened. The one type whose handling is a data-protection
        # decision rather than a workflow choice (ADR-0002).
        workflow="bank_statement_intake",
        # Never opened, and the strategy is where that is now said. A statement
        # rarely names its client in a form this system can use — it names an
        # account — so reading it would cost the guarantee and buy nothing.
        strategy=ProcessingStrategy.USER_METADATA_REQUIRED,
        # Photographed statements are real — a phone held over a printout is a
        # normal way for one to arrive. Narrowing this to PDF would refuse them,
        # and it would buy nothing: whether the file is ever opened is decided
        # by the strategy above, which does not care what format it is in.
        allowed_file_types=("pdf", "jpg", "jpeg", "png"),
        sensitive=True,
        filename=(
            # Both orders, and "account statement" as well as "bank statement".
            # This list is what decides that the file is never opened, so a
            # phrasing it does not know is not a missed label — the document
            # falls through to the reading workflow and gets OCR'd, which for
            # this type is the outcome the whole design exists to prevent.
            # Every bank whose naming we have seen is worth adding here.
            _p(r"(?:bank|account)[_\- ]?statement"),
            _p(r"statement[_\- ]?of[_\- ]?account"),
            _p(r"\b(meezan|hbl|ubl|mcb|alfalah|askari|faysal|js|soneri|allied)\b"),
        ),
        markers=(
            (_p(r"statement of account|account statement"), 3.0),
            (_p(r"\bclosing balance\b"), 2.5),
            (_p(r"\bopening balance\b"), 2.5),
            (re.compile(r"\bdebit\b.*\bcredit\b", re.I | re.S), 1.5),
            (IBAN_PK, 2.0),
        ),
    ),
    FilingType(
        slug="bank_letter",
        label="Bank Letter",
        filename=(_p(r"bank[_\- ]?letter"), _p(r"account[_\- ]?maintenance")),
        markers=(
            (_p(r"account maintenance certificate"), 3.0),
            (_p(r"to whom it may concern"), 2.0),
            (_p(r"\bbranch manager\b"), 2.0),
            (_p(r"certif(y|ies) that"), 1.5),
        ),
    ),
    FilingType(
        slug="salary_slip",
        label="Salary Slip",
        workflow="salary_slip_intake",
        filename=(_p(r"salary[_\- ]?slip"), _p(r"pay[_\- ]?slip")),
        markers=(
            (_p(r"salary slip|pay ?slip"), 3.0),
            (_p(r"\bgross salary\b"), 2.5),
            (_p(r"\bnet pay\b"), 2.5),
            (_p(r"\bdeductions?\b"), 1.5),
        ),
    ),

    # ── Tax ──────────────────────────────────────────────────────────────
    FilingType(
        slug="tax_return",
        label="Tax Return",
        workflow="tax_document_intake",
        filename=(_p(r"(income[_\- ]?)?tax[_\- ]?return"), _p(r"\biris\b")),
        markers=(
            (_p(r"\bincome tax return\b"), 3.0),
            (_p(r"\btaxable income\b"), 2.0),
            (_p(r"\biris\b"), 1.5),
            # Kept, deliberately, even though wealth_statement now owns this
            # phrase: an FBR return contains a wealth statement section, and a
            # return that mentions it still outscores one on its own markers.
            (_p(r"\bwealth statement\b"), 1.0),
        ),
    ),
    FilingType(
        slug="wealth_statement",
        label="Wealth Statement",
        workflow="tax_document_intake",
        filename=(_p(r"wealth[_\- ]?statement"),),
        markers=(
            (_p(r"\bwealth statement\b"), 3.0),
            (_p(r"\breconciliation of net assets\b"), 3.0),
            (_p(r"\bnet assets\b"), 2.0),
            (_p(r"\bassets? (and|&) liabilit"), 2.0),
        ),
    ),
    FilingType(
        slug="tax_certificate",
        label="Tax Certificate",
        workflow="tax_document_intake",
        filename=(_p(r"tax[_\- ]?certificate"), _p(r"\batl\b"), _p(r"withholding")),
        markers=(
            (_p(r"withholding tax certificate"), 3.0),
            (_p(r"active taxpayer"), 3.0),
            (_p(r"tax deducted at source"), 2.5),
            (_p(r"\bsection 236\b|\bsection 231\b"), 2.0),
            (_p(r"\bcertificate\b"), 1.0),
        ),
    ),
    FilingType(
        slug="fbr_notice",
        label="FBR Notice",
        workflow="tax_document_intake",
        filename=(_p(r"\bnotice\b"), _p(r"\bfbr\b")),
        markers=(
            (_p(r"\bfederal board of revenue\b"), 3.0),
            (_p(r"\bnotice u/s\b|\bsection 1[0-9]{2}\b"), 3.0),
            (_p(r"\bcommissioner\b"), 1.5),
        ),
    ),

    # ── Registration ─────────────────────────────────────────────────────
    FilingType(
        slug="ntn_certificate",
        label="NTN Certificate",
        workflow="tax_document_intake",
        filename=(_p(r"\bntn\b"),),
        markers=(
            (_p(r"national tax number"), 3.0),
            (_p(r"\bntn\b"), 2.0),
            (_p(r"taxpayer registration"), 2.5),
        ),
    ),
    FilingType(
        slug="strn_certificate",
        label="STRN Certificate",
        workflow="tax_document_intake",
        filename=(_p(r"\bstrn\b"), _p(r"sales[_\- ]?tax[_\- ]?reg")),
        markers=(
            (_p(r"sales tax registration"), 3.0),
            (_p(r"\bstrn\b"), 2.5),
        ),
    ),

    # ── Assets ───────────────────────────────────────────────────────────
    FilingType(
        slug="property_document",
        label="Property Document",
        workflow="property_intake",
        sensitive=True,
        filename=(_p(r"sale[_\- ]?deed"), _p(r"\bregistry\b"), _p(r"\bmutation\b")),
        markers=(
            (_p(r"\bsale deed\b|\bmutation\b|\ballotment\b"), 3.0),
            (_p(r"\bplot no\b|\bkhasra\b|\bkhewat\b"), 2.5),
            (_p(r"\bregistrar\b"), 1.5),
        ),
    ),
    FilingType(
        slug="vehicle_document",
        # Carries its client's identity on the face of it, so it is read
        # rather than asked about.
        strategy=ProcessingStrategy.OCR_REQUIRED,
        label="Vehicle Document",
        filename=(_p(r"vehicle"), _p(r"registration[_\- ]?book")),
        markers=(
            (_p(r"registration book"), 3.0),
            (_p(r"\bchassis\b|\bengine no\b"), 2.5),
            (_p(r"\bexcise (and|&) taxation\b"), 2.5),
        ),
    ),

    # ── Business ─────────────────────────────────────────────────────────
    FilingType(
        slug="business_document",
        label="Business Document",
        filename=(_p(r"partnership"), _p(r"incorporat"), _p(r"\bsecp\b")),
        markers=(
            (_p(r"certificate of incorporation"), 3.0),
            (_p(r"partnership deed"), 3.0),
            (_p(r"\bsecp\b|securities (and|&) exchange commission"), 2.5),
            (_p(r"memorandum of association"), 2.5),
        ),
    ),
    FilingType(
        slug="financial_statement",
        label="Financial Statement",
        filename=(_p(r"balance[_\- ]?sheet"), _p(r"profit[_\- ]?(and|&|_)?[_\- ]?loss"), _p(r"financial")),
        markers=(
            (_p(r"balance sheet"), 3.0),
            (_p(r"profit (and|&) loss"), 3.0),
            (_p(r"statement of financial position"), 3.0),
            (_p(r"total (assets|liabilities|equity)"), 2.0),
        ),
    ),
    FilingType(
        slug="invoice",
        label="Invoice",
        filename=(_p(r"\binvoice\b"),),
        markers=(
            (_p(r"\binvoice\b"), 3.0),
            (_p(r"invoice (no|number|#)"), 2.5),
            (_p(r"\bsubtotal\b|\bamount due\b"), 2.0),
        ),
    ),
    FilingType(
        slug="receipt",
        label="Receipt",
        filename=(_p(r"\breceipt\b"),),
        markers=(
            (_p(r"\breceipt\b"), 3.0),
            (_p(r"\bpaid\b"), 1.5),
            (_p(r"\bthank you\b"), 1.0),
        ),
    ),

    # ── Everything else ──────────────────────────────────────────────────
    FilingType(
        slug="utility_bill",
        label="Utility Bill",
        filename=(_p(r"\bbill\b"), _p(r"\b(k-?electric|sui|wapda|lesco|iesco)\b")),
        markers=(
            (_p(r"\bbill month\b"), 2.5),
            (_p(r"units consumed|meter reading"), 3.0),
            (_p(r"\b(k-?electric|sui gas|wapda|lesco|iesco)\b"), 3.0),
            (_p(r"\bdue date\b"), 1.0),
        ),
    ),
    FilingType(slug="other", label="Other"),
)


# ── Refinements ───────────────────────────────────────────────────────────
#
# Never scored against a filing type — only within one, after it has been
# chosen. A refinement needs a single marker to win, because by the time it runs
# the question "is this a property document?" has already been answered and all
# that remains is "which kind?".

REFINEMENTS: tuple[Refinement, ...] = (
    # Property
    Refinement("sale_deed", "Sale Deed", "property_document",
               markers=((_p(r"\bsale deed\b"), 3.0),), filename=(_p(r"sale[_\- ]?deed"),)),
    Refinement("purchase_deed", "Purchase Deed", "property_document",
               markers=((_p(r"\bpurchase deed\b"), 3.0),), filename=(_p(r"purchase[_\- ]?deed"),)),
    Refinement("registry", "Registry", "property_document",
               markers=((_p(r"\bregistry\b"), 3.0),), filename=(_p(r"\bregistry\b"),)),
    Refinement("rent_agreement", "Rent Agreement", "property_document",
               markers=((_p(r"\brent agreement\b|\blease agreement\b|\btenancy\b"), 3.0),),
               filename=(_p(r"rent[_\- ]?agreement"), _p(r"lease"))),

    # Tax
    Refinement("income_tax_return", "Income Tax Return", "tax_return",
               markers=((_p(r"\bincome tax return\b"), 3.0),)),
    Refinement("withholding_tax_certificate", "Withholding Tax Certificate", "tax_certificate",
               markers=((_p(r"withholding tax certificate|tax deducted at source"), 3.0),),
               filename=(_p(r"withholding"),)),
    Refinement("atl_certificate", "ATL Certificate", "tax_certificate",
               markers=((_p(r"active taxpayer"), 3.0),), filename=(_p(r"\batl\b"),)),

    # Business
    Refinement("balance_sheet", "Balance Sheet", "financial_statement",
               markers=((_p(r"balance sheet|statement of financial position"), 3.0),),
               filename=(_p(r"balance[_\- ]?sheet"),)),
    Refinement("profit_and_loss", "Profit & Loss Statement", "financial_statement",
               markers=((_p(r"profit (and|&) loss"), 3.0),),
               filename=(_p(r"profit[_\- ]?(and|&|_)?[_\- ]?loss"),)),
    Refinement("company_registration", "Company Registration", "business_document",
               markers=((_p(r"certificate of incorporation|memorandum of association"), 3.0),),
               filename=(_p(r"incorporat"),)),
    Refinement("partnership_deed", "Partnership Deed", "business_document",
               markers=((_p(r"partnership deed"), 3.0),), filename=(_p(r"partnership"),)),

    # Assets
    Refinement("vehicle_registration", "Vehicle Registration", "vehicle_document",
               markers=((_p(r"registration book"), 3.0),), filename=(_p(r"registration[_\- ]?book"),)),
)


# ── Lookups ───────────────────────────────────────────────────────────────

_BY_SLUG: dict[str, FilingType] = {t.slug: t for t in FILING_TYPES}
_REFINEMENTS_BY_FILING: dict[str, list[Refinement]] = {}

for _refinement in REFINEMENTS:
    _REFINEMENTS_BY_FILING.setdefault(_refinement.files_as, []).append(_refinement)


def filing_types() -> tuple[FilingType, ...]:
    return FILING_TYPES


def filing_slugs() -> tuple[str, ...]:
    return tuple(t.slug for t in FILING_TYPES)


def get(slug: str) -> FilingType | None:
    return _BY_SLUG.get(slug)


def permits_format(slug: str | None, suffix_or_name: str) -> bool:
    """Whether this type may be filed in the format it arrived as.

    Permissive where it does not know: an unregistered type, or a file with no
    recognisable extension, is somebody else's decision. This answers one narrow
    question and refusing on ignorance would turn a missing extension into a
    lost document.
    """
    filing = get(slug or "")

    if filing is None:
        return True

    if "." not in suffix_or_name:
        # No extension at all. A name is not a format, and reading one as if it
        # were would refuse "statement" for not being a file type.
        return True

    suffix = suffix_or_name.rsplit(".", 1)[-1].strip().lower()

    if not suffix:
        return True

    return suffix in {allowed.lower().lstrip(".") for allowed in filing.allowed_file_types}


def markers() -> dict[str, tuple[tuple[Pattern[str], float], ...]]:
    """Evidence per filing type, for the rule engine.

    Types with no markers are excluded rather than offered as empty. `other` is
    a filing decision a person makes, not something a classifier recognises, and
    scoring it against nothing would let it win on an empty document.
    """
    return {t.slug: t.markers for t in FILING_TYPES if t.markers}


def refinements_for(filing_slug: str) -> tuple[Refinement, ...]:
    return tuple(_REFINEMENTS_BY_FILING.get(filing_slug, ()))


def refine(filing_slug: str, text: str) -> Refinement | None:
    """The most specific name for something already classified.

    Returns the best-scoring refinement, or None when nothing more precise than
    the filing type can be said — which is the common case and not a failure.
    """
    best: Refinement | None = None
    best_score = 0.0

    for refinement in refinements_for(filing_slug):
        score = sum(weight for pattern, weight in refinement.markers if pattern.search(text))

        if score > best_score:
            best, best_score = refinement, score

    return best


def caption_patterns() -> tuple[tuple[str, str | None, Pattern[str]], ...]:
    """How a person might name each type in a caption.

    Returns `(filing_slug, refinement_slug, pattern)`. The refinement slug is
    None for a filing type named directly.

    Every label is matched automatically, with flexible separators, so a type
    added to the registry is recognisable by name without anybody writing a
    pattern. `aliases` adds to that; it never replaces it.

    Labels are matched **longest first** by the caller. "Income Tax Return"
    contains "Tax Return", and a caption saying the former means the former —
    whichever entry happens to sit earlier in the table.
    """
    patterns: list[tuple[str, str | None, Pattern[str]]] = []

    for filing in FILING_TYPES:
        if filing.slug == "other":
            # Nobody captions a document "other", and a type nothing recognises
            # is the correct home for one nothing recognised.
            continue

        patterns.append((filing.slug, None, _label_pattern(filing.label)))
        patterns.extend((filing.slug, None, alias) for alias in filing.aliases)

    for refinement in REFINEMENTS:
        patterns.append((refinement.files_as, refinement.slug, _label_pattern(refinement.label)))
        patterns.extend(
            (refinement.files_as, refinement.slug, alias) for alias in refinement.aliases
        )

    return tuple(patterns)


def _label_pattern(label: str) -> Pattern[str]:
    """A label as somebody would type it.

    Spaces become "any separator or none", so "Salary Slip" matches
    `salary slip`, `salary_slip`, `salary-slip` and `salaryslip`.

    `&` accepts either spelling. Rewriting the label to "and" — the obvious
    first attempt — made "Profit & Loss Statement" match its own label written
    with "and" and not written with "&", which is the form the registry itself
    uses. A pattern that cannot match the text it was generated from is the kind
    of bug that only a reachability test catches.
    """
    words = [
        r"(?:&|and)" if word == "&" else re.escape(word)
        for word in label.split()
    ]

    return re.compile(r"\b" + r"[\s_\-]*".join(words) + r"\b", re.I)


def label_for(slug: str) -> str:
    """A human name for a filing type or a refinement, whichever this is."""
    if found := _BY_SLUG.get(slug):
        return found.label

    for refinement in REFINEMENTS:
        if refinement.slug == slug:
            return refinement.label

    return slug.replace("_", " ").title()
