"""Everything known about a document before anyone decides what it is.

One object, assembled by whichever intake received the document, so every stage
below reads the same shape and no stage learns where it came from. That is what
makes WhatsApp, email, a manual upload and an API upload the same pipeline
rather than four of them.

## Text is lazy, and that is the point

`text` is a callable, not a string. Reading a document is by far the most
expensive thing this system does — measured at roughly eighteen seconds warm for
one page — and the stages before it exist precisely so it usually does not
happen. A caption that says "bank statement" and a filename that says
`Meezan_Statement_July.pdf` are both free.

So OCR is a function the pipeline may or may not call, and whether it did is
recorded rather than inferred. A stage that wants text asks for it; one that
does not, does not pay for it.
"""

from __future__ import annotations

from collections.abc import Callable
from dataclasses import dataclass, field
from pathlib import Path


@dataclass(slots=True)
class Signals:
    """What arrived, and what can be learned about it without deciding anything."""

    path: Path
    """Where the file is on this machine. It stays there — Level 3 documents do
    not leave the installation (ADR-0002)."""

    source: str = "whatsapp"
    """Which intake received it. Recorded for the audit trail and for metrics;
    no stage branches on it, because a document is what it is regardless of how
    it arrived."""

    caption: str = ""
    """What the sender typed alongside it. The strongest signal there is when it
    carries one, because a person looked at the document and said what it was."""

    filename: str = ""
    """As the sender's device named it. Never used as a path — only as evidence
    — because it is attacker-controlled in every intake this has."""

    mime_type: str = ""
    size_bytes: int = 0

    read_first_page: Callable[[], str] | None = None
    """Reads page one and returns its text. None when no reader is available, in
    which case the OCR stage abstains rather than failing — a deployment without
    OCR still classifies from a caption and a filename."""

    model: object | None = None
    """A model that may name a document from its text, or None.

    None on almost every deployment, which is the point: the model stage is off
    unless one is configured, and it is refused unless it runs on this host
    (ADR-0009). Typed loosely so this module does not import the model package
    — signals are what is known about a document, not how it might be reasoned
    over."""

    _text: str | None = field(default=None, repr=False)
    _read_attempted: bool = field(default=False, repr=False)

    def text(self) -> str:
        """The first page's text, reading it once if it has not been read.

        Cached including the empty result. A document that read as nothing reads
        as nothing the second time too, and paying eighteen seconds again to
        confirm that is the kind of cost that only shows up under load.
        """
        if self._read_attempted:
            return self._text or ""

        self._read_attempted = True

        if self.read_first_page is None:
            return ""

        try:
            self._text = self.read_first_page()
        except Exception:  # noqa: BLE001 - a failed read is an answer, not a crash
            # Recorded as attempted-and-empty. The distinction between "could
            # not read" and "was never read" is one the CMS scores on, and it
            # is carried in the classification rather than guessed at later.
            self._text = ""

        return self._text or ""

    @property
    def was_read(self) -> bool:
        """Whether anything actually opened the document.

        The honest input to the CMS's risk scoring: a document nobody tried to
        read is not the same as one that could not be read, and treating them
        alike marks every cheap classification high risk.
        """
        return self._read_attempted and self.read_first_page is not None

    @property
    def suffix(self) -> str:
        return Path(self.filename or self.path.name).suffix.lower()
