"""Asking a model what a document is — on this host, or not at all.

The last classification stage, and the only one that could break ADR-0002. The
rule it runs under is ADR-0009: a model may read a document's text when, and
only when, it runs on this machine. Level 3 data does not leave the
installation; a model on the installation is not "leaving".

That distinction is one config edit away from being erased, so it is enforced
here rather than documented elsewhere.

## What makes an endpoint acceptable

Loopback, or a private address. Everything else is refused — public addresses,
public DNS names, and anything that will not resolve.

Checked at configuration time *and* again on every call. A hostname that
resolved to loopback at boot and somewhere else an hour later is refused on the
call; trusting the earlier answer is exactly how a long-running process ends up
posting a client's bank statement to somebody's endpoint.

## What is sent

The first page's text. Nothing else — not the file, not the filename, not a
CNIC extracted from anywhere, not the client the document might belong to. The
model is asked one question and given the minimum needed to answer it.

## What is trusted

A type this installation files under, and nothing else. A model naming something
outside the registry abstains rather than guessing, which leaves whatever the
earlier stages decided standing. Its confidence is capped below what a caption
is worth, so a model never overrules a person who said what the document is.
"""

from __future__ import annotations

import ipaddress
import json
import logging
import socket
import urllib.error
import urllib.request
from dataclasses import dataclass
from typing import Protocol, runtime_checkable
from urllib.parse import urlparse

from app.documents import registry

logger = logging.getLogger(__name__)

#: The most a model's answer may be worth.
#:
#: Below CAPTION_CONFIDENCE deliberately. A person who wrote "Salary Slip" on the
#: message is better evidence than a model reading a page, and the ordering must
#: hold even if a model returns 1.0 for everything — which some do.
MAX_CONFIDENCE = 0.85

#: How much text to send.
#:
#: A page is plenty to recognise a type and the tail of a long document adds
#: nothing but latency. Also a soft guard: whatever else changes, this stage
#: cannot ship an entire document's contents anywhere in one call.
MAX_CHARACTERS = 4000


@dataclass(frozen=True, slots=True)
class ModelVerdict:
    """What a model said, once it has been believed."""

    document_type: str
    confidence: float
    reason: str


@runtime_checkable
class ModelClassifier(Protocol):
    """Anything that can name a document from its text."""

    def classify(self, text: str) -> ModelVerdict | None: ...


# ── The locality rule (ADR-0009's enforcement) ────────────────────────────


def is_local_endpoint(url: str) -> bool:
    """Whether this address is somewhere Level 3 data may go.

    Loopback and private ranges pass; everything else is refused, including
    anything that will not resolve. Refusing the unresolvable is the important
    half — a name that cannot be checked is not a name that can be trusted, and
    "it was probably fine" is not a basis for sending somebody's bank statement.

    A private address on the customer's own network is permitted and is not
    necessarily this host. ADR-0009 accepts that: a URL cannot express "this
    machine", and refusing private addresses outright would rule out a model on
    a second box in the same rack, which is a legitimate deployment. What is
    ruled out is anything reachable from the public internet.
    """
    try:
        parsed = urlparse(url)
    except ValueError:
        return False

    host = parsed.hostname

    if not host:
        return False

    try:
        # Every address the name resolves to, not just the first. A name that
        # answers with one loopback address and one public one is not local.
        resolved = socket.getaddrinfo(host, None)
    except (socket.gaierror, UnicodeError):
        return False

    addresses = {info[4][0] for info in resolved}

    if not addresses:
        return False

    for address in addresses:
        try:
            ip = ipaddress.ip_address(address)
        except ValueError:
            return False

        if not (ip.is_loopback or ip.is_private):
            return False

    return True


# ── A model spoken to over HTTP ───────────────────────────────────────────


class LocalHttpModel:
    """An OpenAI-compatible chat endpoint on this host.

    That shape rather than any one product's: Ollama, llama.cpp's server and LM
    Studio all speak it, so a deployment picks its runtime without this module
    knowing which.

    Every failure returns None. A model that is slow, down, or answering
    nonsense must leave the document exactly where the earlier stages left it —
    this stage exists to recognise a residue, and turning that residue into
    failed runs would be worse than not having it.
    """

    def __init__(self, url: str, model: str, timeout: float = 20.0) -> None:
        self._url = url.rstrip("/")
        self._model = model
        self._timeout = timeout

    @property
    def url(self) -> str:
        return self._url

    def classify(self, text: str) -> ModelVerdict | None:
        if not text.strip():
            return None

        # Re-checked on every call, not trusted from construction. See the
        # module docstring: a name's answer can change under a long-running
        # process, and this is the only thing standing between a client's
        # documents and somebody else's endpoint.
        if not is_local_endpoint(self._url):
            logger.error(
                "Refusing to send document text to %s: it is not a local endpoint "
                "(ADR-0009). The model stage is doing nothing.",
                _describe(self._url),
            )

            return None

        try:
            payload = json.dumps(
                {
                    "model": self._model,
                    "temperature": 0,
                    "messages": [
                        {"role": "system", "content": _SYSTEM_PROMPT},
                        {"role": "user", "content": text[:MAX_CHARACTERS]},
                    ],
                }
            ).encode("utf-8")

            request = urllib.request.Request(
                f"{self._url}/v1/chat/completions",
                data=payload,
                headers={"Content-Type": "application/json"},
                method="POST",
            )

            with urllib.request.urlopen(request, timeout=self._timeout) as response:
                body = json.loads(response.read().decode("utf-8"))
        except (urllib.error.URLError, TimeoutError, ValueError, OSError) as exc:
            logger.warning("The local model did not answer: %s", exc)

            return None

        return _believe(body)


def _describe(url: str) -> str:
    """A URL for a log line, without its credentials.

    A model endpoint should not carry any — but "should not" is how secrets get
    logged, and this line is written precisely when somebody has configured
    something unexpected.
    """
    parsed = urlparse(url)

    return f"{parsed.scheme}://{parsed.hostname or '?'}"


_SYSTEM_PROMPT = (
    "You identify Pakistani tax-practice documents. Reply with JSON only: "
    '{"document_type": "<slug>", "confidence": <0-1>, "reason": "<short>"}. '
    "The slug must be one of: " + ", ".join(registry.filing_slugs()) + ". "
    'If none fits, reply {"document_type": "unknown", "confidence": 0, '
    '"reason": "..."}. Never invent a slug.'
)


def _believe(body: dict) -> ModelVerdict | None:
    """Turn a model's reply into a verdict, or into nothing.

    Everything is checked. A model is the one component here that can return
    confident, well-formed, entirely invented output, and the cost of believing
    it is a document filed under a type nobody chose.
    """
    try:
        content = body["choices"][0]["message"]["content"]
    except (KeyError, IndexError, TypeError):
        return None

    answer = _parse_json(content)

    if answer is None:
        return None

    document_type = str(answer.get("document_type") or "").strip().lower()

    # The registry is the vocabulary. A slug it does not hold is refused rather
    # than passed on — the CMS would reject the filing anyway, and a reviewer
    # would be looking at a proposal nobody could approve.
    if document_type in ("", "unknown") or registry.get(document_type) is None:
        return None

    try:
        confidence = float(answer.get("confidence", 0))
    except (TypeError, ValueError):
        confidence = 0.0

    # Capped, not trusted. Some local models answer 1.0 to everything.
    confidence = max(0.0, min(confidence, MAX_CONFIDENCE))

    reason = str(answer.get("reason") or "").strip()[:200]

    return ModelVerdict(
        document_type=document_type,
        confidence=confidence,
        reason=reason or "A local model recognised it.",
    )


def _parse_json(content: str) -> dict | None:
    """The JSON in a reply that may not be only JSON.

    Small models wrap answers in prose or fences however firmly they are asked
    not to. Finding the object is worth doing; guessing at prose is not, so
    anything that will not parse is refused.
    """
    text = (content or "").strip()

    if text.startswith("```"):
        text = text.strip("`")
        text = text.split("\n", 1)[-1] if "\n" in text else text

    start, end = text.find("{"), text.rfind("}")

    if start == -1 or end <= start:
        return None

    try:
        parsed = json.loads(text[start : end + 1])
    except ValueError:
        return None

    return parsed if isinstance(parsed, dict) else None


def build(url: str | None, model: str | None, timeout: float = 20.0) -> LocalHttpModel | None:
    """The model this deployment will use, or None.

    None is the ordinary answer: the stage is off unless a URL is configured,
    so a deployment that sets nothing sends nothing.

    A configured but non-local URL returns None too, loudly. Raising instead
    would stop a deployment from starting over a stage it can run perfectly well
    without — and the document intake it would take down is the part that
    matters.
    """
    if not url:
        return None

    if not model:
        logger.warning("A model URL is set but no model name; the model stage is off.")

        return None

    if not is_local_endpoint(url):
        logger.error(
            "TAXPILOT_MODEL_URL points at %s, which is not a local endpoint. "
            "The model stage is disabled — document text may not leave this "
            "installation (ADR-0002, ADR-0009).",
            _describe(url),
        )

        return None

    logger.info("Local model classification is on, using %s.", _describe(url))

    return LocalHttpModel(url, model, timeout)
