"""Is it alive, and can it do anything useful?

Two questions, and conflating them is the classic mistake.

**Liveness** decides whether to *restart* the container. It must therefore only
consider things a restart would fix — a wedged loop, a deadlock. If liveness
checked the CMS, then a CMS taken down for ten minutes of maintenance would have
the orchestrator restart the AI over and over: throwing away in-flight work,
achieving nothing, and slowing recovery once the CMS came back.

**Readiness** decides whether to *send work*. External dependencies belong here.
A deployment whose database is unreachable is not broken, it is not ready — and
the right response is to wait, not to kill it.

The third state matters too. Missing OCR is **degraded**: documents still reach a
reviewer, unread. That must not fail readiness, or a missing optional component
would take a working service offline.
"""

from __future__ import annotations

import logging
from dataclasses import dataclass, field
from datetime import UTC, datetime
from enum import StrEnum

logger = logging.getLogger(__name__)


class Status(StrEnum):
    OK = "ok"
    DEGRADED = "degraded"
    FAILING = "failing"


@dataclass(frozen=True, slots=True)
class Component:
    name: str
    status: Status
    detail: str = ""

    @property
    def is_failing(self) -> bool:
        return self.status is Status.FAILING


@dataclass(slots=True)
class Report:
    components: list[Component] = field(default_factory=list)
    checked_at: datetime = field(default_factory=lambda: datetime.now(UTC))

    @property
    def status(self) -> Status:
        if any(c.is_failing for c in self.components):
            return Status.FAILING

        if any(c.status is Status.DEGRADED for c in self.components):
            return Status.DEGRADED

        return Status.OK

    @property
    def is_ready(self) -> bool:
        """Degraded is ready. A missing optional component must not take a
        working service out of rotation."""
        return not any(c.is_failing for c in self.components)

    def to_dict(self) -> dict:
        """The response body.

        Component names and coarse states only — never a hostname, a DSN or the
        text of a driver error. This endpoint is the most likely thing on the box
        to be exposed by accident, and "connection to 10.0.0.4:5432 failed" is a
        map of the infrastructure for whoever reads it.
        """
        from app.release.version import VERSION

        return {
            "status": self.status.value,
            # Which release is answering. The first question asked when a
            # deployment misbehaves is "what is actually running on it", and
            # without this the answer involves logging into the box.
            "version": VERSION,
            "checked_at": self.checked_at.isoformat(),
            "components": [
                {"name": c.name, "status": c.status.value, "detail": c.detail}
                for c in self.components
            ],
        }


def liveness(daemon) -> Report:
    """Only what a restart would fix."""
    ticking = daemon is None or daemon.is_ticking()

    return Report([
        Component(
            "loop",
            Status.OK if ticking else Status.FAILING,
            _loop_detail(daemon) if ticking else "the daemon loop has stalled",
        )
    ])


def _loop_detail(daemon) -> str:
    """What the loop is doing, not merely that it is alive.

    A probe that answers "running" while a document has been in OCR for two
    minutes tells an operator nothing they can act on. Naming the job and its
    age turns the one moment this endpoint is ever read — something looks wrong
    — into an answer rather than another question.
    """
    since = getattr(daemon, "running_since", None)

    if daemon is None or since is None:
        return "running"

    seconds = (datetime.now(UTC) - since).total_seconds()
    job = getattr(daemon, "running_job", None) or "a job"

    return f"running — busy in '{job}' for {seconds:.0f}s"


def readiness(container, daemon=None) -> Report:
    """Everything needed to do useful work."""
    components: list[Component] = [_configuration(container)]

    if daemon is not None:
        components.append(liveness(daemon).components[0])

    components.extend([
        _database(container),
        _cms(container),
        _compatibility(container),
        _ocr(container),
        _memory(container),
        _whatsapp(container),
        _intake_outbox(container),
    ])

    return Report(components)


# ── Individual checks ─────────────────────────────────────────────────────
#
# Each returns rather than raises. One component being unreachable must not stop
# the others being reported — an operator reading this wants the whole picture,
# not the first thing that went wrong.


def _configuration(container) -> Component:
    try:
        container.settings

        return Component("configuration", Status.OK, "loaded")
    except Exception:  # noqa: BLE001
        # Unreachable in practice: the process refuses to start without it.
        return Component("configuration", Status.FAILING, "incomplete")


def _database(container) -> Component:
    if container.connect is None:
        return Component("database", Status.FAILING, "not configured")

    try:
        with container.connect() as connection, connection.cursor() as cursor:
            cursor.execute("SELECT 1")
            cursor.fetchone()

        return Component("database", Status.OK, "reachable")
    except Exception:  # noqa: BLE001
        # Deliberately not the driver's message, which carries host and port.
        return Component("database", Status.FAILING, "unreachable")


def _cms(container) -> Component:
    try:
        # The handshake rather than a bare whoami. It costs the same call and
        # refreshes the compatibility verdict, so a CMS that updates itself
        # under a running process is noticed by the next readiness probe rather
        # than at the next restart.
        container.cms.handshake()
        identity = container.cms.identity or {}
    except Exception:  # noqa: BLE001
        return Component("cms", Status.FAILING, "unreachable or rejected")

    granted = set(identity.get("permissions") or [])
    missing = {"clients.read", "proposals.submit"} - granted

    if missing:
        # Reachable but useless: every document read would have nowhere to go.
        return Component("cms", Status.FAILING, f"missing {', '.join(sorted(missing))}")

    return Component("cms", Status.OK, "authenticated")


def _compatibility(container) -> Component:
    """Whether this release and the CMS still agree on the contract.

    FAILING when they do not, which takes the deployment out of rotation. That
    is the right answer: an incompatible agent cannot do useful work, and the
    gate in CmsClient is already refusing its requests. Reporting it as merely
    degraded would leave an orchestrator happily routing to a service that
    cannot file anything.
    """
    from app.api.compatibility import Level

    verdict = getattr(container.cms, "compatibility", None)

    if verdict is None:
        return Component("compatibility", Status.DEGRADED, "not yet checked")

    if verdict.level is Level.INCOMPATIBLE:
        return Component("compatibility", Status.FAILING, verdict.detail)

    status = Status.DEGRADED if verdict.level is Level.DEGRADED else Status.OK

    return Component("compatibility", status, verdict.detail)


def _ocr(container) -> Component:
    name = getattr(container.ocr, "name", "unknown")

    if name == "null":
        # Degraded, never failing. Documents still reach a reviewer, unread —
        # failing readiness here would take a working service offline over a
        # component that only makes it better.
        return Component("ocr", Status.DEGRADED, "no engine; documents reach a reviewer unread")

    return Component("ocr", Status.OK, name)


def _memory(container) -> Component:
    try:
        container.memory.recall("health", limit=1)

        return Component("memory", Status.OK, "responding")
    except Exception:  # noqa: BLE001
        # Degraded: memory improves later runs and nothing depends on it to file
        # a document correctly.
        return Component("memory", Status.DEGRADED, "not responding")


def _intake_outbox(container) -> Component:
    """Documents that arrived and the CMS has not been told about (ADR-0010).

    Degraded rather than OK the moment this is non-empty, and deliberately so.
    Everything counted here is a document somebody sent which appears nowhere in
    the CMS — the firm cannot see their own post, and the only reason that is
    tolerable at all is that it is loudly visible while it lasts.

    Failing outright at capacity, because at that point the agent has stopped
    accepting documents altogether.
    """
    try:
        held = len(container.intake_outbox)
        capacity = container.intake_outbox.capacity
    except Exception:  # noqa: BLE001
        return Component("intake outbox", Status.DEGRADED, "cannot be read")

    if held == 0:
        return Component("intake outbox", Status.OK, "empty")

    if held >= capacity:
        return Component(
            "intake outbox",
            Status.FAILING,
            f"full at {held}; documents are being refused until the CMS returns",
        )

    return Component(
        "intake outbox",
        Status.DEGRADED,
        f"{held} document(s) arrived but the CMS has not been told",
    )


def _whatsapp(container) -> Component:
    """Whether the inbox knows whose account it is attached to.

    Degraded rather than failing when it does not: the deployment still polls
    for decisions and finishes workflows already in flight. It simply accepts
    nothing new, which is the only safe answer while nobody can say whose
    messages these would be.

    No longer a configuration question. The answer is either "an account is
    linked" or "link one on the settings page" — there is nothing to fill in.
    """
    inbox = container.inbox

    # Resolved, not read: a readiness probe asking "is an account linked" must
    # go and find out, rather than report the answer a message would have left
    # behind had one arrived.
    policy = inbox.current_policy() if hasattr(inbox, "current_policy") else inbox.policy

    if not policy.is_configured:
        return Component(
            "whatsapp",
            Status.DEGRADED,
            "no WhatsApp account linked; every message is ignored",
        )

    # An account being linked is not the same as messages being able to flow,
    # and the difference is not academic: on 2026-08-03 this check said "ok"
    # through three separate outages — Evolution's process died twice and its
    # stream died once — because all it ever looked at was the policy. The
    # transport is this deployment's only way to receive a document, so a
    # transport that cannot deliver is FAILING, not a nuance.
    #
    # Providers that cannot report a session (Meta, test stubs) keep the old
    # judgement: "we cannot tell" must not condemn a working deployment.
    provider = getattr(container, "whatsapp", None)

    if provider is not None and callable(getattr(provider, "session_status", None)):
        from app.whatsapp.session import SessionState, status_of

        session = status_of(provider)

        if session.state is not SessionState.CONNECTED:
            return Component(
                "whatsapp",
                Status.FAILING,
                f"cannot receive documents — {session.state.value}"
                + (f": {session.detail}" if session.detail else ""),
            )

        return Component("whatsapp", Status.OK, f"connected; reading {policy.describe()}")

    return Component("whatsapp", Status.OK, f"reading {policy.describe()}")
