"""Work the CMS has asked this deployment to do.

THE DIRECTION, AND WHY IT IS THIS WAY ROUND

A customer presses "Connect WhatsApp" in their own dashboard. The thing that has
to act is this process — which lives on other infrastructure and, by ADR-0005,
is the client in this relationship. The CMS never calls out.

So the CMS writes down what it wants and this collects it, exactly as it already
collects proposal decisions. Nothing new is exposed: no inbound port here, no
credential for us stored there, no second direction to secure.

WHAT A RESTART COSTS

Nothing. Every command's state lives in the CMS, so a process killed mid-flight
leaves a claim behind rather than losing work; the CMS returns it to the queue
after a few minutes and this picks it up again on a later poll. There is no
local queue to lose, which is the entire reason the queue is not local.

RETRIES ARE THE CMS'S DECISION, NOT OURS

This reports what happened and whether it looked worth trying again. The backoff,
the attempt count and the giving-up all live on the CMS side, where they are
durable and auditable. A retry loop here would be invisible to the customer
watching the screen.
"""

from __future__ import annotations

import logging
from collections.abc import Callable
from dataclasses import dataclass, field
from typing import Any

from app.api.client import AgentApiError

logger = logging.getLogger(__name__)

#: A handler: given the command's payload, return what to report back.
Handler = Callable[[dict[str, Any]], "CommandResult"]


@dataclass(frozen=True, slots=True)
class CommandResult:
    """What to tell the CMS about a command."""

    ok: bool
    data: dict[str, Any] = field(default_factory=dict)
    error: str = ""

    retryable: bool = True
    """Whether trying again could plausibly work.

    Only this side knows. "Evolution was unreachable" is worth another attempt;
    "this provider cannot be linked by QR" never will be, and retrying it three
    times just delays telling the customer something true.
    """

    @classmethod
    def succeeded(cls, **data: Any) -> CommandResult:
        return cls(ok=True, data=data)

    @classmethod
    def failed(cls, error: str, retryable: bool = True) -> CommandResult:
        return cls(ok=False, error=error, retryable=retryable)


@dataclass(frozen=True, slots=True)
class Command:
    """One instruction, as the CMS handed it over."""

    id: str
    type: str
    payload: dict[str, Any] = field(default_factory=dict)
    attempt: int = 1
    max_attempts: int = 3

    @classmethod
    def from_response(cls, data: dict[str, Any]) -> Command | None:
        identifier = str(data.get("id") or "")
        kind = str(data.get("type") or "")

        if not identifier or not kind:
            return None

        return cls(
            id=identifier,
            type=kind,
            payload=data.get("payload") or {},
            attempt=int(data.get("attempt") or 1),
            max_attempts=int(data.get("max_attempts") or 3),
        )


class CommandProcessor:
    """Claims one command per pass, runs it, reports the outcome."""

    def __init__(self, cms, handlers: dict[str, Handler] | None = None) -> None:
        self._cms = cms
        self._handlers: dict[str, Handler] = dict(handlers or {})

    def register(self, command_type: str, handler: Handler) -> None:
        self._handlers[command_type] = handler

    def poll(self, limit: int = 5) -> list[str]:
        """Work through what is waiting. Returns the ids handled.

        Bounded per pass so one busy installation cannot monopolise the loop
        that also polls decisions and reports health.
        """
        handled: list[str] = []

        for _ in range(max(1, limit)):
            try:
                command = self._claim()
            except AgentApiError as exc:
                # An unreachable CMS is not a command failure — there is nothing
                # to report a failure *about*. The next pass tries again.
                logger.warning("Could not claim a command: %s", exc)
                break

            if command is None:
                break

            self._run(command)
            handled.append(command.id)

        return handled

    # ── Internals ─────────────────────────────────────────────────────────

    def _claim(self) -> Command | None:
        response = self._cms.claim_command()

        if not response:
            return None

        return Command.from_response(response)

    def _run(self, command: Command) -> None:
        handler = self._handlers.get(command.type)

        if handler is None:
            # A CMS newer than this release. Not retryable: the next attempt
            # would find the same missing handler, and three identical failures
            # tell the customer nothing the first did not.
            logger.error("No handler for command type '%s'.", command.type)
            self._report(command, CommandResult.failed(
                f"this release does not handle '{command.type}'", retryable=False
            ))

            return

        try:
            result = handler(command.payload)
        except Exception as exc:  # noqa: BLE001 - a handler must not kill the loop
            logger.exception("Command %s (%s) raised: %s", command.id, command.type, exc)
            self._report(command, CommandResult.failed(str(exc)))

            return

        self._report(command, result)

    def _report(self, command: Command, result: CommandResult) -> None:
        try:
            self._cms.report_command(
                command.id,
                status="completed" if result.ok else "failed",
                result=result.data,
                error=result.error,
                retryable=result.retryable,
            )
        except AgentApiError as exc:
            # The work happened; the CMS did not hear. Its stale-claim sweep
            # returns the command to the queue and it runs again — which is why
            # handlers have to be safe to repeat.
            logger.warning("Could not report command %s: %s", command.id, exc)
