"""Turning remembered text into vectors.

**The embedder is a port for a data-protection reason, not just a design one.**
Embedding sends text to a model. If that model is hosted, the text has left the
installation — so an embedder is an egress path, and ADR-0002 applies to it
exactly as it applies to a chat completion. A hosted embedder may only ever see
Level 1 or redacted Level 2 content.

The default below is local and deterministic, so a deployment embeds nothing to
anyone until someone deliberately configures otherwise.
"""

from __future__ import annotations

import hashlib
import math
import re
from typing import Protocol, runtime_checkable

DIMENSIONS = 384
"""Matches the usual small sentence-transformer size, so swapping in a real
local model later does not require a schema migration."""


@runtime_checkable
class Embedder(Protocol):
    """Anything that can turn text into a vector."""

    name: str
    dimensions: int

    def embed(self, text: str) -> list[float]: ...


class HashingEmbedder:
    """A local, dependency-free, deterministic embedder.

    **This is a lexical embedder, not a semantic one.** It hashes token n-grams
    into a fixed vector, so "invoice unpaid" and "unpaid invoice" land close
    together, while "bill outstanding" does not — a real sentence-transformer
    would place all three together.

    It is here because it makes the whole memory pipeline real and testable with
    zero dependencies and zero data egress, and because a deployment with no
    model configured should still be able to store and retrieve. Swap in a local
    sentence-transformer for genuine semantic recall; the port makes that a
    configuration change rather than a rewrite.

    Overstating this as semantic search would be the sort of quiet
    misrepresentation that costs someone a document later.
    """

    name = "hashing-local"
    dimensions = DIMENSIONS

    _TOKEN = re.compile(r"[a-z0-9]+")

    def embed(self, text: str) -> list[float]:
        vector = [0.0] * self.dimensions
        tokens = self._TOKEN.findall(text.lower())

        if not tokens:
            return vector

        # Unigrams and bigrams: bigrams give a little word-order sensitivity,
        # which is what stops "paid" and "not paid" colliding entirely.
        features = tokens + [f"{a}_{b}" for a, b in zip(tokens, tokens[1:], strict=False)]

        for feature in features:
            digest = hashlib.blake2b(feature.encode("utf-8"), digest_size=8).digest()
            index = int.from_bytes(digest[:4], "big") % self.dimensions
            # Signed, so unrelated features cancel instead of all pushing the
            # vector in one direction and making everything look similar.
            sign = 1.0 if digest[4] % 2 == 0 else -1.0
            vector[index] += sign

        return _normalise(vector)


def _normalise(vector: list[float]) -> list[float]:
    magnitude = math.sqrt(sum(v * v for v in vector))

    if magnitude == 0.0:
        return vector

    return [v / magnitude for v in vector]


def cosine_similarity(a: list[float], b: list[float]) -> float:
    """Similarity of two vectors, in [-1, 1].

    Both vectors are normalised on creation, so this is a dot product — but it
    is computed defensively anyway, because a vector arriving from storage may
    have been written by a different embedder.
    """
    if len(a) != len(b):
        raise ValueError(f"Vectors differ in length: {len(a)} vs {len(b)}.")

    dot = sum(x * y for x, y in zip(a, b, strict=True))
    magnitude = math.sqrt(sum(x * x for x in a)) * math.sqrt(sum(y * y for y in b))

    if magnitude == 0.0:
        return 0.0

    return dot / magnitude
