"""Regressions for the Phase 9 security audit.

Each test here corresponds to a weakness that was present in the code and has
been fixed. They exist so the fix cannot quietly come undone — a masked repr is
one careless `@dataclass` regeneration away from leaking again, and a size limit
is one refactor away from being dropped.

Findings and their fixes:

  S-01  Secrets appeared in dataclass reprs, so any traceback or log line
        carrying a Settings or MetaConfig object wrote a working credential to
        disk.                                                    → masked reprs
  S-02  The HTTP server read Content-Length bytes with no bound, so a claimed
        ten-gigabyte body exhausted memory on an internet-facing endpoint.
                                                        → 413 before any read
  S-03  OCR had no input size limit, and `max_side` was configured but never
        applied — dead config promising a protection that did not exist.
                                    → file-size bound, and max_side wired through
  S-05  And wiring max_side through made things worse: PaddleOCR defaults
        text_det_limit_type to "min", which scales the SHORTEST side *up*, so a
        side length alone enlarges images. 96.9s vs 19.3s, identical output.
                                              → text_det_limit_type set to "max"
"""

from __future__ import annotations

import urllib.error
import urllib.request

import pytest

from app.config.settings import Settings
from app.ocr.config import OcrConfig
from app.ocr.paddle import PaddleOcrEngine
from app.runtime.server import MAX_BODY_BYTES, HttpServer, Request
from app.whatsapp.meta import MetaConfig

SECRET = "sk-live-THIS-IS-THE-SECRET-VALUE"


class TestS01SecretsNeverAppearInARepr:
    """A dataclass repr is how credentials reach a log file.

    `logger.exception("failed: %s", settings)`, a traceback that prints locals,
    or a crash reporter all render these — and logs are shipped, aggregated and
    kept far longer than anyone intends.
    """

    def test_settings_masks_its_secret(self):
        rendered = repr(Settings("https://cms.test", "tpa_key_value", SECRET))

        assert SECRET not in rendered
        assert "cms.test" in rendered      # still useful for diagnosis

    def test_settings_masks_the_api_key(self):
        assert "tpa_key_value" not in repr(Settings("https://cms.test", "tpa_key_value", SECRET))

    def test_meta_masks_its_access_token(self):
        rendered = repr(MetaConfig("PN1", SECRET, app_secret=SECRET))

        assert SECRET not in rendered
        assert "PN1" in rendered           # not a secret, and identifies the number

    def test_a_masked_value_still_distinguishes_two_credentials(self):
        first = repr(Settings("https://cms.test", "k", "secret-value-AAAA"))
        second = repr(Settings("https://cms.test", "k", "secret-value-BBBB"))

        # Enough to tell "the wrong key is configured" from "the key is right";
        # not enough to use either.
        assert first != second

    def test_a_short_secret_is_not_revealed_by_the_mask(self):
        # Showing the last four of an eight-character secret gives away half of
        # it, so short values are masked entirely.
        assert "abc" not in repr(Settings("https://cms.test", "k", "abcd"))

    def test_str_is_masked_too(self):
        # An f-string calls __str__ rather than __repr__, and dataclasses derive
        # one from the other — but only while nobody overrides just one of them.
        assert SECRET not in f"{Settings('https://cms.test', 'k', SECRET)}"

    def test_an_exception_carrying_settings_does_not_leak(self):
        settings = Settings("https://cms.test", "k", SECRET)

        try:
            raise RuntimeError(f"could not start with {settings!r}")
        except RuntimeError as exc:
            assert SECRET not in str(exc)


class TestS02RequestBodiesAreBounded:
    """`rfile.read(n)` allocates n. An attacker need not send what they claim."""

    @pytest.fixture
    def server(self):
        server = HttpServer("127.0.0.1", 0)
        server.route("POST", "/echo", lambda request: (200, {"length": len(request.body)}))
        server.start()

        yield server

        server.stop()

    def test_an_oversized_body_is_refused(self, server):
        oversized = b"x" * (MAX_BODY_BYTES + 1)

        status, _ = _post(server, "/echo", oversized)

        assert status == 413

    def test_a_lying_content_length_is_refused_before_reading(self, server):
        # The attack that costs nothing to mount: claim ten gigabytes, send
        # almost nothing, and let the server allocate.
        #
        # Sent over a raw socket because urllib recomputes Content-Length from
        # the body — so a header set through it does not survive, and the test
        # would pass or fail depending on which value won.
        assert _raw_status(server, content_length=str(10 * 1024 * 1024 * 1024)) == 413

    def test_a_malformed_content_length_is_refused(self, server):
        assert _raw_status(server, content_length="not-a-number") == 400

    def test_an_ordinary_body_still_works(self, server):
        status, body = _post(server, "/echo", b'{"entry":[]}')

        # A real WhatsApp webhook carries a media *id*, not the media, so a
        # legitimate body is kilobytes.
        assert status == 200
        assert body["length"] == 12

    def test_the_limit_is_generous_for_real_traffic(self):
        """A megabyte was not generous, and this test said it was.

        The floor was 1 MB, resting on the same wrong assumption the constant
        carried: that a webhook holds a media id rather than the media. With
        `webhookBase64: true` the image is in the body, so an ordinary phone
        photograph — three to four megabytes before base64 adds a third — was
        refused, and this assertion passed throughout.

        Floored against a real document instead. The CMS accepts 20 MB; base64
        carries that past 26 MB, and anything lower refuses documents the other
        half of the system would have taken.
        """
        assert MAX_BODY_BYTES >= 27 * 1024 * 1024

    def test_a_phone_photograph_of_a_cnic_is_accepted(self):
        """The exact case that failed in production on 2026-08-04.

        4,415,811 bytes — the body Evolution re-posted every few minutes for 45
        minutes while the document sat unprocessed. A literal, because it is
        evidence rather than an estimate.
        """
        assert MAX_BODY_BYTES > 4_415_811


class TestS03OcrInputIsBounded:
    """A compressed image expands to width × height × channels once decoded."""

    def test_an_oversized_document_is_refused_without_decoding(self, tmp_path):
        engine = PaddleOcrEngine(OcrConfig(max_file_bytes=1024))

        class Exploding:
            def predict(self, path):
                raise AssertionError("the file should never have reached the decoder")

        engine._reader = Exploding()  # noqa: SLF001

        document = tmp_path / "bomb.png"
        document.write_bytes(b"x" * 2048)

        # Refused on the stat, before anything reads it.
        assert engine.read(document).is_empty

    def test_a_normal_document_is_still_read(self, tmp_path):
        from app.ocr.engine import OcrResult, TextBlock

        engine = PaddleOcrEngine(OcrConfig())

        class Reader:
            def predict(self, path):
                return [{"rec_texts": ["hello"], "rec_scores": [0.9],
                         "rec_polys": [[[0, 0], [10, 0], [10, 10], [0, 10]]]}]

        engine._reader = Reader()  # noqa: SLF001

        document = tmp_path / "ok.png"
        document.write_bytes(b"x" * 100)

        assert engine.read(document).text == "hello"

    def test_the_limit_is_configurable(self):
        config = OcrConfig.from_env({"TAXPILOT_OCR_MAX_FILE_BYTES": "1234"})

        assert config.max_file_bytes == 1234

    def test_max_side_is_actually_passed_to_the_detector(self):
        """It was configured and never used until this audit.

        Dead config is worse than no config: it documents a bound that does not
        exist, and anyone reading it concludes the protection is in place.
        """
        import inspect

        source = inspect.getsource(PaddleOcrEngine._build_reader)

        assert "text_det_limit_side_len" in source
        assert "self._config.max_side" in source

        # And as a CAP. PaddleOCR defaults limit_type to "min", which scales the
        # shortest side *up* — so the side length alone makes images larger.
        # Measured on one document: 96.9s with "min", 19.3s with "max", identical
        # output. Without this line the setting does the opposite of its purpose.
        assert '"text_det_limit_type": "max"' in source


def _raw_status(server: HttpServer, content_length: str, body: bytes = b"tiny") -> int:
    """Send a request with an exact Content-Length, whatever the body really is.

    A raw socket, because every HTTP client recomputes that header — which is
    precisely the value under test here. Through urllib the test passed or
    failed depending on whose value won.
    """
    import socket

    head = "\r\n".join([
        "POST /echo HTTP/1.1",
        "Host: 127.0.0.1",
        "Content-Type: application/json",
        f"Content-Length: {content_length}",
        "Connection: close",
        "",
        "",
    ])

    with socket.create_connection(("127.0.0.1", server.port), timeout=10) as sock:
        sock.sendall(head.encode() + body)

        response = b""
        while b"\r\n" not in response:
            chunk = sock.recv(4096)

            if not chunk:
                break

            response += chunk

    return int(response.split()[1]) if response else 0


def _post(server: HttpServer, path: str, body: bytes, content_length: str | None = None):
    import json

    request = urllib.request.Request(
        f"http://127.0.0.1:{server.port}{path}", data=body, method="POST"
    )

    if content_length is not None:
        # Overriding what urllib would compute — which is the whole point of the
        # lying-length case.
        request.add_header("Content-Length", content_length)

    try:
        with urllib.request.urlopen(request, timeout=5) as response:
            return response.status, json.loads(response.read())
    except urllib.error.HTTPError as exc:
        return exc.code, json.loads(exc.read())
