"""Reading a PDF's own text instead of photographing it.

The nine documents this exists for are real client bank statements and salary
slips — Level 3 under ADR-0002, so none of them are here. The PDFs below are
generated, and they reproduce the two shapes that matter: an export carrying a
text layer, and a scan carrying none.
"""

from __future__ import annotations

from pathlib import Path

import pytest

from app.ocr.engine import OcrResult, TextBlock
from app.ocr.text_layer import (
    MINIMUM_CHARACTERS,
    TextLayerFirst,
    extract,
    is_usable,
)


class RecordingEngine:
    """Stands in for PaddleOCR, and remembers whether it was asked to work."""

    name = "recording"

    def __init__(self, result: OcrResult | None = None) -> None:
        self.calls: list[Path] = []
        self._result = result or OcrResult(
            blocks=[TextBlock(text="from ocr", confidence=0.8)], engine="recording"
        )

    def read(self, path: Path) -> OcrResult:
        self.calls.append(path)

        return self._result


def pdf_with_text(path: Path, pages: int = 4, per_page: int = 400) -> Path:
    """A PDF carrying a real text layer, as an exported statement does."""
    fitz = pytest.importorskip("fitz")

    document = fitz.open()

    for number in range(pages):
        page = document.new_page()
        # Wrapped into lines: one enormous line would be laid out off the page
        # and pypdfium2 would not return it.
        body = f"Statement page {number + 1}. " + ("Account 1234567890 balance 5000. " * 12)
        page.insert_textbox(fitz.Rect(40, 40, 550, 780), body[:per_page], fontsize=9)

    document.save(str(path))
    document.close()

    return path


def pdf_without_text(path: Path, pages: int = 3) -> Path:
    """A scan: pages of image data and no text layer at all."""
    fitz = pytest.importorskip("fitz")

    document = fitz.open()

    for _ in range(pages):
        document.new_page()

    document.save(str(path))
    document.close()

    return path


class TestWhatCountsAsUsable:
    def test_a_real_text_layer_is_usable(self):
        assert is_usable("x" * 5000, pages=5) is True

    def test_a_few_stray_characters_from_a_scanner_are_not(self):
        # The producer string a scanner stamps on an otherwise imageless page.
        assert is_usable("Scanned by CamScanner", pages=9) is False

    def test_plenty_of_text_but_spread_over_far_too_many_pages_is_not(self):
        # A title page of metadata in front of forty scanned pages.
        assert is_usable("x" * (MINIMUM_CHARACTERS + 50), pages=40) is False

    def test_nothing_at_all_is_not(self):
        assert is_usable("", pages=4) is False

    def test_a_document_with_no_pages_is_not(self):
        assert is_usable("x" * 5000, pages=0) is False


class TestExtract:
    def test_it_reads_the_text_and_counts_the_pages(self, tmp_path: Path):
        text, pages = extract(pdf_with_text(tmp_path / "statement.pdf", pages=4))

        assert pages == 4
        assert "Statement page 1" in text
        assert "Statement page 4" in text

    def test_a_scan_yields_nothing_rather_than_failing(self, tmp_path: Path):
        text, pages = extract(pdf_without_text(tmp_path / "scan.pdf"))

        assert text == ""

    def test_a_file_that_is_not_a_pdf_yields_nothing_rather_than_raising(self, tmp_path: Path):
        path = tmp_path / "broken.pdf"
        path.write_bytes(b"this is not a pdf")

        assert extract(path) == ("", 0)


class TestTheEngineWrapper:
    """The property that matters: OCR is skipped exactly when it should be."""

    def test_a_pdf_with_text_never_reaches_ocr(self, tmp_path: Path):
        inner = RecordingEngine()
        engine = TextLayerFirst(inner)

        result = engine.read(pdf_with_text(tmp_path / "statement.pdf"))

        assert inner.calls == [], 'OCR was run on a document that could simply be read'
        assert "Statement page 1" in result.text
        assert result.engine == "pdftext"

    def test_a_scanned_pdf_still_goes_to_ocr(self, tmp_path: Path):
        inner = RecordingEngine()
        engine = TextLayerFirst(inner)

        result = engine.read(pdf_without_text(tmp_path / "scan.pdf"))

        assert len(inner.calls) == 1, 'a scan has no text layer and must still be read'
        assert result.text == "from ocr"

    def test_a_photograph_goes_straight_to_ocr(self, tmp_path: Path):
        inner = RecordingEngine()
        path = tmp_path / "cnic.jpg"
        path.write_bytes(b"not really a jpeg")

        TextLayerFirst(inner).read(path)

        assert len(inner.calls) == 1

    def test_an_unreadable_pdf_falls_through_rather_than_losing_the_document(
        self, tmp_path: Path
    ):
        inner = RecordingEngine()
        path = tmp_path / "corrupt.pdf"
        path.write_bytes(b"%PDF-1.4 truncated")

        result = TextLayerFirst(inner).read(path)

        assert len(inner.calls) == 1
        assert result.text == "from ocr"

    def test_the_text_layer_is_reported_as_certain(self, tmp_path: Path):
        """Not a recognition score.

        A downstream confidence threshold discarding an exact read would be a
        strange way to lose a document.
        """
        result = TextLayerFirst(RecordingEngine()).read(
            pdf_with_text(tmp_path / "statement.pdf")
        )

        assert result.confidence == 1.0
