"""Reading the text a PDF already contains, instead of photographing it.

Measured on the nine documents that production could not read between 1 and 6
August 2026. Every one of them timed out — the reader spent five minutes per
attempt, twice, and returned nothing:

    pages  embedded text  OCR outcome        text layer
      6        8,034      timed out, 0 chars    50 ms
      8       10,876      timed out, 0 chars    38 ms
      9       11,883      timed out, 0 chars    32 ms
      9       11,802      timed out, 0 chars    32 ms
      5        5,076      timed out, 0 chars    56 ms
      4        9,583      timed out, 0 chars    96 ms
      8       11,481      timed out, 0 chars    34 ms
      4       11,188      timed out, 0 chars    31 ms
     16            0      timed out, 0 chars     7 ms   <- genuinely scanned

Eight of the nine were never images. They are bank statements and salary slips
exported straight to PDF, carrying a perfect text layer, and the pipeline was
rendering each page to a bitmap and asking a neural network to guess at glyphs
it could have simply read. The ninth is a real scan and still needs OCR.

## Why the failure looked like something else

The nine had nothing in common by size — 70 KB to 3.1 MB — which is what sent
the first investigation towards image resolution. The shared property was page
count: everything that failed had four or more pages, everything that succeeded
had one or two. The deadline is per document, so a sixteen-page scan gets the
same five minutes as a single photograph and cannot possibly finish.

That deadline is not wrong. It is a backstop against a wedged read, and this
module removes most of what was hitting it rather than raising it.

## Why this is not merely faster

OCR is a guess with a confidence attached. An embedded text layer is what the
program that produced the document wrote, so the digits of an account number
come back exactly, not probably. Confidence is reported as 1.0 here for that
reason — it is not a recognition score, and pretending it is one would let a
downstream threshold discard a perfect read.
"""

from __future__ import annotations

import logging
from pathlib import Path

from app.ocr.engine import OcrEngine, OcrResult, TextBlock

logger = logging.getLogger(__name__)

#: Total characters below which the layer is not worth trusting.
#:
#: A PDF of scanned pages often still carries a few stray characters — a
#: producer string, a page number stamped by the scanner — and treating those
#: as "the document reads fine" would skip OCR on exactly the documents that
#: need it. The successful reads in the sample above sat at 5,076 and up; the
#: scanned ones at 0, 10 and 20.
MINIMUM_CHARACTERS = 200

#: And per page, so a title page of metadata in front of forty scanned pages is
#: not mistaken for a readable document.
MINIMUM_PER_PAGE = 50


def extract(path: Path) -> tuple[str, int]:
    """The PDF's own text, and how many pages it has.

    Returns an empty string for anything that is not a readable PDF — a scan
    with no text layer, a corrupt file, a missing library. Never raises: this
    sits in front of OCR, and a failure here must fall through to it rather
    than cost the document.
    """
    try:
        import pypdfium2 as pdfium
    except ImportError:
        # Arrives with PaddleOCR's own dependencies, so this should not happen
        # — but a missing optional import must not stop a document being read.
        logger.debug("pypdfium2 is not available; every PDF will go to OCR.")

        return "", 0

    document = None

    try:
        document = pdfium.PdfDocument(str(path))
        pages = len(document)
        parts = []

        for index in range(pages):
            page = document[index]
            parts.append(page.get_textpage().get_text_range())

        return "\n".join(parts).strip(), pages
    except Exception:  # noqa: BLE001 - any failure means "use OCR instead"
        logger.debug("Could not read a text layer from %s.", path.name, exc_info=True)

        return "", 0
    finally:
        if document is not None:
            try:
                document.close()
            except Exception:  # noqa: BLE001
                pass


def is_usable(text: str, pages: int) -> bool:
    """Whether this text layer is the document, or just debris around a scan."""
    if pages <= 0:
        return False

    stripped = text.strip()

    return len(stripped) >= MINIMUM_CHARACTERS and len(stripped) / pages >= MINIMUM_PER_PAGE


class TextLayerFirst:
    """Reads a PDF's own text when it has one, and falls through to OCR when not.

    An OcrEngine itself, so nothing downstream knows this is here — the same
    shape as FallbackOcrEngine, and for the same reason.
    """

    def __init__(self, engine: OcrEngine) -> None:
        self._engine = engine

    @property
    def name(self) -> str:
        return f"pdftext+{self._engine.name}"

    def read(self, path: Path) -> OcrResult:
        if path.suffix.lower() != ".pdf":
            return self._engine.read(path)

        text, pages = extract(path)

        if not is_usable(text, pages):
            return self._engine.read(path)

        logger.info(
            "Read %s from its own text layer: %d characters over %d page(s), no OCR needed.",
            path.name, len(text), pages,
        )

        # One block per page, not one per document: downstream weights
        # confidence by block length, and a single enormous block would make
        # every page's contribution indistinguishable.
        blocks = [
            TextBlock(text=part.strip(), confidence=1.0)
            for part in text.split("\n")
            if part.strip()
        ]

        return OcrResult(blocks=blocks, pages=pages, engine="pdftext")
