"""Turning a photograph of a document into the document.

A client's file should hold a clean rectangular card, not a phone snap of a desk
with a hand in it. This finds the document inside the picture, straightens it,
cleans it mildly, and checks the result is genuinely better before anything uses
it.

## Falling back is the normal outcome, not the failure case

Every function here answers "could not" rather than raising, and `prepare`
returns the original image whenever any step is unsure. That is the whole design
rule: **a crop that loses part of a document is far worse than no crop at all.**
A missing corner takes the last digit off a CNIC, and the result still looks
like a clean document — nobody reviewing it would know. An uncropped photograph
is merely untidy, and untidy is recoverable.

So the bar for replacing the original is deliberately high, and every rejection
is recorded with a reason rather than silently swallowed.

## Nothing here knows about workflows

Pure functions over arrays, with one `prepare` at the bottom that touches disk.
That is what makes the hard cases testable: a tilted card, a blurred one, a card
against a busy background can all be generated, and a pipeline that can only be
tested against real photographs cannot honestly be tested at all.
"""

from __future__ import annotations

import logging
import time
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any

logger = logging.getLogger(__name__)

#: Formats worth attempting. A PDF is not a photograph of anything and is
#: excluded by the caller; this is the second line of that same rule.
IMAGE_SUFFIXES = frozenset({".jpg", ".jpeg", ".png"})


@dataclass(frozen=True, slots=True)
class Thresholds:
    """Where "good enough" sits, in one place.

    Values rather than constants scattered through the steps, so an installation
    photographing in poor light can be tuned without touching the pipeline.
    """

    min_area_ratio: float = 0.06
    """How much of the frame the document must fill.

    Below this it is more likely a card-shaped object in the scene — a phone, a
    tile, a picture frame — than the document being sent.

    Set from measurement, not intuition. The first value here was 0.18, which
    sounded conservative and silently rejected every test image: somebody
    holding a phone over a CNIC on a desk produces a card filling about an
    eighth of the frame, not a fifth. A threshold above the ordinary case
    rejects everything while looking like caution."""

    max_area_ratio: float = 0.995
    """Above this the "document" is essentially the whole frame, which means
    nothing was detected and the edges found were the photograph's own."""

    min_edge_margin: int = 2
    """How far the detected quadrilateral must sit from the frame edge.

    A document touching the edge was probably photographed with part of it
    outside the frame, and cropping to what is visible would silently remove
    whatever was cut off."""

    min_side_pixels: int = 320
    """A crop smaller than this in either dimension has thrown away the
    resolution OCR needs, whatever it looks like."""

    min_sharpness: float = 60.0
    """Variance of the Laplacian. Low means blurred; OCR on a blurred crop
    produces confident nonsense, which is worse than a refusal."""

    dark_limit: float = 45.0
    bright_limit: float = 218.0
    """Mean intensity bounds. Outside them the image is under- or overexposed
    far enough that detail is gone rather than merely dim."""

    min_rectangularity: float = 0.72
    """Quadrilateral area against its own bounding box. A real document
    photographed at an angle is still convex and roughly rectangular; a random
    contour through clutter is not."""

    frame_filled_contrast: float = 12.0
    """Below this difference in mean brightness between the frame's border and
    its middle, the photograph IS the document — held up to the camera with no
    surface visible around it.

    There is nothing to crop in that case and no quadrilateral to find, so
    detection declines and the original is kept, which is right. The number
    exists so the result can SAY that, rather than reporting "no document was
    found in the picture" about a picture that is nothing but document."""


@dataclass(slots=True)
class Prepared:
    """What the caller should use, and the account of how it was chosen."""

    path: Path
    """The image to read and to file. The cropped one if it earned its place,
    the original otherwise."""

    cropped: bool = False
    reason: str = ""
    """Why the original was kept. Empty when the crop was used."""

    metadata: dict[str, Any] = field(default_factory=dict)

    @property
    def uploaded_version(self) -> str:
        return "cropped" if self.cropped else "original"


class _Unusable(Exception):
    """A step could not produce something worth using.

    Internal. Raised freely between steps and turned into a fallback with a
    reason at the boundary, so no caller of this module ever handles it.
    """


# ── The steps ─────────────────────────────────────────────────────────────


def detect_document(image, limits: Thresholds | None = None):
    """The four corners of the largest document-shaped thing in the picture.

    Edges rather than colour: a white card on a white desk has almost no colour
    difference and a very clear edge, and colour segmentation fails exactly on
    the documents that matter most.

    Returns the corners ordered top-left, top-right, bottom-right, bottom-left,
    or None when nothing convincing was found — which is a normal answer.
    """
    import cv2
    import numpy as np

    limits = limits or Thresholds()
    height, width = image.shape[:2]
    frame_area = float(height * width)

    grey = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)

    # Every mask is tried and the largest answer wins, rather than the first
    # mask that produces any answer at all.
    #
    # The masks are not ranked by quality — they are three different ways of
    # seeing, each of which wins on some photographs. Returning on the first
    # meant a marginal quadrilateral from the edge mask beat a cleaner one the
    # threshold mask would have found a moment later, purely on running order.
    best = None
    best_area = 0.0

    for mask in _candidate_masks(grey, height, width):
        # RETR_LIST, not RETR_EXTERNAL. The document is very often *inside*
        # another contour — the edge of a desk, a patterned surface, or a noisy
        # background that closes into one blob. Taking only outermost contours
        # finds that blob, rejects it for filling the frame, and reports no
        # document while the card sits plainly inside it.
        contours, _ = cv2.findContours(mask, cv2.RETR_LIST, cv2.CHAIN_APPROX_SIMPLE)

        found = _best_quad(contours, frame_area, limits, height, width)

        if found is None:
            continue

        area = cv2.contourArea(found.reshape(-1, 1, 2).astype("float32"))

        if area > best_area:
            best, best_area = found, area

    return _ordered(best) if best is not None else None


def fills_frame(image, limits: Thresholds | None = None) -> bool:
    """Is the photograph already the document, edge to edge?

    A close-up of a CNIC held to the camera has no background, so there is
    nothing to crop and no quadrilateral to find. Detection correctly declines
    and the original is correctly kept — but "no document was found in the
    picture" is the wrong thing to say about a picture that is nothing but
    document, and it made these look like failures when they were not.

    Judged on the border rather than on any contour: the frame's edge and its
    middle being equally bright means the document reaches the edge, which is
    precisely the situation where a contour detector has nothing to key on.
    """
    import cv2
    import numpy as np

    limits = limits or Thresholds()
    height, width = image.shape[:2]
    band = max(4, min(height, width) // 25)
    grey = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY).astype(float)

    border = np.concatenate([
        grey[:band].ravel(), grey[-band:].ravel(),
        grey[:, :band].ravel(), grey[:, -band:].ravel(),
    ])
    centre = grey[height // 4: 3 * height // 4, width // 4: 3 * width // 4].ravel()

    return abs(float(border.mean()) - float(centre.mean())) < limits.frame_filled_contrast


def _candidate_masks(grey, height: int, width: int):
    """Ways of separating a document from what it is lying on, in order.

    Two, because one is not enough on real photographs. Edges are precise where
    a card meets a contrasting surface, and on a real forward of a CNIC on a
    desk they fragmented into 1361 contours whose largest was 2% of the frame —
    the card, at 27%, was never traced at all. Adaptive thresholding found the
    same card as a convex quadrilateral covering 56%.

    Neither wins everywhere, so both are tried and the first that yields a
    plausible document is taken.
    """
    import cv2
    import numpy as np

    # Scaled to the image: a fixed kernel that closes a gap on a 900px photo
    # does nothing on a 4000px one.
    span = max(9, (int(min(height, width) * 0.023) | 1))
    kernel = np.ones((span, span), np.uint8)

    # 1. Edges. Blurred first, because sensor noise makes thousands of tiny
    #    edges that findContours will happily join into a convincing shape.
    blurred = cv2.GaussianBlur(grey, (5, 5), 0)
    yield cv2.morphologyEx(cv2.Canny(blurred, 60, 180), cv2.MORPH_CLOSE, kernel)

    # 2. Local thresholding, for the case edges cannot see: a pale card on a
    #    pale surface, or a border broken by shadow.
    threshold = cv2.adaptiveThreshold(
        cv2.medianBlur(grey, 7), 255,
        cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY, 31, 8,
    )

    yield cv2.morphologyEx(255 - threshold, cv2.MORPH_CLOSE, kernel)

    # 3. One global threshold, closed only lightly.
    #
    #    The one that actually found the card on the first real desk photo. The
    #    two above either could not see its border or, with a kernel wide enough
    #    to bridge the gaps, welded the card to its own shadow into a blob
    #    covering 56% of the frame and touching every edge. Otsu separates a
    #    bright card from a darker surface in one step, and a light kernel keeps
    #    the two apart: the same card came back at 20% with a 67px margin.
    light = np.ones((max(5, span // 3) | 1,) * 2, np.uint8)
    _, otsu = cv2.threshold(cv2.GaussianBlur(grey, (7, 7), 0), 0, 255,
                            cv2.THRESH_BINARY + cv2.THRESH_OTSU)

    yield cv2.morphologyEx(otsu, cv2.MORPH_CLOSE, light)


def _best_quad(contours, frame_area: float, limits: Thresholds, height: int, width: int):
    """The largest contour that could plausibly be a document.

    "Largest" is qualified by one rule learned from a real photograph: a
    quadrilateral touching the frame edge is discarded here rather than at
    validation. A blob that has swallowed the document and run to the edges is
    always bigger than the document, so left in the running it wins every time
    and the correctly-margined card behind it is never seen.
    """
    import cv2

    best = None
    best_area = 0.0

    for contour in sorted(contours, key=cv2.contourArea, reverse=True)[:12]:
        area = cv2.contourArea(contour)

        if area / frame_area < limits.min_area_ratio:
            # Sorted by area, so everything after this is smaller still.
            break

        approx = _as_quad(contour, area, limits)

        if approx is None:
            continue

        box_w, box_h = cv2.boundingRect(approx)[2:]

        if not box_w or not box_h:
            continue

        if area / float(box_w * box_h) < limits.min_rectangularity:
            continue

        # Filtered here rather than after choosing a winner. A contour tracing
        # the photograph's own border is quadrilateral, convex and larger than
        # anything else, so it wins `best` every time — and rejecting it at the
        # end then reports "no document" while the card sits plainly inside it.
        if area / frame_area > limits.max_area_ratio:
            continue

        points = approx.reshape(4, 2).astype("float32")
        margin = min(
            points[:, 0].min(), points[:, 1].min(),
            width - 1 - points[:, 0].max(), height - 1 - points[:, 1].max(),
        )

        if margin < limits.min_edge_margin:
            continue

        if area > best_area:
            best, best_area = points, area

    return best


def _as_quad(contour, area: float, limits: Thresholds):
    """Four corners for this contour, or None if it is not document-shaped.

    ## Why the tolerance is swept rather than fixed

    approxPolyDP simplifies a contour until it is within `epsilon` of the
    original, and epsilon was pinned at 2% of the perimeter. That is the right
    number for a card lying flat on a desk with four crisp corners, and too
    tight for most of what actually arrives: a slightly bent CNIC, a corner
    lifting off the surface, a shadow softening one edge. Those simplify to five
    or six points at 2% and were thrown away as "not a document" while being
    plainly, visibly a document.

    Tried tightest-first, so a photograph that WAS clean still yields the precise
    quadrilateral it always did. Only the ones that failed get the looser passes.

    ## Why a rotated box is an acceptable last resort

    A contour that survives every earlier gate is already large, and about to be
    checked for rectangularity against its own bounding box, sharpness, exposure
    and edge margin. If it passes all of those and still will not simplify to
    four points, the shape is a rectangle whose outline is merely noisy — and
    minAreaRect is the right description of it. The rectangularity test is what
    keeps this from becoming "draw a box around any old blob": a contour that
    does not fill its own rotated box is rejected before it gets here.
    """
    import cv2

    perimeter = cv2.arcLength(contour, True)

    for epsilon in (0.02, 0.035, 0.05):
        approx = cv2.approxPolyDP(contour, epsilon * perimeter, True)

        if len(approx) == 4 and cv2.isContourConvex(approx):
            return approx

    rect = cv2.minAreaRect(contour)
    (_, (rect_w, rect_h), _) = rect

    if not rect_w or not rect_h:
        return None

    # How much of its own rotated box the contour actually occupies. A real
    # document fills nearly all of it; an L-shaped smear of clutter does not.
    if area / float(rect_w * rect_h) < limits.min_rectangularity:
        return None

    return cv2.boxPoints(rect).astype("int32").reshape(4, 1, 2)


def four_point_transform(image, corners):
    """The document, seen straight on.

    The output size is taken from the longest opposing edges rather than from a
    fixed size, so a card photographed close up is not downsampled to match one
    photographed far away.
    """
    import cv2
    import numpy as np

    tl, tr, br, bl = corners

    width = int(max(_distance(br, bl), _distance(tr, tl)))
    height = int(max(_distance(tr, br), _distance(tl, bl)))

    if width < 2 or height < 2:
        raise _Unusable("the detected corners enclose nothing")

    target = np.array(
        [[0, 0], [width - 1, 0], [width - 1, height - 1], [0, height - 1]],
        dtype="float32",
    )

    return cv2.warpPerspective(image, cv2.getPerspectiveTransform(corners, target), (width, height))


def enhance(image):
    """Mild, deliberately.

    Every operation here can make OCR worse if pushed. Aggressive denoising
    eats thin strokes, hard thresholding turns a faint digit into background,
    and strong sharpening manufactures edges that the recogniser reads as
    punctuation. The aim is a document a person would call clean, not a
    maximally processed one.
    """
    import cv2

    lab = cv2.cvtColor(image, cv2.COLOR_BGR2LAB)
    lightness, a, b = cv2.split(lab)

    # Local contrast, clipped low. Shadows across a card lift without the
    # bright half blowing out, which a global equalisation would do.
    lightness = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8)).apply(lightness)

    evened = cv2.cvtColor(cv2.merge((lightness, a, b)), cv2.COLOR_LAB2BGR)

    # Edge-preserving, so strokes survive while sensor noise goes.
    denoised = cv2.bilateralFilter(evened, 5, 45, 45)

    # Unsharp masking at low weight — visibly crisper, not haloed.
    blurred = cv2.GaussianBlur(denoised, (0, 0), 2.0)

    return cv2.addWeighted(denoised, 1.35, blurred, -0.35, 0)


def sharpness(image) -> float:
    """Variance of the Laplacian: high for crisp edges, near zero for blur."""
    import cv2

    grey = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY) if image.ndim == 3 else image

    return float(cv2.Laplacian(grey, cv2.CV_64F).var())


def exposure(image) -> float:
    """Mean intensity, 0 (black) to 255 (white)."""
    import cv2

    grey = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY) if image.ndim == 3 else image

    return float(grey.mean())


def validate(cropped, corners, frame_shape, limits: Thresholds | None = None) -> list[str]:
    """Every reason this crop should not replace the original.

    Returns a list rather than a verdict: an operator asking why a document was
    not cropped deserves the actual reasons, and a boolean would throw them away.
    """
    limits = limits or Thresholds()
    problems: list[str] = []

    height, width = cropped.shape[:2]
    frame_h, frame_w = frame_shape[:2]

    if min(height, width) < limits.min_side_pixels:
        problems.append(f"the crop is only {width}x{height}; too small to read reliably")

    # A document touching the frame edge was probably photographed with part of
    # it outside. Cropping to what is visible would remove the rest silently.
    margin = limits.min_edge_margin
    xs = [float(p[0]) for p in corners]
    ys = [float(p[1]) for p in corners]

    if min(xs) <= margin or min(ys) <= margin or max(xs) >= frame_w - 1 - margin or max(ys) >= frame_h - 1 - margin:
        problems.append("the document touches the edge of the photograph; part of it may be outside")

    if (value := sharpness(cropped)) < limits.min_sharpness:
        problems.append(f"the crop is blurred (sharpness {value:.0f})")

    if (mean := exposure(cropped)) < limits.dark_limit:
        problems.append(f"the crop is underexposed (mean {mean:.0f})")
    elif mean > limits.bright_limit:
        problems.append(f"the crop is overexposed (mean {mean:.0f})")

    return problems


# ── The one function that touches disk ────────────────────────────────────


def prepare(
    path: Path | str,
    *,
    debug: bool = False,
    limits: Thresholds | None = None,
) -> Prepared:
    """Produce the image that should be read and filed.

    Never raises. Every failure — an unreadable file, a missing OpenCV, a crop
    that did not earn its place — returns the original path with a reason, so a
    caller can use the result without a try block and a document is never lost
    to a preprocessing bug.
    """
    started = time.perf_counter()
    source = Path(path)
    limits = limits or Thresholds()
    metadata: dict[str, Any] = {"uploaded_version": "original", "crop_success": False}

    def fall_back(reason: str) -> Prepared:
        metadata["processing_time_ms"] = round((time.perf_counter() - started) * 1000)
        metadata["fallback_reason"] = reason
        logger.info("Keeping the original of %s: %s", source.name, reason)

        return Prepared(path=source, cropped=False, reason=reason, metadata=metadata)

    if source.suffix.lower() not in IMAGE_SUFFIXES:
        return fall_back("not an image")

    try:
        import cv2
    except Exception:  # noqa: BLE001 - an optional dependency, not a failure
        return fall_back("OpenCV is not available")

    try:
        image = cv2.imread(str(source))

        if image is None:
            return fall_back("the file could not be read as an image")

        metadata["original_resolution"] = f"{image.shape[1]}x{image.shape[0]}"
        metadata["original_sharpness"] = round(sharpness(image), 1)
        metadata["original_exposure"] = round(exposure(image), 1)

        corners = detect_document(image, limits)

        if corners is None:
            # Said accurately rather than as a catch-all. A close-up with no
            # surface around it is not a detection failure — it is a photograph
            # that needs no cropping, and the original is already the right
            # answer. Calling both "no document was found" made a correct
            # outcome indistinguishable from a miss, in the logs and in every
            # measurement taken from them.
            if fills_frame(image, limits):
                metadata["already_cropped"] = True

                return fall_back("the document already fills the frame")

            return fall_back("no document was found in the picture")

        metadata["corners_detected"] = True

        straightened = four_point_transform(image, corners)
        metadata["perspective_corrected"] = True

        cleaned = enhance(straightened)

        if problems := validate(cleaned, corners, image.shape, limits):
            metadata["validation_problems"] = problems

            return fall_back(problems[0])

        target = source.with_name(f"{source.stem}-cropped{source.suffix}")

        if not cv2.imwrite(str(target), cleaned):
            return fall_back("the cropped image could not be written")

        metadata.update(
            crop_success=True,
            uploaded_version="cropped",
            cropped_resolution=f"{cleaned.shape[1]}x{cleaned.shape[0]}",
            cropped_sharpness=round(sharpness(cleaned), 1),
            cropped_exposure=round(exposure(cleaned), 1),
            processing_time_ms=round((time.perf_counter() - started) * 1000),
            debug=debug,
        )

        if not debug:
            # Production keeps the document, not the working copy. In debug both
            # survive, which is the only way to see why a crop went wrong.
            metadata["original_kept"] = False

        logger.info(
            "Cropped %s to %s", source.name, metadata["cropped_resolution"]
        )

        return Prepared(path=target, cropped=True, metadata=metadata)
    except _Unusable as unusable:
        return fall_back(str(unusable))
    except Exception as exc:  # noqa: BLE001
        # A bug in here must never cost a document. The original is always a
        # correct answer; this pipeline only ever offers a better one.
        logger.exception("Preprocessing %s failed", source.name)

        return fall_back(f"preprocessing failed: {exc}")


# ── Internals ─────────────────────────────────────────────────────────────


def _ordered(points):
    """Corners as top-left, top-right, bottom-right, bottom-left.

    By sums and differences rather than by angle: the top-left has the smallest
    x+y and the top-right the smallest y-x, which holds for any rotation up to
    the point where "top" stops meaning anything.
    """
    import numpy as np

    ordered = np.zeros((4, 2), dtype="float32")
    total = points.sum(axis=1)
    diff = np.diff(points, axis=1).ravel()

    ordered[0] = points[np.argmin(total)]
    ordered[2] = points[np.argmax(total)]
    ordered[1] = points[np.argmin(diff)]
    ordered[3] = points[np.argmax(diff)]

    return ordered


def _distance(a, b) -> float:
    return float(((a[0] - b[0]) ** 2 + (a[1] - b[1]) ** 2) ** 0.5)
