"""Putting recognised regions into reading order.

An OCR engine returns regions in detection order, which is roughly top-to-bottom
and is not reading order. That matters more than it sounds: everything
downstream — classification and every extraction pattern — runs over
``OcrResult.text``, which is the blocks joined by newlines.

Get the order wrong and a CNIC printed as "Identity Number:" on one line and
"35202-1234567-1" on the next arrives with a paragraph between them, and the
pattern that would have matched does not. The document is then "unreadable" for
a reason that has nothing to do with the scan.

Pure geometry, no engine: testable exactly.
"""

from __future__ import annotations

from dataclasses import dataclass


@dataclass(frozen=True, slots=True)
class Region:
    """A recognised region and where it sat on the page."""

    text: str
    confidence: float
    left: float
    top: float
    right: float
    bottom: float

    @property
    def height(self) -> float:
        return max(self.bottom - self.top, 1.0)

    @property
    def middle(self) -> float:
        return (self.top + self.bottom) / 2


def from_polygon(text: str, confidence: float, polygon) -> Region:
    """Build a Region from a detection polygon.

    Engines return four corners, and a photographed document is never square to
    the camera — so the corners are not axis-aligned and the bounding box is
    taken from their extremes rather than from two of them.
    """
    xs = [float(point[0]) for point in polygon]
    ys = [float(point[1]) for point in polygon]

    return Region(
        text=text,
        confidence=confidence,
        left=min(xs),
        top=min(ys),
        right=max(xs),
        bottom=max(ys),
    )


def reading_order(regions: list[Region], line_tolerance: float = 0.6) -> list[list[Region]]:
    """Group regions into lines, top to bottom, each ordered left to right.

    Regions are on the same line when their vertical middles are within
    ``line_tolerance`` of a typical character height — a fraction rather than a
    fixed pixel count, because a phone photograph and a flatbed scan of the same
    document differ by an order of magnitude in resolution, and a threshold in
    pixels works for exactly one of them.

    Grouping by exact `top` would split a line wherever one word sits a pixel
    higher, which on a photographed document is most of them.
    """
    if not regions:
        return []

    ordered = sorted(regions, key=lambda r: (r.middle, r.left))
    threshold = _typical_height(ordered) * line_tolerance

    lines: list[list[Region]] = [[ordered[0]]]

    for region in ordered[1:]:
        current = lines[-1]
        # Compared against the line's own first region rather than the previous
        # one, so a line does not drift downward across a long row of regions
        # each slightly lower than the last.
        if abs(region.middle - current[0].middle) <= threshold:
            current.append(region)
        else:
            lines.append([region])

    return [sorted(line, key=lambda r: r.left) for line in lines]


def _typical_height(regions: list[Region]) -> float:
    """The median region height.

    Median, not mean: one full-width heading or a mis-detected block spanning
    half the page would drag a mean upward and merge genuinely separate lines
    into one.
    """
    heights = sorted(r.height for r in regions)

    return heights[len(heights) // 2]
