"""Reading order.

Everything downstream runs over the joined text, so the order regions come back
in decides whether extraction works at all. A CNIC printed as a label on one line
and digits on the next only matches if those two arrive adjacent.

Pure geometry — no engine, no images, exactly testable.
"""

from __future__ import annotations

from app.ocr.layout import Region, from_polygon, reading_order


def region(text: str, top: float, left: float, height: float = 20, width: float = 100) -> Region:
    return Region(
        text=text,
        confidence=0.9,
        left=left,
        top=top,
        right=left + width,
        bottom=top + height,
    )


def flatten(lines: list[list[Region]]) -> list[str]:
    return [" ".join(r.text for r in line) for line in lines]


class TestGrouping:
    def test_regions_on_one_line_stay_together(self):
        lines = reading_order([
            region("Identity", top=100, left=10),
            region("Number:", top=100, left=120),
            region("35202-1234567-1", top=100, left=240),
        ])

        assert flatten(lines) == ["Identity Number: 35202-1234567-1"]

    def test_lines_read_top_to_bottom(self):
        lines = reading_order([
            region("second", top=140, left=10),
            region("first", top=100, left=10),
            region("third", top=180, left=10),
        ])

        assert flatten(lines) == ["first", "second", "third"]

    def test_words_read_left_to_right(self):
        lines = reading_order([
            region("Ali", top=100, left=200),
            region("Muhammad", top=100, left=60),
            region("Name:", top=100, left=10),
        ])

        assert flatten(lines) == ["Name: Muhammad Ali"]

    def test_a_slightly_uneven_line_is_not_split(self):
        # A photographed document is never square to the camera. Grouping by
        # exact `top` would split most lines on most real inputs.
        lines = reading_order([
            region("Date", top=100, left=10),
            region("of", top=103, left=70),
            region("Birth", top=98, left=110),
        ])

        assert flatten(lines) == ["Date of Birth"]

    def test_genuinely_separate_lines_are_kept_apart(self):
        lines = reading_order([
            region("Name: Muhammad Ali", top=100, left=10),
            region("CNIC: 35202-1234567-1", top=140, left=10),
        ])

        assert len(lines) == 2

    def test_a_long_row_does_not_drift_into_the_next_line(self):
        # Each region a little lower than the last. Compared against the previous
        # region rather than the line's first, the tolerance would creep and
        # eventually swallow the row below.
        drifting = [region(f"w{i}", top=100 + i * 4, left=i * 60) for i in range(8)]
        below = region("next line", top=160, left=10)

        lines = reading_order([*drifting, below])

        assert lines[-1][0].text == "next line"

    def test_no_regions_is_no_lines(self):
        assert reading_order([]) == []

    def test_one_region_is_one_line(self):
        assert flatten(reading_order([region("alone", top=0, left=0)])) == ["alone"]


class TestScaleIndependence:
    """The same document photographed and scanned differs by an order of magnitude."""

    def _document(self, scale: float) -> list[Region]:
        return [
            region("Name:", top=100 * scale, left=10 * scale, height=20 * scale, width=60 * scale),
            region("Ali", top=100 * scale, left=80 * scale, height=20 * scale, width=40 * scale),
            region("CNIC:", top=140 * scale, left=10 * scale, height=20 * scale, width=60 * scale),
        ]

    def test_a_small_scan_groups_correctly(self):
        assert flatten(reading_order(self._document(1))) == ["Name: Ali", "CNIC:"]

    def test_a_large_photograph_groups_identically(self):
        # A tolerance in pixels would work for exactly one of these.
        assert flatten(reading_order(self._document(6))) == ["Name: Ali", "CNIC:"]


class TestTypicalHeight:
    def test_one_huge_heading_does_not_merge_the_body(self):
        heading = region("GOVERNMENT OF PAKISTAN", top=0, left=0, height=120, width=600)
        body = [
            region("Name: Ali", top=200, left=10, height=18),
            region("CNIC: 35202", top=230, left=10, height=18),
        ]

        # A mean height would be dragged up by the heading and the two body lines
        # would fall inside one tolerance band.
        lines = reading_order([heading, *body])

        assert len(lines) == 3


class TestPolygons:
    def test_a_bounding_box_is_taken_from_the_extremes(self):
        # A photographed document is rotated, so the corners are not axis
        # aligned and picking two of them gives the wrong box.
        rotated = [[12, 100], [210, 92], [214, 118], [16, 126]]

        built = from_polygon("Name: Ali", 0.93, rotated)

        assert built.left == 12
        assert built.top == 92
        assert built.right == 214
        assert built.bottom == 126

    def test_it_carries_the_text_and_confidence(self):
        built = from_polygon("35202-1234567-1", 0.87, [[0, 0], [10, 0], [10, 10], [0, 10]])

        assert built.text == "35202-1234567-1"
        assert built.confidence == 0.87

    def test_a_zero_height_region_does_not_break_the_maths(self):
        # A detection collapsed to a line. Height feeds a division.
        flat = from_polygon("x", 0.5, [[0, 50], [10, 50], [10, 50], [0, 50]])

        assert flat.height >= 1.0
        assert reading_order([flat])
