"""QA matching must be reproducible and must credit the best candidate. The greedy IoU matcher decides which candidate is reported as a match and which becomes false-positive evidence for an operator. If that decision depends on database row order, the same run produces different scores and points reviewers at the wrong geometry. """ from __future__ import annotations from shapely.geometry import box from app.services.qa_service import QaService REFERENCE = [({"id": "R1"}, box(0.0, 0.0, 10.0, 10.0))] # Two detections of the same building. ``sloppy`` is far too tall (IoU 0.51), # ``accurate`` is nearly exact (IoU 0.96). SLOPPY = box(0.0, 0.0, 10.0, 19.5) ACCURATE = box(0.0, 0.0, 10.0, 10.4) def _candidates(order: list[tuple[str, float]]) -> list[tuple[dict, object]]: geometries = {"sloppy": SLOPPY, "accurate": ACCURATE} return [ ({"id": name, "confidence": confidence}, geometries[name]) for name, confidence in order ] def test_matching_is_independent_of_candidate_row_order() -> None: forward = QaService._match_io_u_evidence( _candidates([("sloppy", 0.42), ("accurate", 0.91)]), REFERENCE, 0.5 ) reverse = QaService._match_io_u_evidence( _candidates([("accurate", 0.91), ("sloppy", 0.42)]), REFERENCE, 0.5 ) assert forward.matches == reverse.matches assert forward.false_positives == reverse.false_positives assert forward.false_negatives == reverse.false_negatives assert forward.match_iou_values == reverse.match_iou_values assert forward.match_evidence == reverse.match_evidence assert forward.false_positive_evidence == reverse.false_positive_evidence def test_highest_confidence_candidate_claims_the_reference() -> None: evidence = QaService._match_io_u_evidence( _candidates([("sloppy", 0.42), ("accurate", 0.91)]), REFERENCE, 0.5 ) assert evidence.matches == 1 assert evidence.match_evidence[0]["candidate_feature_id"] == "accurate" assert evidence.false_positive_evidence == [{"candidate_feature_id": "sloppy"}] assert evidence.match_iou_values[0] > 0.9 def test_matching_without_confidence_is_still_deterministic() -> None: """Vector-vs-vector QA has no confidence; identity keeps it reproducible.""" left = [ ({"id": "b-second"}, SLOPPY), ({"id": "a-first"}, ACCURATE), ] right = list(reversed(left)) assert QaService._match_io_u_evidence(left, REFERENCE, 0.5).match_evidence == ( QaService._match_io_u_evidence(right, REFERENCE, 0.5).match_evidence ) def test_evidence_is_ordered_by_confidence_for_review() -> None: evidence = QaService._match_io_u_evidence( [ ({"id": "low", "confidence": 0.30}, box(30.0, 30.0, 31.0, 31.0)), ({"id": "high", "confidence": 0.95}, box(40.0, 40.0, 41.0, 41.0)), ({"id": "mid", "confidence": 0.60}, box(50.0, 50.0, 51.0, 51.0)), ], REFERENCE, 0.5, ) assert [item["candidate_feature_id"] for item in evidence.false_positive_evidence] == [ "high", "mid", "low", ]