Files
geointel/backend/tests/test_reviewed_detection_metrics.py
T
Jens faeb58ef6d
GeoIntel release gates / Compile, test, contracts and builds (push) Successful in 1m49s
GeoIntel release gates / Python and npm vulnerability policy (push) Successful in 21s
GeoIntel release gates / Production AI image, SBOM and container scan (push) Successful in 5m39s
GeoIntel release gates / Deploy exact gated revision to Unraid (push) Failing after 58m43s
Initial public release
2026-08-31 21:56:53 +02:00

255 lines
9.6 KiB
Python

"""An operator's adjudication must reach the score.
The review vocabulary already distinguishes a model error from a reference gap
— the product's own position is that official footprints are not automatically
perfect ground truth. But the reviews were only counted. An operator who
inspects forty false positives and establishes that twelve are buildings the
reference simply lacks still sees a precision that counts all forty against the
model, and that they have personally disproved.
Because part of the evidence is usually still unreviewed, the honest answer is
an interval, not a single corrected number: pessimistic assumes every
unreviewed item is a model error, optimistic assumes none is.
"""
from __future__ import annotations
import pytest
from app.services.reviewed_metrics_service import ReviewedMetricsService
def _counts(**decisions: int) -> dict[str, int]:
return decisions
class TestAdjudication:
def test_a_reference_gap_stops_counting_against_precision(self) -> None:
result = ReviewedMetricsService.adjudicate(
matches=80,
false_positives=20,
false_negatives=10,
false_positive_decisions=_counts(reference_gap_or_change=20),
false_negative_decisions={},
)
# Every false positive was the reference missing a real building.
assert result["adjudicated"]["false_positives"] == 0
assert result["adjudicated"]["precision"] == pytest.approx(1.0)
def test_a_confirmed_model_error_keeps_counting(self) -> None:
result = ReviewedMetricsService.adjudicate(
matches=80,
false_positives=20,
false_negatives=0,
false_positive_decisions=_counts(confirmed_model_false_positive=20),
false_negative_decisions={},
)
assert result["adjudicated"]["false_positives"] == 20
assert result["adjudicated"]["precision"] == pytest.approx(0.8)
def test_an_alignment_mismatch_is_not_a_model_error(self) -> None:
"""Both the detection and the footprint were right; the matching failed."""
result = ReviewedMetricsService.adjudicate(
matches=80,
false_positives=20,
false_negatives=0,
false_positive_decisions=_counts(qa_alignment_mismatch=20),
false_negative_decisions={},
)
assert result["adjudicated"]["false_positives"] == 0
def test_a_reference_gap_on_a_miss_stops_counting_against_recall(self) -> None:
result = ReviewedMetricsService.adjudicate(
matches=80,
false_positives=0,
false_negatives=20,
false_positive_decisions={},
false_negative_decisions=_counts(reference_gap_or_change=20),
)
# The reference held twenty footprints that are not there.
assert result["adjudicated"]["false_negatives"] == 0
assert result["adjudicated"]["recall"] == pytest.approx(1.0)
def test_an_uncertain_verdict_keeps_counting_against_the_model(self) -> None:
"""Not being able to judge is not evidence in the model's favour."""
result = ReviewedMetricsService.adjudicate(
matches=80,
false_positives=20,
false_negatives=0,
false_positive_decisions=_counts(uncertain=10, imagery_obscured_or_uncertain=10),
false_negative_decisions={},
)
assert result["adjudicated"]["false_positives"] == 20
class TestBounds:
def test_a_partly_reviewed_check_reports_an_interval(self) -> None:
result = ReviewedMetricsService.adjudicate(
matches=80,
false_positives=20,
false_negatives=0,
false_positive_decisions=_counts(reference_gap_or_change=10),
false_negative_decisions={},
)
# Ten unreviewed: pessimistically all model errors, optimistically none.
assert result["pessimistic"]["precision"] == pytest.approx(80 / 90)
assert result["optimistic"]["precision"] == pytest.approx(1.0)
assert result["review_complete"] is False
def test_a_fully_reviewed_check_collapses_the_interval(self) -> None:
result = ReviewedMetricsService.adjudicate(
matches=80,
false_positives=20,
false_negatives=5,
false_positive_decisions=_counts(reference_gap_or_change=12, confirmed_model_false_positive=8),
false_negative_decisions=_counts(confirmed_model_false_negative=5),
)
assert result["review_complete"] is True
assert result["pessimistic"]["precision"] == pytest.approx(result["optimistic"]["precision"])
assert result["adjudicated"]["precision"] == pytest.approx(80 / 88)
def test_an_unreviewed_check_reports_the_raw_numbers_unchanged(self) -> None:
result = ReviewedMetricsService.adjudicate(
matches=80,
false_positives=20,
false_negatives=10,
false_positive_decisions={},
false_negative_decisions={},
)
assert result["review_complete"] is False
assert result["adjudicated"]["precision"] == pytest.approx(result["raw"]["precision"])
assert result["adjudicated"]["recall"] == pytest.approx(result["raw"]["recall"])
def test_the_raw_score_is_always_reported_alongside(self) -> None:
result = ReviewedMetricsService.adjudicate(
matches=80,
false_positives=20,
false_negatives=0,
false_positive_decisions=_counts(reference_gap_or_change=20),
false_negative_decisions={},
)
assert result["raw"]["precision"] == pytest.approx(0.8)
assert result["adjudicated"]["precision"] == pytest.approx(1.0)
class TestEdges:
def test_a_check_without_findings_makes_no_claim(self) -> None:
result = ReviewedMetricsService.adjudicate(
matches=0,
false_positives=0,
false_negatives=0,
false_positive_decisions={},
false_negative_decisions={},
)
assert result["adjudicated"]["precision"] is None
assert result["adjudicated"]["recall"] is None
assert result["review_complete"] is True
def test_more_decisions_than_findings_cannot_invent_a_negative_count(self) -> None:
result = ReviewedMetricsService.adjudicate(
matches=10,
false_positives=2,
false_negatives=0,
false_positive_decisions=_counts(reference_gap_or_change=99),
false_negative_decisions={},
)
assert result["adjudicated"]["false_positives"] == 0
def test_an_unknown_decision_is_treated_as_no_judgement(self) -> None:
result = ReviewedMetricsService.adjudicate(
matches=10,
false_positives=5,
false_negatives=0,
false_positive_decisions=_counts(something_new_from_a_later_release=5),
false_negative_decisions={},
)
assert result["adjudicated"]["false_positives"] == 5
assert result["review_complete"] is False
class TestThroughTheReviewPanel:
"""The score the panel shows, not just the arithmetic behind it."""
def _quality_check(self, quality_check_id, project_id):
from app.models import QualityCheck
return QualityCheck(
id=quality_check_id,
project_id=project_id,
reference_dataset_id=__import__("uuid").uuid4(),
check_type="detections_vs_reference",
status="ok",
findings_json={
"matches": 80,
"false_positives": 20,
"false_negatives": 0,
"false_positive_evidence": [
{"candidate_feature_id": f"detection-{index}"} for index in range(20)
],
"false_negative_evidence": [],
},
)
def test_adjudicated_reference_gaps_raise_the_reported_precision(self) -> None:
import uuid
from app.models import DetectionReview, QualityCheck
from app.services.detection_review_service import DetectionReviewService
quality_check_id, project_id = uuid.uuid4(), uuid.uuid4()
quality_check = self._quality_check(quality_check_id, project_id)
reviews = [
DetectionReview(
id=uuid.uuid4(),
quality_check_id=quality_check_id,
evidence_role="false_positive",
evidence_feature_id=f"detection-{index}",
decision="reference_gap_or_change",
)
for index in range(12)
]
class _Query:
def __init__(self, rows):
self.rows = rows
def filter(self, *_args):
return self
def all(self):
return self.rows
class _Session:
def get(self, model, item_id):
return quality_check if model is QualityCheck and item_id == quality_check_id else None
def query(self, _model):
return _Query(reviews)
result = DetectionReviewService.list_reviews(
_Session(), project_id=project_id, quality_check_id=quality_check_id
)
metrics = result.summary.reviewed_metrics
assert metrics is not None
assert metrics["raw"]["precision"] == pytest.approx(0.8)
# Twelve of the twenty were the reference missing a building.
assert metrics["adjudicated"]["precision"] == pytest.approx(80 / 88)
assert metrics["review_complete"] is False
assert metrics["false_positive_breakdown"]["exonerated"] == 12
assert metrics["false_positive_breakdown"]["unreviewed"] == 8