GeoIntel release gates / Compile, test, contracts and builds (push) Successful in 1m49s
GeoIntel release gates / Python and npm vulnerability policy (push) Successful in 21s
GeoIntel release gates / Production AI image, SBOM and container scan (push) Successful in 5m39s
GeoIntel release gates / Deploy exact gated revision to Unraid (push) Failing after 58m43s
255 lines
9.6 KiB
Python
255 lines
9.6 KiB
Python
"""An operator's adjudication must reach the score.
|
|
|
|
The review vocabulary already distinguishes a model error from a reference gap
|
|
— the product's own position is that official footprints are not automatically
|
|
perfect ground truth. But the reviews were only counted. An operator who
|
|
inspects forty false positives and establishes that twelve are buildings the
|
|
reference simply lacks still sees a precision that counts all forty against the
|
|
model, and that they have personally disproved.
|
|
|
|
Because part of the evidence is usually still unreviewed, the honest answer is
|
|
an interval, not a single corrected number: pessimistic assumes every
|
|
unreviewed item is a model error, optimistic assumes none is.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import pytest
|
|
|
|
from app.services.reviewed_metrics_service import ReviewedMetricsService
|
|
|
|
|
|
def _counts(**decisions: int) -> dict[str, int]:
|
|
return decisions
|
|
|
|
|
|
class TestAdjudication:
|
|
def test_a_reference_gap_stops_counting_against_precision(self) -> None:
|
|
result = ReviewedMetricsService.adjudicate(
|
|
matches=80,
|
|
false_positives=20,
|
|
false_negatives=10,
|
|
false_positive_decisions=_counts(reference_gap_or_change=20),
|
|
false_negative_decisions={},
|
|
)
|
|
|
|
# Every false positive was the reference missing a real building.
|
|
assert result["adjudicated"]["false_positives"] == 0
|
|
assert result["adjudicated"]["precision"] == pytest.approx(1.0)
|
|
|
|
def test_a_confirmed_model_error_keeps_counting(self) -> None:
|
|
result = ReviewedMetricsService.adjudicate(
|
|
matches=80,
|
|
false_positives=20,
|
|
false_negatives=0,
|
|
false_positive_decisions=_counts(confirmed_model_false_positive=20),
|
|
false_negative_decisions={},
|
|
)
|
|
|
|
assert result["adjudicated"]["false_positives"] == 20
|
|
assert result["adjudicated"]["precision"] == pytest.approx(0.8)
|
|
|
|
def test_an_alignment_mismatch_is_not_a_model_error(self) -> None:
|
|
"""Both the detection and the footprint were right; the matching failed."""
|
|
|
|
result = ReviewedMetricsService.adjudicate(
|
|
matches=80,
|
|
false_positives=20,
|
|
false_negatives=0,
|
|
false_positive_decisions=_counts(qa_alignment_mismatch=20),
|
|
false_negative_decisions={},
|
|
)
|
|
|
|
assert result["adjudicated"]["false_positives"] == 0
|
|
|
|
def test_a_reference_gap_on_a_miss_stops_counting_against_recall(self) -> None:
|
|
result = ReviewedMetricsService.adjudicate(
|
|
matches=80,
|
|
false_positives=0,
|
|
false_negatives=20,
|
|
false_positive_decisions={},
|
|
false_negative_decisions=_counts(reference_gap_or_change=20),
|
|
)
|
|
|
|
# The reference held twenty footprints that are not there.
|
|
assert result["adjudicated"]["false_negatives"] == 0
|
|
assert result["adjudicated"]["recall"] == pytest.approx(1.0)
|
|
|
|
def test_an_uncertain_verdict_keeps_counting_against_the_model(self) -> None:
|
|
"""Not being able to judge is not evidence in the model's favour."""
|
|
|
|
result = ReviewedMetricsService.adjudicate(
|
|
matches=80,
|
|
false_positives=20,
|
|
false_negatives=0,
|
|
false_positive_decisions=_counts(uncertain=10, imagery_obscured_or_uncertain=10),
|
|
false_negative_decisions={},
|
|
)
|
|
|
|
assert result["adjudicated"]["false_positives"] == 20
|
|
|
|
|
|
class TestBounds:
|
|
def test_a_partly_reviewed_check_reports_an_interval(self) -> None:
|
|
result = ReviewedMetricsService.adjudicate(
|
|
matches=80,
|
|
false_positives=20,
|
|
false_negatives=0,
|
|
false_positive_decisions=_counts(reference_gap_or_change=10),
|
|
false_negative_decisions={},
|
|
)
|
|
|
|
# Ten unreviewed: pessimistically all model errors, optimistically none.
|
|
assert result["pessimistic"]["precision"] == pytest.approx(80 / 90)
|
|
assert result["optimistic"]["precision"] == pytest.approx(1.0)
|
|
assert result["review_complete"] is False
|
|
|
|
def test_a_fully_reviewed_check_collapses_the_interval(self) -> None:
|
|
result = ReviewedMetricsService.adjudicate(
|
|
matches=80,
|
|
false_positives=20,
|
|
false_negatives=5,
|
|
false_positive_decisions=_counts(reference_gap_or_change=12, confirmed_model_false_positive=8),
|
|
false_negative_decisions=_counts(confirmed_model_false_negative=5),
|
|
)
|
|
|
|
assert result["review_complete"] is True
|
|
assert result["pessimistic"]["precision"] == pytest.approx(result["optimistic"]["precision"])
|
|
assert result["adjudicated"]["precision"] == pytest.approx(80 / 88)
|
|
|
|
def test_an_unreviewed_check_reports_the_raw_numbers_unchanged(self) -> None:
|
|
result = ReviewedMetricsService.adjudicate(
|
|
matches=80,
|
|
false_positives=20,
|
|
false_negatives=10,
|
|
false_positive_decisions={},
|
|
false_negative_decisions={},
|
|
)
|
|
|
|
assert result["review_complete"] is False
|
|
assert result["adjudicated"]["precision"] == pytest.approx(result["raw"]["precision"])
|
|
assert result["adjudicated"]["recall"] == pytest.approx(result["raw"]["recall"])
|
|
|
|
def test_the_raw_score_is_always_reported_alongside(self) -> None:
|
|
result = ReviewedMetricsService.adjudicate(
|
|
matches=80,
|
|
false_positives=20,
|
|
false_negatives=0,
|
|
false_positive_decisions=_counts(reference_gap_or_change=20),
|
|
false_negative_decisions={},
|
|
)
|
|
|
|
assert result["raw"]["precision"] == pytest.approx(0.8)
|
|
assert result["adjudicated"]["precision"] == pytest.approx(1.0)
|
|
|
|
|
|
class TestEdges:
|
|
def test_a_check_without_findings_makes_no_claim(self) -> None:
|
|
result = ReviewedMetricsService.adjudicate(
|
|
matches=0,
|
|
false_positives=0,
|
|
false_negatives=0,
|
|
false_positive_decisions={},
|
|
false_negative_decisions={},
|
|
)
|
|
|
|
assert result["adjudicated"]["precision"] is None
|
|
assert result["adjudicated"]["recall"] is None
|
|
assert result["review_complete"] is True
|
|
|
|
def test_more_decisions_than_findings_cannot_invent_a_negative_count(self) -> None:
|
|
result = ReviewedMetricsService.adjudicate(
|
|
matches=10,
|
|
false_positives=2,
|
|
false_negatives=0,
|
|
false_positive_decisions=_counts(reference_gap_or_change=99),
|
|
false_negative_decisions={},
|
|
)
|
|
|
|
assert result["adjudicated"]["false_positives"] == 0
|
|
|
|
def test_an_unknown_decision_is_treated_as_no_judgement(self) -> None:
|
|
result = ReviewedMetricsService.adjudicate(
|
|
matches=10,
|
|
false_positives=5,
|
|
false_negatives=0,
|
|
false_positive_decisions=_counts(something_new_from_a_later_release=5),
|
|
false_negative_decisions={},
|
|
)
|
|
|
|
assert result["adjudicated"]["false_positives"] == 5
|
|
assert result["review_complete"] is False
|
|
|
|
|
|
class TestThroughTheReviewPanel:
|
|
"""The score the panel shows, not just the arithmetic behind it."""
|
|
|
|
def _quality_check(self, quality_check_id, project_id):
|
|
from app.models import QualityCheck
|
|
|
|
return QualityCheck(
|
|
id=quality_check_id,
|
|
project_id=project_id,
|
|
reference_dataset_id=__import__("uuid").uuid4(),
|
|
check_type="detections_vs_reference",
|
|
status="ok",
|
|
findings_json={
|
|
"matches": 80,
|
|
"false_positives": 20,
|
|
"false_negatives": 0,
|
|
"false_positive_evidence": [
|
|
{"candidate_feature_id": f"detection-{index}"} for index in range(20)
|
|
],
|
|
"false_negative_evidence": [],
|
|
},
|
|
)
|
|
|
|
def test_adjudicated_reference_gaps_raise_the_reported_precision(self) -> None:
|
|
import uuid
|
|
|
|
from app.models import DetectionReview, QualityCheck
|
|
from app.services.detection_review_service import DetectionReviewService
|
|
|
|
quality_check_id, project_id = uuid.uuid4(), uuid.uuid4()
|
|
quality_check = self._quality_check(quality_check_id, project_id)
|
|
reviews = [
|
|
DetectionReview(
|
|
id=uuid.uuid4(),
|
|
quality_check_id=quality_check_id,
|
|
evidence_role="false_positive",
|
|
evidence_feature_id=f"detection-{index}",
|
|
decision="reference_gap_or_change",
|
|
)
|
|
for index in range(12)
|
|
]
|
|
|
|
class _Query:
|
|
def __init__(self, rows):
|
|
self.rows = rows
|
|
|
|
def filter(self, *_args):
|
|
return self
|
|
|
|
def all(self):
|
|
return self.rows
|
|
|
|
class _Session:
|
|
def get(self, model, item_id):
|
|
return quality_check if model is QualityCheck and item_id == quality_check_id else None
|
|
|
|
def query(self, _model):
|
|
return _Query(reviews)
|
|
|
|
result = DetectionReviewService.list_reviews(
|
|
_Session(), project_id=project_id, quality_check_id=quality_check_id
|
|
)
|
|
metrics = result.summary.reviewed_metrics
|
|
|
|
assert metrics is not None
|
|
assert metrics["raw"]["precision"] == pytest.approx(0.8)
|
|
# Twelve of the twenty were the reference missing a building.
|
|
assert metrics["adjudicated"]["precision"] == pytest.approx(80 / 88)
|
|
assert metrics["review_complete"] is False
|
|
assert metrics["false_positive_breakdown"]["exonerated"] == 12
|
|
assert metrics["false_positive_breakdown"]["unreviewed"] == 8
|