Files
geointel/backend/app/services/detection_review_service.py
T
JensandClaude Opus 5 ff4a15aa74 let an operator's adjudication reach the score
The review vocabulary already separates a model error from a reference gap,
because the product's position is that official footprints are not
automatically perfect ground truth. Those verdicts were only counted. An
operator who inspected forty false positives and established that twelve are
buildings the reference simply lacks still saw a precision counting all forty
against the model — a number they had personally disproved, on the panel where
they disproved it.

Applying the verdicts gives an adjudicated score reported next to the raw one,
so nothing is quietly improved. Not being able to judge is not evidence in the
model's favour, so uncertain and obscured verdicts keep counting, as does a
decision from a later release that this runtime does not recognise.

Because part of the evidence is usually still unreviewed, the honest form is an
interval rather than a single corrected number: pessimistic assumes every
unreviewed finding is a model error, optimistic assumes none is, and the
headline equals the pessimistic reading so a partly reviewed check never
presents as a settled one.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-22 20:43:22 +02:00

295 lines
13 KiB
Python

from __future__ import annotations
from collections import Counter
from typing import Any
from uuid import UUID
from sqlalchemy.orm import Session
from app.core.errors import AppError
from app.services.reviewed_metrics_service import ReviewedMetricsService
from app.models import Detection, DetectionReview, QualityCheck, VectorFeature
from app.schemas.detection_review import (
DetectionReviewList,
DetectionReviewRead,
DetectionReviewSummary,
DetectionReviewUpsert,
)
class DetectionReviewService:
ALLOWED_DECISIONS = {
"false_positive": {
"confirmed_model_false_positive",
"reference_gap_or_change",
"qa_alignment_mismatch",
"uncertain",
"unreviewed",
},
"false_negative": {
"confirmed_model_false_negative",
"reference_gap_or_change",
"qa_alignment_mismatch",
"imagery_obscured_or_uncertain",
"uncertain",
"unreviewed",
},
}
@staticmethod
def _quality_check(db: Session, project_id: UUID, quality_check_id: UUID) -> QualityCheck:
quality_check = db.get(QualityCheck, quality_check_id)
if not quality_check or quality_check.project_id != project_id:
raise AppError(code="QUALITY_CHECK_NOT_FOUND", message="Quality check not found", status_code=404)
if quality_check.check_type != "detections_vs_reference":
raise AppError(
code="DETECTION_REVIEW_UNSUPPORTED",
message="Only persisted detection-versus-reference quality checks can be reviewed",
status_code=422,
)
return quality_check
@staticmethod
def _evidence_items(quality_check: QualityCheck) -> list[dict[str, str]]:
findings = quality_check.findings_json or {}
items: list[dict[str, str]] = []
for role, key, id_key in (
("false_positive", "false_positive_evidence", "candidate_feature_id"),
("false_negative", "false_negative_evidence", "reference_feature_id"),
):
evidence_rows = findings.get(key)
if not isinstance(evidence_rows, list):
continue
for evidence in evidence_rows:
if not isinstance(evidence, dict):
continue
value = str(evidence.get(id_key) or "").strip()
if value:
items.append({"evidence_role": role, "evidence_feature_id": value})
return items
@staticmethod
def _uuid(value: str) -> UUID | None:
try:
return UUID(value)
except (TypeError, ValueError):
return None
@staticmethod
def _review_index(db: Session, quality_check_id: UUID) -> dict[tuple[str, str], DetectionReview]:
rows = db.query(DetectionReview).filter(DetectionReview.quality_check_id == quality_check_id).all()
return {(row.evidence_role, row.evidence_feature_id): row for row in rows}
@staticmethod
def _summary(
evidence: list[dict[str, str]],
reviews: dict[tuple[str, str], DetectionReview],
quality_check: QualityCheck | None = None,
) -> DetectionReviewSummary:
evidence_keys = {(item["evidence_role"], item["evidence_feature_id"]) for item in evidence}
decisions = Counter(
reviews[key].decision if key in reviews else "unreviewed"
for key in evidence_keys
)
reviewed = sum(count for decision, count in decisions.items() if decision != "unreviewed")
false_positive_total = sum(1 for item in evidence if item["evidence_role"] == "false_positive")
false_negative_total = sum(1 for item in evidence if item["evidence_role"] == "false_negative")
return DetectionReviewSummary(
total=len(evidence),
reviewed=reviewed,
remaining=max(len(evidence) - reviewed, 0),
false_positive_total=false_positive_total,
false_negative_total=false_negative_total,
decision_counts=dict(sorted(decisions.items())),
reviewed_metrics=DetectionReviewService._reviewed_metrics(evidence_keys, reviews, quality_check),
)
@staticmethod
def _reviewed_metrics(
evidence_keys: set[tuple[str, str]],
reviews: dict[tuple[str, str], DetectionReview],
quality_check: QualityCheck | None,
) -> dict | None:
"""The score with the operator's verdicts applied.
Without this the panel shows a precision the operator has already
disproved: a false positive adjudicated as a reference gap is not the
model's error, and the raw number keeps counting it as one.
"""
if quality_check is None:
return None
findings = quality_check.findings_json if isinstance(quality_check.findings_json, dict) else {}
matches = findings.get("matches")
false_positives = findings.get("false_positives")
false_negatives = findings.get("false_negatives")
if not all(isinstance(value, int) for value in (matches, false_positives, false_negatives)):
return None
per_role: dict[str, Counter] = {"false_positive": Counter(), "false_negative": Counter()}
for role, feature_id in evidence_keys:
review = reviews.get((role, feature_id))
if review is not None and role in per_role:
per_role[role][review.decision] += 1
return ReviewedMetricsService.adjudicate(
matches=int(matches),
false_positives=int(false_positives),
false_negatives=int(false_negatives),
false_positive_decisions=dict(per_role["false_positive"]),
false_negative_decisions=dict(per_role["false_negative"]),
)
@staticmethod
def _read_item(
db: Session,
quality_check: QualityCheck,
evidence: dict[str, str],
review: DetectionReview | None,
) -> DetectionReviewRead:
role = evidence["evidence_role"]
feature_id = evidence["evidence_feature_id"]
feature_uuid = DetectionReviewService._uuid(feature_id)
detection = db.get(Detection, feature_uuid) if role == "false_positive" and feature_uuid else None
reference = db.get(VectorFeature, feature_uuid) if role == "false_negative" and feature_uuid else None
return DetectionReviewRead(
id=review.id if review else None,
project_id=quality_check.project_id,
quality_check_id=quality_check.id,
analysis_run_id=quality_check.analysis_run_id,
evidence_role=role,
evidence_feature_id=feature_id,
detection_id=detection.id if detection else review.detection_id if review else None,
reference_feature_id=reference.id if reference else review.reference_feature_id if review else None,
decision=review.decision if review else "unreviewed",
notes=review.notes if review else None,
reviewed_by=review.reviewed_by if review else None,
confidence=detection.confidence if detection else None,
class_name=(detection.class_name if detection else reference.feature_class if reference else None),
source_tile_path=detection.source_tile_path if detection else None,
created_at=review.created_at if review else None,
updated_at=review.updated_at if review else None,
)
@staticmethod
def list_reviews(
db: Session,
*,
project_id: UUID,
quality_check_id: UUID,
evidence_role: str | None = None,
decision: str | None = None,
reviewed: bool | None = None,
limit: int = 50,
offset: int = 0,
) -> DetectionReviewList:
quality_check = DetectionReviewService._quality_check(db, project_id, quality_check_id)
evidence = DetectionReviewService._evidence_items(quality_check)
reviews = DetectionReviewService._review_index(db, quality_check_id)
filtered = [item for item in evidence if evidence_role is None or item["evidence_role"] == evidence_role]
if decision is not None:
filtered = [
item
for item in filtered
if (reviews.get((item["evidence_role"], item["evidence_feature_id"])).decision
if reviews.get((item["evidence_role"], item["evidence_feature_id"]))
else "unreviewed")
== decision
]
if reviewed is not None:
filtered = [
item
for item in filtered
if (
(reviews.get((item["evidence_role"], item["evidence_feature_id"])).decision
if reviews.get((item["evidence_role"], item["evidence_feature_id"]))
else "unreviewed")
!= "unreviewed"
)
== reviewed
]
page = filtered[offset : offset + limit]
return DetectionReviewList(
items=[
DetectionReviewService._read_item(
db,
quality_check,
item,
reviews.get((item["evidence_role"], item["evidence_feature_id"])),
)
for item in page
],
total=len(filtered),
limit=limit,
offset=offset,
summary=DetectionReviewService._summary(evidence, reviews, quality_check),
)
@staticmethod
def upsert_review(
db: Session,
*,
project_id: UUID,
quality_check_id: UUID,
payload: DetectionReviewUpsert,
) -> DetectionReviewRead:
quality_check = DetectionReviewService._quality_check(db, project_id, quality_check_id)
if payload.decision not in DetectionReviewService.ALLOWED_DECISIONS[payload.evidence_role]:
raise AppError(
code="INVALID_DETECTION_REVIEW_DECISION",
message="The review decision is not valid for this evidence role",
details={"evidence_role": payload.evidence_role, "decision": payload.decision},
status_code=422,
)
evidence = DetectionReviewService._evidence_items(quality_check)
evidence_key = (payload.evidence_role, payload.evidence_feature_id)
if evidence_key not in {(item["evidence_role"], item["evidence_feature_id"]) for item in evidence}:
raise AppError(
code="DETECTION_REVIEW_EVIDENCE_NOT_FOUND",
message="The evidence feature does not belong to this quality check",
status_code=404,
)
feature_uuid = DetectionReviewService._uuid(payload.evidence_feature_id)
detection = db.get(Detection, feature_uuid) if payload.evidence_role == "false_positive" and feature_uuid else None
reference = db.get(VectorFeature, feature_uuid) if payload.evidence_role == "false_negative" and feature_uuid else None
if payload.evidence_role == "false_positive" and (not detection or detection.analysis_run_id != quality_check.analysis_run_id):
raise AppError(code="DETECTION_REVIEW_EVIDENCE_NOT_FOUND", message="Persisted detection evidence was not found", status_code=404)
if payload.evidence_role == "false_negative" and (not reference or reference.dataset_id != quality_check.reference_dataset_id):
raise AppError(code="DETECTION_REVIEW_EVIDENCE_NOT_FOUND", message="Persisted reference evidence was not found", status_code=404)
review = (
db.query(DetectionReview)
.filter(
DetectionReview.quality_check_id == quality_check_id,
DetectionReview.evidence_role == payload.evidence_role,
DetectionReview.evidence_feature_id == payload.evidence_feature_id,
)
.first()
)
if review is None:
review = DetectionReview(
project_id=project_id,
quality_check_id=quality_check_id,
analysis_run_id=quality_check.analysis_run_id,
evidence_role=payload.evidence_role,
evidence_feature_id=payload.evidence_feature_id,
detection_id=detection.id if detection else None,
reference_feature_id=reference.id if reference else None,
decision=payload.decision,
notes=payload.notes.strip() if payload.notes and payload.notes.strip() else None,
reviewed_by=payload.reviewed_by.strip(),
)
else:
review.decision = payload.decision
review.notes = payload.notes.strip() if payload.notes and payload.notes.strip() else None
review.reviewed_by = payload.reviewed_by.strip()
db.add(review)
db.commit()
db.refresh(review)
return DetectionReviewService._read_item(
db,
quality_check,
{"evidence_role": payload.evidence_role, "evidence_feature_id": payload.evidence_feature_id},
review,
)