Persist QA feature evidence
GeoIntel CI / docs-smoke (push) Has been cancelled
GeoIntel CI / contract-smoke (push) Has been cancelled

This commit is contained in:
Codex
2026-06-25 02:57:50 +02:00
parent 49e58cd1c7
commit 368c1ad738
15 changed files with 425 additions and 81 deletions
+8
View File
@@ -7,6 +7,14 @@
# Changelog # Changelog
## Sprint 111 QA feature evidence persistence (2026-06-25)
- Added feature-level QA evidence to dataset, detection and segmentation QA matching.
- Persisted matched feature ids, false-positive feature ids and false-negative feature ids inside `quality_checks.findings_json`.
- Extended QA/QC drilldown with compact matched/false-positive/false-negative feature id lists beside the existing metrics and raw findings JSON.
- Updated API contracts to document `match_evidence`, `false_positive_evidence` and `false_negative_evidence`.
- No migration, new table, provider fetching, AI behavior or new product domain was introduced.
## Sprint 110 Map QA evidence drilldown (2026-06-25) ## Sprint 110 Map QA evidence drilldown (2026-06-25)
- Extended the Map workspace QA/QC shortcut with inline evidence after comparing a saved derived selection dataset. - Extended the Map workspace QA/QC shortcut with inline evidence after comparing a saved derived selection dataset.
+3
View File
@@ -59,6 +59,9 @@ def compare_candidate_with_reference(
"warnings": result_json.get("warnings", []), "warnings": result_json.get("warnings", []),
"unsupported_geometry": result_json.get("unsupported_geometry", False), "unsupported_geometry": result_json.get("unsupported_geometry", False),
"unsupported_geometries": result_json.get("unsupported_geometries", []), "unsupported_geometries": result_json.get("unsupported_geometries", []),
"match_evidence": result_json.get("match_evidence", []),
"false_positive_evidence": result_json.get("false_positive_evidence", []),
"false_negative_evidence": result_json.get("false_negative_evidence", []),
}, },
metrics={ metrics={
"precision": result_json.get("precision"), "precision": result_json.get("precision"),
+3
View File
@@ -28,6 +28,9 @@ class QaProviderComparisonResult(BaseModel):
iou_threshold: float iou_threshold: float
unsupported_geometry: bool = False unsupported_geometry: bool = False
unsupported_geometries: list[str] = Field(default_factory=list) unsupported_geometries: list[str] = Field(default_factory=list)
match_evidence: list[dict] = Field(default_factory=list)
false_positive_evidence: list[dict] = Field(default_factory=list)
false_negative_evidence: list[dict] = Field(default_factory=list)
generated_at: datetime generated_at: datetime
+22 -16
View File
@@ -287,18 +287,18 @@ class DetectionService:
candidate_geometries = [({"id": str(row.id), "class_name": row.class_name}, to_shape(row.geometry)) for row in detections] candidate_geometries = [({"id": str(row.id), "class_name": row.class_name}, to_shape(row.geometry)) for row in detections]
reference_geometries = [({"id": str(row.id), "feature_class": row.feature_class}, to_shape(row.geometry)) for row in references] reference_geometries = [({"id": str(row.id), "feature_class": row.feature_class}, to_shape(row.geometry)) for row in references]
matches, false_positives, false_negatives, match_iou_values, warnings, unsupported = QaService._match_io_u_metrics( evidence = QaService._match_io_u_evidence(
candidate_geometries, candidate_geometries,
reference_geometries, reference_geometries,
iou_threshold, iou_threshold,
) )
mean_iou = None if not match_iou_values else sum(match_iou_values) / len(match_iou_values) mean_iou = None if not evidence.match_iou_values else sum(evidence.match_iou_values) / len(evidence.match_iou_values)
precision = matches / (matches + false_positives) if matches + false_positives > 0 else None precision = evidence.matches / (evidence.matches + evidence.false_positives) if evidence.matches + evidence.false_positives > 0 else None
recall = matches / (matches + false_negatives) if matches + false_negatives > 0 else None recall = evidence.matches / (evidence.matches + evidence.false_negatives) if evidence.matches + evidence.false_negatives > 0 else None
f1_score = None f1_score = None
if precision is not None and recall is not None: if precision is not None and recall is not None:
f1_score = (2 * precision * recall) / (precision + recall) if precision + recall > 0 else 0.0 f1_score = (2 * precision * recall) / (precision + recall) if precision + recall > 0 else 0.0
status = "unsupported" if unsupported else "ok" status = "unsupported" if evidence.unsupported else "ok"
quality_check = QualityService.persist_quality_check( quality_check = QualityService.persist_quality_check(
db=db, db=db,
project_id=run.project_id, project_id=run.project_id,
@@ -316,19 +316,22 @@ class DetectionService:
"min_confidence": min_confidence, "min_confidence": min_confidence,
}, },
findings={ findings={
"matches": matches, "matches": evidence.matches,
"false_positives": false_positives, "false_positives": evidence.false_positives,
"false_negatives": false_negatives, "false_negatives": evidence.false_negatives,
"warnings": warnings, "warnings": evidence.warnings,
"unsupported_geometry": unsupported, "unsupported_geometry": evidence.unsupported,
"match_evidence": evidence.match_evidence,
"false_positive_evidence": evidence.false_positive_evidence,
"false_negative_evidence": evidence.false_negative_evidence,
}, },
metrics={ metrics={
"precision": precision, "precision": precision,
"recall": recall, "recall": recall,
"f1": f1_score, "f1": f1_score,
"mean_iou": mean_iou, "mean_iou": mean_iou,
"false_positive_count": false_positives, "false_positive_count": evidence.false_positives,
"false_negative_count": false_negatives, "false_negative_count": evidence.false_negatives,
}, },
) )
return { return {
@@ -338,15 +341,18 @@ class DetectionService:
"reference_dataset_id": str(reference_dataset_id), "reference_dataset_id": str(reference_dataset_id),
"candidate_feature_count": len(candidate_geometries), "candidate_feature_count": len(candidate_geometries),
"reference_feature_count": len(reference_geometries), "reference_feature_count": len(reference_geometries),
"matches": matches, "matches": evidence.matches,
"false_positives": false_positives, "false_positives": evidence.false_positives,
"false_negatives": false_negatives, "false_negatives": evidence.false_negatives,
"precision": precision, "precision": precision,
"recall": recall, "recall": recall,
"f1_score": f1_score, "f1_score": f1_score,
"mean_iou": mean_iou, "mean_iou": mean_iou,
"iou_threshold": iou_threshold, "iou_threshold": iou_threshold,
"warnings": warnings, "warnings": evidence.warnings,
"match_evidence": evidence.match_evidence,
"false_positive_evidence": evidence.false_positive_evidence,
"false_negative_evidence": evidence.false_negative_evidence,
} }
@staticmethod @staticmethod
+106 -37
View File
@@ -1,5 +1,6 @@
from __future__ import annotations from __future__ import annotations
from dataclasses import dataclass, field
from datetime import datetime, timezone from datetime import datetime, timezone
from typing import Any from typing import Any
from uuid import UUID from uuid import UUID
@@ -16,6 +17,19 @@ from app.schemas.qa import QaProviderComparisonResult
from app.services.vector_operations_service import VectorOperationsService from app.services.vector_operations_service import VectorOperationsService
@dataclass
class QaMatchEvidence:
matches: int = 0
false_positives: int = 0
false_negatives: int = 0
match_iou_values: list[float] = field(default_factory=list)
warnings: list[str] = field(default_factory=list)
unsupported: bool = False
match_evidence: list[dict[str, Any]] = field(default_factory=list)
false_positive_evidence: list[dict[str, Any]] = field(default_factory=list)
false_negative_evidence: list[dict[str, Any]] = field(default_factory=list)
def _extract_crs_warnings(source_dataset: Dataset, reference_dataset: Dataset) -> list[str]: def _extract_crs_warnings(source_dataset: Dataset, reference_dataset: Dataset) -> list[str]:
warnings: list[str] = [] warnings: list[str] = []
for dataset, label in ((source_dataset, "candidate"), (reference_dataset, "reference")): for dataset, label in ((source_dataset, "candidate"), (reference_dataset, "reference")):
@@ -92,17 +106,34 @@ class QaService:
return area_geometry return area_geometry
@staticmethod @staticmethod
def _match_io_u_metrics( def _feature_identifier(feature: dict[str, Any], fallback_prefix: str, index: int) -> str:
feature_id = feature.get("id")
if feature_id is not None:
return str(feature_id)
properties = feature.get("properties")
if isinstance(properties, dict):
for key in ("vector_feature_id", "source_feature_id", "detection_id", "segmentation_id", "id", "name"):
value = properties.get(key)
if value is not None:
return str(value)
for key in ("vector_feature_id", "source_feature_id", "detection_id", "segmentation_id", "class_name", "feature_class"):
value = feature.get(key)
if value is not None:
return str(value)
return f"{fallback_prefix}-{index + 1}"
@staticmethod
def _match_io_u_evidence(
source_geometries: list[tuple[dict[str, Any], BaseGeometry]], source_geometries: list[tuple[dict[str, Any], BaseGeometry]],
reference_geometries: list[tuple[dict[str, Any], BaseGeometry]], reference_geometries: list[tuple[dict[str, Any], BaseGeometry]],
iou_threshold: float, iou_threshold: float,
) -> tuple[int, int, int, list[float], list[str], bool]: ) -> QaMatchEvidence:
source_supported = [ source_supported = [
(feature, geom) for feature, geom in source_geometries if geom.geom_type in QaService.SUPPORTED_GEOMETRY_TYPES (index, feature, geom) for index, (feature, geom) in enumerate(source_geometries) if geom.geom_type in QaService.SUPPORTED_GEOMETRY_TYPES
] ]
reference_supported = [ reference_supported = [
(feature, geom) (index, feature, geom)
for feature, geom in reference_geometries for index, (feature, geom) in enumerate(reference_geometries)
if geom.geom_type in QaService.SUPPORTED_GEOMETRY_TYPES if geom.geom_type in QaService.SUPPORTED_GEOMETRY_TYPES
] ]
@@ -114,29 +145,35 @@ class QaService:
} }
) )
if not source_supported or not reference_supported: if not source_supported or not reference_supported:
return ( return QaMatchEvidence(
0, false_positives=len(source_supported),
len(source_supported), false_negatives=len(reference_supported),
len(reference_supported), warnings=[f"Unsupported geometry types: {unsupported}"] if unsupported else [],
[], unsupported=True,
[f"Unsupported geometry types: {unsupported}"] if unsupported else [], false_positive_evidence=[
True, {"candidate_feature_id": QaService._feature_identifier(feature, "candidate", source_index)}
for source_index, feature, _ in source_supported
],
false_negative_evidence=[
{"reference_feature_id": QaService._feature_identifier(feature, "reference", reference_index)}
for reference_index, feature, _ in reference_supported
],
) )
unmatched_reference_indices = set(range(len(reference_supported))) unmatched_reference_indices = set(range(len(reference_supported)))
matches = 0 evidence = QaMatchEvidence(warnings=[f"Unsupported geometry types: {unsupported}"] if unsupported else [], unsupported=bool(unsupported))
match_iou_values: list[float] = []
false_positives = 0
for _, source_geom in source_supported: for source_index, source_feature, source_geom in source_supported:
source_feature_id = QaService._feature_identifier(source_feature, "candidate", source_index)
if source_geom.area <= 0: if source_geom.area <= 0:
false_positives += 1 evidence.false_positives += 1
evidence.false_positive_evidence.append({"candidate_feature_id": source_feature_id})
continue continue
best_iou = 0.0 best_iou = 0.0
best_index = None best_index = None
for reference_index in list(unmatched_reference_indices): for reference_index in list(unmatched_reference_indices):
_, reference_geom = reference_supported[reference_index] _, _, reference_geom = reference_supported[reference_index]
if reference_geom.area <= 0: if reference_geom.area <= 0:
unmatched_reference_indices.discard(reference_index) unmatched_reference_indices.discard(reference_index)
continue continue
@@ -161,16 +198,45 @@ class QaService:
best_index = reference_index best_index = reference_index
if best_index is not None and best_iou >= iou_threshold: if best_index is not None and best_iou >= iou_threshold:
matches += 1 reference_original_index, reference_feature, _ = reference_supported[best_index]
match_iou_values.append(best_iou) evidence.matches += 1
evidence.match_iou_values.append(best_iou)
evidence.match_evidence.append(
{
"candidate_feature_id": source_feature_id,
"reference_feature_id": QaService._feature_identifier(reference_feature, "reference", reference_original_index),
"iou": best_iou,
}
)
unmatched_reference_indices.discard(best_index) unmatched_reference_indices.discard(best_index)
else: else:
false_positives += 1 evidence.false_positives += 1
evidence.false_positive_evidence.append({"candidate_feature_id": source_feature_id})
false_negatives = len(unmatched_reference_indices) evidence.false_negatives = len(unmatched_reference_indices)
warnings: list[str] = [f"Unsupported geometry types: {unsupported}"] if unsupported else [] for reference_index in sorted(unmatched_reference_indices):
reference_original_index, reference_feature, _ = reference_supported[reference_index]
evidence.false_negative_evidence.append(
{"reference_feature_id": QaService._feature_identifier(reference_feature, "reference", reference_original_index)}
)
return matches, false_positives, false_negatives, match_iou_values, warnings, bool(unsupported) return evidence
@staticmethod
def _match_io_u_metrics(
source_geometries: list[tuple[dict[str, Any], BaseGeometry]],
reference_geometries: list[tuple[dict[str, Any], BaseGeometry]],
iou_threshold: float,
) -> tuple[int, int, int, list[float], list[str], bool]:
evidence = QaService._match_io_u_evidence(source_geometries, reference_geometries, iou_threshold)
return (
evidence.matches,
evidence.false_positives,
evidence.false_negatives,
evidence.match_iou_values,
evidence.warnings,
evidence.unsupported,
)
@staticmethod @staticmethod
def compare_candidate_with_reference( def compare_candidate_with_reference(
@@ -206,7 +272,7 @@ class QaService:
candidate_geometries = QaService._apply_area_filter(candidate_geometries, area_geometry, dataset_id=candidate_dataset.id) candidate_geometries = QaService._apply_area_filter(candidate_geometries, area_geometry, dataset_id=candidate_dataset.id)
reference_geometries = QaService._apply_area_filter(reference_geometries, area_geometry, dataset_id=reference_dataset.id) reference_geometries = QaService._apply_area_filter(reference_geometries, area_geometry, dataset_id=reference_dataset.id)
matches, false_positives, false_negatives, match_iou_values, warnings, unsupported = QaService._match_io_u_metrics( evidence = QaService._match_io_u_evidence(
candidate_geometries, candidate_geometries,
reference_geometries, reference_geometries,
iou_threshold, iou_threshold,
@@ -214,35 +280,38 @@ class QaService:
candidate_feature_count = len(candidate_payload.get("features", [])) if isinstance(candidate_payload, dict) else 0 candidate_feature_count = len(candidate_payload.get("features", [])) if isinstance(candidate_payload, dict) else 0
reference_feature_count = len(reference_payload.get("features", [])) if isinstance(reference_payload, dict) else 0 reference_feature_count = len(reference_payload.get("features", [])) if isinstance(reference_payload, dict) else 0
mean_iou = None if not match_iou_values else sum(match_iou_values) / len(match_iou_values) mean_iou = None if not evidence.match_iou_values else sum(evidence.match_iou_values) / len(evidence.match_iou_values)
precision = None precision = None
if matches + false_positives > 0: if evidence.matches + evidence.false_positives > 0:
precision = matches / (matches + false_positives) precision = evidence.matches / (evidence.matches + evidence.false_positives)
recall = None recall = None
if matches + false_negatives > 0: if evidence.matches + evidence.false_negatives > 0:
recall = matches / (matches + false_negatives) recall = evidence.matches / (evidence.matches + evidence.false_negatives)
f1_score = None f1_score = None
if precision is not None and recall is not None and precision + recall > 0: if precision is not None and recall is not None and precision + recall > 0:
f1_score = (2 * precision * recall) / (precision + recall) f1_score = (2 * precision * recall) / (precision + recall)
status = "unsupported" if unsupported else "ok" status = "unsupported" if evidence.unsupported else "ok"
return QaProviderComparisonResult( return QaProviderComparisonResult(
status=status, status=status,
warnings=_extract_crs_warnings(candidate_dataset, reference_dataset) + warnings, warnings=_extract_crs_warnings(candidate_dataset, reference_dataset) + evidence.warnings,
candidate_feature_count=candidate_feature_count, candidate_feature_count=candidate_feature_count,
reference_feature_count=reference_feature_count, reference_feature_count=reference_feature_count,
matches=matches, matches=evidence.matches,
false_positives=false_positives, false_positives=evidence.false_positives,
false_negatives=false_negatives, false_negatives=evidence.false_negatives,
precision=precision, precision=precision,
recall=recall, recall=recall,
f1_score=f1_score, f1_score=f1_score,
mean_iou=mean_iou, mean_iou=mean_iou,
iou_threshold=iou_threshold, iou_threshold=iou_threshold,
unsupported_geometry=unsupported, unsupported_geometry=evidence.unsupported,
unsupported_geometries=warnings, unsupported_geometries=evidence.warnings,
match_evidence=evidence.match_evidence,
false_positive_evidence=evidence.false_positive_evidence,
false_negative_evidence=evidence.false_negative_evidence,
generated_at=datetime.now(timezone.utc), generated_at=datetime.now(timezone.utc),
) )
+22 -16
View File
@@ -251,18 +251,18 @@ class SegmentationService:
candidate_geometries = [({"id": str(row.id), "class_name": row.class_name}, to_shape(row.geometry)) for row in segmentations] candidate_geometries = [({"id": str(row.id), "class_name": row.class_name}, to_shape(row.geometry)) for row in segmentations]
reference_geometries = [({"id": str(row.id), "feature_class": row.feature_class}, to_shape(row.geometry)) for row in references] reference_geometries = [({"id": str(row.id), "feature_class": row.feature_class}, to_shape(row.geometry)) for row in references]
matches, false_positives, false_negatives, match_iou_values, warnings, unsupported = QaService._match_io_u_metrics( evidence = QaService._match_io_u_evidence(
candidate_geometries, candidate_geometries,
reference_geometries, reference_geometries,
iou_threshold, iou_threshold,
) )
mean_iou = None if not match_iou_values else sum(match_iou_values) / len(match_iou_values) mean_iou = None if not evidence.match_iou_values else sum(evidence.match_iou_values) / len(evidence.match_iou_values)
precision = matches / (matches + false_positives) if matches + false_positives > 0 else None precision = evidence.matches / (evidence.matches + evidence.false_positives) if evidence.matches + evidence.false_positives > 0 else None
recall = matches / (matches + false_negatives) if matches + false_negatives > 0 else None recall = evidence.matches / (evidence.matches + evidence.false_negatives) if evidence.matches + evidence.false_negatives > 0 else None
f1_score = None f1_score = None
if precision is not None and recall is not None: if precision is not None and recall is not None:
f1_score = (2 * precision * recall) / (precision + recall) if precision + recall > 0 else 0.0 f1_score = (2 * precision * recall) / (precision + recall) if precision + recall > 0 else 0.0
status = "unsupported" if unsupported else "ok" status = "unsupported" if evidence.unsupported else "ok"
quality_check = QualityService.persist_quality_check( quality_check = QualityService.persist_quality_check(
db=db, db=db,
project_id=run.project_id, project_id=run.project_id,
@@ -280,19 +280,22 @@ class SegmentationService:
"min_confidence": min_confidence, "min_confidence": min_confidence,
}, },
findings={ findings={
"matches": matches, "matches": evidence.matches,
"false_positives": false_positives, "false_positives": evidence.false_positives,
"false_negatives": false_negatives, "false_negatives": evidence.false_negatives,
"warnings": warnings, "warnings": evidence.warnings,
"unsupported_geometry": unsupported, "unsupported_geometry": evidence.unsupported,
"match_evidence": evidence.match_evidence,
"false_positive_evidence": evidence.false_positive_evidence,
"false_negative_evidence": evidence.false_negative_evidence,
}, },
metrics={ metrics={
"precision": precision, "precision": precision,
"recall": recall, "recall": recall,
"f1": f1_score, "f1": f1_score,
"mean_iou": mean_iou, "mean_iou": mean_iou,
"false_positive_count": false_positives, "false_positive_count": evidence.false_positives,
"false_negative_count": false_negatives, "false_negative_count": evidence.false_negatives,
}, },
) )
return { return {
@@ -302,15 +305,18 @@ class SegmentationService:
"reference_dataset_id": str(reference_dataset_id), "reference_dataset_id": str(reference_dataset_id),
"candidate_feature_count": len(candidate_geometries), "candidate_feature_count": len(candidate_geometries),
"reference_feature_count": len(reference_geometries), "reference_feature_count": len(reference_geometries),
"matches": matches, "matches": evidence.matches,
"false_positives": false_positives, "false_positives": evidence.false_positives,
"false_negatives": false_negatives, "false_negatives": evidence.false_negatives,
"precision": precision, "precision": precision,
"recall": recall, "recall": recall,
"f1_score": f1_score, "f1_score": f1_score,
"mean_iou": mean_iou, "mean_iou": mean_iou,
"iou_threshold": iou_threshold, "iou_threshold": iou_threshold,
"warnings": warnings, "warnings": evidence.warnings,
"match_evidence": evidence.match_evidence,
"false_positive_evidence": evidence.false_positive_evidence,
"false_negative_evidence": evidence.false_negative_evidence,
} }
@staticmethod @staticmethod
+73 -10
View File
@@ -22,23 +22,30 @@ class FakeSession:
return None return None
def _feature(feature_id: str, coordinates: list[list[list[float]]]) -> dict:
return {
"type": "Feature",
"id": feature_id,
"properties": {"source_feature_id": feature_id},
"geometry": {
"type": "Polygon",
"coordinates": coordinates,
},
}
def _write_dataset(path: Path, coordinates: list[list[list[float]]]) -> None: def _write_dataset(path: Path, coordinates: list[list[list[float]]]) -> None:
payload = { payload = {
"type": "FeatureCollection", "type": "FeatureCollection",
"features": [ "features": [_feature("feature-1", coordinates)],
{
"type": "Feature",
"properties": {},
"geometry": {
"type": "Polygon",
"coordinates": coordinates,
},
},
],
} }
path.write_text(json.dumps(payload), encoding="utf-8") path.write_text(json.dumps(payload), encoding="utf-8")
def _write_features(path: Path, features: list[dict]) -> None:
path.write_text(json.dumps({"type": "FeatureCollection", "features": features}), encoding="utf-8")
def test_qa_compare_candidate_with_reference_returns_metrics(tmp_path) -> None: def test_qa_compare_candidate_with_reference_returns_metrics(tmp_path) -> None:
project_id = uuid4() project_id = uuid4()
candidate_id = uuid4() candidate_id = uuid4()
@@ -87,6 +94,62 @@ def test_qa_compare_candidate_with_reference_returns_metrics(tmp_path) -> None:
assert result.f1_score == 1.0 assert result.f1_score == 1.0
def test_qa_compare_candidate_with_reference_returns_feature_level_evidence(tmp_path) -> None:
project_id = uuid4()
candidate_id = uuid4()
reference_id = uuid4()
candidate_path = tmp_path / "candidate.geojson"
reference_path = tmp_path / "reference.geojson"
matched_candidate = [[[4.0, 51.0], [4.1, 51.0], [4.1, 51.1], [4.0, 51.1], [4.0, 51.0]]]
matched_reference = [[[4.0, 51.0], [4.1, 51.0], [4.1, 51.1], [4.0, 51.1], [4.0, 51.0]]]
false_positive = [[[4.5, 51.5], [4.6, 51.5], [4.6, 51.6], [4.5, 51.6], [4.5, 51.5]]]
false_negative = [[[4.8, 51.8], [4.9, 51.8], [4.9, 51.9], [4.8, 51.9], [4.8, 51.8]]]
_write_features(candidate_path, [_feature("candidate-match", matched_candidate), _feature("candidate-extra", false_positive)])
_write_features(reference_path, [_feature("reference-match", matched_reference), _feature("reference-missing", false_negative)])
candidate = Dataset(
id=candidate_id,
project_id=project_id,
name="candidate.geojson",
dataset_type="vector",
source="test",
storage_path=str(candidate_path),
crs="EPSG:4326",
metadata_json={"crs_assumed": False},
)
reference = Dataset(
id=reference_id,
project_id=project_id,
name="reference.geojson",
dataset_type="vector",
source="test",
storage_path=str(reference_path),
crs="EPSG:4326",
metadata_json={"crs_assumed": False},
)
result = QaService.compare_candidate_with_reference(
db=FakeSession([candidate, reference]),
project_id=project_id,
candidate_dataset_id=candidate_id,
reference_dataset_id=reference_id,
iou_threshold=0.5,
)
assert result.matches == 1
assert result.false_positives == 1
assert result.false_negatives == 1
assert result.match_evidence == [
{
"candidate_feature_id": "candidate-match",
"reference_feature_id": "reference-match",
"iou": 1.0,
}
]
assert result.false_positive_evidence == [{"candidate_feature_id": "candidate-extra"}]
assert result.false_negative_evidence == [{"reference_feature_id": "reference-missing"}]
def test_dataset_reference_metadata_migration_declares_required_columns() -> None: def test_dataset_reference_metadata_migration_declares_required_columns() -> None:
migration_path = Path(__file__).parents[1] / "alembic" / "versions" / "202606120001_add_dataset_reference_metadata.py" migration_path = Path(__file__).parents[1] / "alembic" / "versions" / "202606120001_add_dataset_reference_metadata.py"
migration_text = migration_path.read_text(encoding="utf-8") migration_text = migration_path.read_text(encoding="utf-8")
@@ -0,0 +1,39 @@
from pathlib import Path
ROOT = Path(__file__).resolve().parents[2]
def read_text(relative_path: str) -> str:
return (ROOT / relative_path).read_text(encoding="utf-8")
def test_qa_feature_evidence_contract_is_documented_and_rendered() -> None:
schema = read_text("backend/app/schemas/qa.py")
quality_panel = read_text("frontend/src/components/quality/QualityResultsPanel.tsx")
api_contracts = read_text("docs/API_CONTRACTS.md")
for field_name in ("match_evidence", "false_positive_evidence", "false_negative_evidence"):
assert field_name in schema
assert field_name in quality_panel
assert field_name in api_contracts
assert "Feature-level QA/QC evidence" in quality_panel
assert "Matched feature ids" in quality_panel
assert "False positive feature ids" in quality_panel
assert "False negative feature ids" in quality_panel
assert "evidenceLabel" in quality_panel
def test_qa_services_persist_feature_evidence_without_new_migrations() -> None:
qa_route = read_text("backend/app/api/routes/qa.py")
detection_service = read_text("backend/app/services/detection_service.py")
segmentation_service = read_text("backend/app/services/segmentation_service.py")
migrations = "\n".join(path.name for path in (ROOT / "backend" / "alembic" / "versions").glob("*.py"))
for field_name in ("match_evidence", "false_positive_evidence", "false_negative_evidence"):
assert field_name in qa_route
assert field_name in detection_service
assert field_name in segmentation_service
assert "quality_check_items" not in migrations
@@ -267,6 +267,15 @@ def test_qa_route_persists_quality_check_domain_record(monkeypatch) -> None:
"mean_iou": 1.0, "mean_iou": 1.0,
"iou_threshold": 0.5, "iou_threshold": 0.5,
"warnings": [], "warnings": [],
"match_evidence": [
{
"candidate_feature_id": "candidate-1",
"reference_feature_id": "reference-1",
"iou": 1.0,
}
],
"false_positive_evidence": [{"candidate_feature_id": "candidate-extra"}],
"false_negative_evidence": [{"reference_feature_id": "reference-missing"}],
} }
}, },
)(), )(),
@@ -287,6 +296,9 @@ def test_qa_route_persists_quality_check_domain_record(monkeypatch) -> None:
assert persisted_quality_checks[0].job_id == job_id assert persisted_quality_checks[0].job_id == job_id
assert persisted_quality_checks[0].candidate_dataset_id == candidate_dataset_id assert persisted_quality_checks[0].candidate_dataset_id == candidate_dataset_id
assert persisted_quality_checks[0].reference_dataset_id == reference_dataset_id assert persisted_quality_checks[0].reference_dataset_id == reference_dataset_id
assert persisted_quality_checks[0].findings_json["match_evidence"][0]["candidate_feature_id"] == "candidate-1"
assert persisted_quality_checks[0].findings_json["false_positive_evidence"][0]["candidate_feature_id"] == "candidate-extra"
assert persisted_quality_checks[0].findings_json["false_negative_evidence"][0]["reference_feature_id"] == "reference-missing"
assert [metric.metric_key for metric in persisted_metrics] == [ assert [metric.metric_key for metric in persisted_metrics] == [
"precision", "precision",
"recall", "recall",
+30 -1
View File
@@ -988,6 +988,14 @@ Request:
Response is wrapped in the job envelope. On success, `result_json` includes precision, recall, F1, mean IoU, false positives, false negatives and `quality_check_id`. Response is wrapped in the job envelope. On success, `result_json` includes precision, recall, F1, mean IoU, false positives, false negatives and `quality_check_id`.
Sprint 111 also includes feature-level evidence arrays for map/review handoff:
- `match_evidence`: matched candidate/reference feature ids with IoU.
- `false_positive_evidence`: unmatched candidate feature ids.
- `false_negative_evidence`: unmatched reference feature ids.
These arrays are derived from the same persisted/source geometries used for IoU matching. They are not separate QA records yet; they are persisted inside `quality_checks.findings_json`.
Sprint 7A persists the QA/QC result as: Sprint 7A persists the QA/QC result as:
- `jobs`: execution state. - `jobs`: execution state.
@@ -1016,7 +1024,28 @@ Response:
"status": "ok", "status": "ok",
"score": 0.5, "score": 0.5,
"parameters_json": {}, "parameters_json": {},
"findings_json": {}, "findings_json": {
"matches": 1,
"false_positives": 1,
"false_negatives": 1,
"match_evidence": [
{
"candidate_feature_id": "candidate-feature-id",
"reference_feature_id": "reference-feature-id",
"iou": 0.83
}
],
"false_positive_evidence": [
{
"candidate_feature_id": "candidate-extra-id"
}
],
"false_negative_evidence": [
{
"reference_feature_id": "reference-missing-id"
}
]
},
"metrics": [ "metrics": [
{ {
"metric_key": "precision", "metric_key": "precision",
+34
View File
@@ -1,3 +1,37 @@
## Sprint 111 QA feature evidence persistence (2026-06-25)
Changed:
- Added feature-level evidence extraction to the shared QA IoU matcher.
- Dataset QA now returns and persists `match_evidence`, `false_positive_evidence` and `false_negative_evidence`.
- Detection QA and segmentation QA now use the same evidence-aware matcher and persist the same evidence keys in `quality_checks.findings_json`.
- Extended the QA/QC drilldown with compact matched, false-positive and false-negative feature id lists before the raw findings JSON.
- Updated `docs/API_CONTRACTS.md`, `frontend/README.md`, `CHANGELOG.md` and `docs/TODO.md`.
- Added/extended regression coverage in `backend/tests/test_qa_service.py`, `backend/tests/test_sprint7a_persistence_foundation.py` and `backend/tests/test_sprint111_qa_feature_evidence.py`.
Validation:
- RED: `python -m pytest backend\tests\test_qa_service.py -q` failed before implementation because `QaProviderComparisonResult` had no `match_evidence`.
- RED: `python -m pytest backend\tests\test_sprint111_qa_feature_evidence.py -q` failed before docs were updated because `docs/API_CONTRACTS.md` did not document the evidence keys.
- `python -m pytest backend\tests\test_qa_service.py -q` passed: 3 tests.
- `python -m pytest backend\tests\test_sprint8c_detection_visualization_qa.py backend\tests\test_sprint9_segmentation_foundation.py -q` passed: 18 tests.
- `python -m pytest backend\tests\test_sprint111_qa_feature_evidence.py -q` passed: 2 tests.
- `python -m pytest backend\tests\test_qa_service.py backend\tests\test_sprint7a_persistence_foundation.py -q` passed: 10 tests.
- `python -m compileall backend/app` passed.
- `cd backend && python -m pytest -q` passed: 352 tests.
- `cd frontend && npm run typecheck` passed.
- `cd frontend && npm run build` passed.
- `cd backend && python -m alembic heads` passed: `202606120900 (head)`.
- `cd backend && python -m alembic upgrade head --sql` passed.
- `bash -n scripts/live_migration_smoke.sh` passed.
- `bash scripts/run_readiness_check.sh` passed.
Limitations:
- Feature-level evidence is persisted as ids/IoU metadata in `quality_checks.findings_json`; no `quality_check_items` table or first-class evidence geometry table was introduced.
- Evidence map overlays can now be built from persisted ids, but overlay generation remains future work.
- No migration, provider fetching, AI dependency, real model behavior or new product domain was added.
Next recommended pass:
- Add QA evidence overlay generation by resolving persisted evidence ids back to candidate/reference geometries and rendering false positives/false negatives as MapLibre layers.
## Sprint 110 Map QA evidence drilldown (2026-06-25) ## Sprint 110 Map QA evidence drilldown (2026-06-25)
Changed: Changed:
+1
View File
@@ -377,3 +377,4 @@ This file now starts with the current implementation status. Older preparation/b
- [x] Persist map area selections as reusable derived vector datasets indexed into `vector_features`. - [x] Persist map area selections as reusable derived vector datasets indexed into `vector_features`.
- [x] Add Map workspace QA/QC shortcut for saved derived selection datasets. - [x] Add Map workspace QA/QC shortcut for saved derived selection datasets.
- [x] Add Map workspace QA/QC evidence drilldown handoff for saved selection comparisons. - [x] Add Map workspace QA/QC evidence drilldown handoff for saved selection comparisons.
- [x] Persist QA/QC feature-level evidence for matches, false positives and false negatives.
+1
View File
@@ -287,6 +287,7 @@ AI Lab run controls explicitly explain when no raster dataset is available, inst
- Raster controls show the latest generated tile manifest path from persisted `raster.tile` jobs and can hand that path directly to Detection Lab or Segmentation Lab with the selected raster dataset. - Raster controls show the latest generated tile manifest path from persisted `raster.tile` jobs and can hand that path directly to Detection Lab or Segmentation Lab with the selected raster dataset.
- The QA/QC workspace shows candidate/reference handoff cards and resolves persisted quality-check dataset IDs back to dataset names when the datasets are loaded in the current project context. - The QA/QC workspace shows candidate/reference handoff cards and resolves persisted quality-check dataset IDs back to dataset names when the datasets are loaded in the current project context.
- The QA/QC workspace includes a selected-check evidence drilldown with candidate/reference provenance, false-positive/negative metric evidence, map handoff context and parameters/findings JSON. - The QA/QC workspace includes a selected-check evidence drilldown with candidate/reference provenance, false-positive/negative metric evidence, map handoff context and parameters/findings JSON.
- QA/QC findings now persist feature-level evidence in `findings_json`: matched candidate/reference feature ids with IoU, false-positive candidate feature ids and false-negative reference feature ids. The QA/QC drilldown renders these as compact evidence lists before the raw JSON.
## Raster dependency visibility ## Raster dependency visibility
@@ -46,6 +46,20 @@ function metricByKey(check: QualityCheckRead | null, metricKey: string): MetricR
return check?.metrics.find((metric) => metric.metric_key === metricKey) return check?.metrics.find((metric) => metric.metric_key === metricKey)
} }
function findingEvidenceList(check: QualityCheckRead | null, key: string): Record<string, unknown>[] {
const value = check?.findings_json?.[key]
return Array.isArray(value) ? value.filter((item): item is Record<string, unknown> => Boolean(item) && typeof item === 'object' && !Array.isArray(item)) : []
}
function evidenceLabel(item: Record<string, unknown>): string {
const candidate = item['candidate_feature_id']
const reference = item['reference_feature_id']
const iou = item['iou']
const pairLabel = [candidate ? `Candidate ${candidate}` : null, reference ? `Reference ${reference}` : null].filter(Boolean).join(' / ')
const iouValue = Number(iou)
return iou === null || iou === undefined || !Number.isFinite(iouValue) ? pairLabel || JSON.stringify(item) : `${pairLabel || 'Match'} / IoU ${iouValue.toFixed(3)}`
}
function formatQualityTimestamp(value?: string | null): string { function formatQualityTimestamp(value?: string | null): string {
return value || 'n/a' return value || 'n/a'
} }
@@ -108,6 +122,9 @@ export function QualityResultsPanel({
const selectedReferenceName = selectedQualityCheck const selectedReferenceName = selectedQualityCheck
? datasetNameById.get(selectedQualityCheck.reference_dataset_id) ?? selectedQualityCheck.reference_dataset_id ? datasetNameById.get(selectedQualityCheck.reference_dataset_id) ?? selectedQualityCheck.reference_dataset_id
: 'n/a' : 'n/a'
const selectedMatchEvidence = findingEvidenceList(selectedQualityCheck, 'match_evidence')
const selectedFalsePositiveEvidence = findingEvidenceList(selectedQualityCheck, 'false_positive_evidence')
const selectedFalseNegativeEvidence = findingEvidenceList(selectedQualityCheck, 'false_negative_evidence')
const qualityStatuses = useMemo(() => Array.from(new Set(qualityChecks.map((check) => check.status))).sort(), [qualityChecks]) const qualityStatuses = useMemo(() => Array.from(new Set(qualityChecks.map((check) => check.status))).sort(), [qualityChecks])
const qualityTypes = useMemo(() => Array.from(new Set(qualityChecks.map((check) => check.check_type))).sort(), [qualityChecks]) const qualityTypes = useMemo(() => Array.from(new Set(qualityChecks.map((check) => check.check_type))).sort(), [qualityChecks])
const filteredQualityChecks = useMemo( const filteredQualityChecks = useMemo(
@@ -239,6 +256,44 @@ export function QualityResultsPanel({
<p>Use the candidate and reference datasets as map layers for spatial review.</p> <p>Use the candidate and reference datasets as map layers for spatial review.</p>
</div> </div>
</div> </div>
<div className="quality-feature-evidence-grid" aria-label="Feature-level QA/QC evidence">
<div>
<span>Matched feature ids</span>
{selectedMatchEvidence.length > 0 ? (
<ul>
{selectedMatchEvidence.slice(0, 6).map((item, index) => (
<li key={`${evidenceLabel(item)}-${index}`}>{evidenceLabel(item)}</li>
))}
</ul>
) : (
<p>No matched feature ids persisted for this check.</p>
)}
</div>
<div>
<span>False positive feature ids</span>
{selectedFalsePositiveEvidence.length > 0 ? (
<ul>
{selectedFalsePositiveEvidence.slice(0, 6).map((item, index) => (
<li key={`${evidenceLabel(item)}-${index}`}>{evidenceLabel(item)}</li>
))}
</ul>
) : (
<p>No false-positive feature ids persisted for this check.</p>
)}
</div>
<div>
<span>False negative feature ids</span>
{selectedFalseNegativeEvidence.length > 0 ? (
<ul>
{selectedFalseNegativeEvidence.slice(0, 6).map((item, index) => (
<li key={`${evidenceLabel(item)}-${index}`}>{evidenceLabel(item)}</li>
))}
</ul>
) : (
<p>No false-negative feature ids persisted for this check.</p>
)}
</div>
</div>
<div className="quality-provenance-grid"> <div className="quality-provenance-grid">
<div> <div>
<span>Parameters</span> <span>Parameters</span>
+16 -1
View File
@@ -1618,6 +1618,7 @@ button.entity-card {
.quality-drilldown-grid, .quality-drilldown-grid,
.quality-evidence-token-grid, .quality-evidence-token-grid,
.quality-feature-evidence-grid,
.quality-provenance-grid { .quality-provenance-grid {
display: grid; display: grid;
grid-template-columns: repeat(auto-fit, minmax(12rem, 1fr)); grid-template-columns: repeat(auto-fit, minmax(12rem, 1fr));
@@ -1641,6 +1642,7 @@ button.entity-card {
.quality-drilldown-grid > div, .quality-drilldown-grid > div,
.quality-evidence-token-grid > div, .quality-evidence-token-grid > div,
.quality-feature-evidence-grid > div,
.quality-provenance-grid > div { .quality-provenance-grid > div {
min-width: 0; min-width: 0;
border: 1px solid var(--line); border: 1px solid var(--line);
@@ -1663,6 +1665,7 @@ button.entity-card {
.quality-handoff-grid span, .quality-handoff-grid span,
.quality-drilldown-grid span, .quality-drilldown-grid span,
.quality-evidence-token-grid span, .quality-evidence-token-grid span,
.quality-feature-evidence-grid span,
.quality-provenance-grid span, .quality-provenance-grid span,
.quality-score-row span, .quality-score-row span,
.latest-export-card span, .latest-export-card span,
@@ -1703,7 +1706,8 @@ button.entity-card {
} }
.quality-drilldown-grid p, .quality-drilldown-grid p,
.quality-evidence-token-grid p { .quality-evidence-token-grid p,
.quality-feature-evidence-grid p {
margin: 0.24rem 0 0; margin: 0.24rem 0 0;
color: var(--muted); color: var(--muted);
font-size: 0.8rem; font-size: 0.8rem;
@@ -1711,6 +1715,17 @@ button.entity-card {
overflow-wrap: anywhere; overflow-wrap: anywhere;
} }
.quality-feature-evidence-grid ul {
display: grid;
gap: 0.28rem;
margin: 0.38rem 0 0;
padding-left: 1.1rem;
color: var(--text);
font-size: 0.78rem;
line-height: 1.35;
overflow-wrap: anywhere;
}
.quality-provenance-pre { .quality-provenance-pre {
max-height: 11rem; max-height: 11rem;
margin: 0.38rem 0 0; margin: 0.38rem 0 0;