Persist QA feature evidence
This commit is contained in:
@@ -7,6 +7,14 @@
|
||||
|
||||
# Changelog
|
||||
|
||||
## Sprint 111 QA feature evidence persistence (2026-06-25)
|
||||
|
||||
- Added feature-level QA evidence to dataset, detection and segmentation QA matching.
|
||||
- Persisted matched feature ids, false-positive feature ids and false-negative feature ids inside `quality_checks.findings_json`.
|
||||
- Extended QA/QC drilldown with compact matched/false-positive/false-negative feature id lists beside the existing metrics and raw findings JSON.
|
||||
- Updated API contracts to document `match_evidence`, `false_positive_evidence` and `false_negative_evidence`.
|
||||
- No migration, new table, provider fetching, AI behavior or new product domain was introduced.
|
||||
|
||||
## Sprint 110 Map QA evidence drilldown (2026-06-25)
|
||||
|
||||
- Extended the Map workspace QA/QC shortcut with inline evidence after comparing a saved derived selection dataset.
|
||||
|
||||
@@ -59,6 +59,9 @@ def compare_candidate_with_reference(
|
||||
"warnings": result_json.get("warnings", []),
|
||||
"unsupported_geometry": result_json.get("unsupported_geometry", False),
|
||||
"unsupported_geometries": result_json.get("unsupported_geometries", []),
|
||||
"match_evidence": result_json.get("match_evidence", []),
|
||||
"false_positive_evidence": result_json.get("false_positive_evidence", []),
|
||||
"false_negative_evidence": result_json.get("false_negative_evidence", []),
|
||||
},
|
||||
metrics={
|
||||
"precision": result_json.get("precision"),
|
||||
|
||||
@@ -28,6 +28,9 @@ class QaProviderComparisonResult(BaseModel):
|
||||
iou_threshold: float
|
||||
unsupported_geometry: bool = False
|
||||
unsupported_geometries: list[str] = Field(default_factory=list)
|
||||
match_evidence: list[dict] = Field(default_factory=list)
|
||||
false_positive_evidence: list[dict] = Field(default_factory=list)
|
||||
false_negative_evidence: list[dict] = Field(default_factory=list)
|
||||
generated_at: datetime
|
||||
|
||||
|
||||
|
||||
@@ -287,18 +287,18 @@ class DetectionService:
|
||||
|
||||
candidate_geometries = [({"id": str(row.id), "class_name": row.class_name}, to_shape(row.geometry)) for row in detections]
|
||||
reference_geometries = [({"id": str(row.id), "feature_class": row.feature_class}, to_shape(row.geometry)) for row in references]
|
||||
matches, false_positives, false_negatives, match_iou_values, warnings, unsupported = QaService._match_io_u_metrics(
|
||||
evidence = QaService._match_io_u_evidence(
|
||||
candidate_geometries,
|
||||
reference_geometries,
|
||||
iou_threshold,
|
||||
)
|
||||
mean_iou = None if not match_iou_values else sum(match_iou_values) / len(match_iou_values)
|
||||
precision = matches / (matches + false_positives) if matches + false_positives > 0 else None
|
||||
recall = matches / (matches + false_negatives) if matches + false_negatives > 0 else None
|
||||
mean_iou = None if not evidence.match_iou_values else sum(evidence.match_iou_values) / len(evidence.match_iou_values)
|
||||
precision = evidence.matches / (evidence.matches + evidence.false_positives) if evidence.matches + evidence.false_positives > 0 else None
|
||||
recall = evidence.matches / (evidence.matches + evidence.false_negatives) if evidence.matches + evidence.false_negatives > 0 else None
|
||||
f1_score = None
|
||||
if precision is not None and recall is not None:
|
||||
f1_score = (2 * precision * recall) / (precision + recall) if precision + recall > 0 else 0.0
|
||||
status = "unsupported" if unsupported else "ok"
|
||||
status = "unsupported" if evidence.unsupported else "ok"
|
||||
quality_check = QualityService.persist_quality_check(
|
||||
db=db,
|
||||
project_id=run.project_id,
|
||||
@@ -316,19 +316,22 @@ class DetectionService:
|
||||
"min_confidence": min_confidence,
|
||||
},
|
||||
findings={
|
||||
"matches": matches,
|
||||
"false_positives": false_positives,
|
||||
"false_negatives": false_negatives,
|
||||
"warnings": warnings,
|
||||
"unsupported_geometry": unsupported,
|
||||
"matches": evidence.matches,
|
||||
"false_positives": evidence.false_positives,
|
||||
"false_negatives": evidence.false_negatives,
|
||||
"warnings": evidence.warnings,
|
||||
"unsupported_geometry": evidence.unsupported,
|
||||
"match_evidence": evidence.match_evidence,
|
||||
"false_positive_evidence": evidence.false_positive_evidence,
|
||||
"false_negative_evidence": evidence.false_negative_evidence,
|
||||
},
|
||||
metrics={
|
||||
"precision": precision,
|
||||
"recall": recall,
|
||||
"f1": f1_score,
|
||||
"mean_iou": mean_iou,
|
||||
"false_positive_count": false_positives,
|
||||
"false_negative_count": false_negatives,
|
||||
"false_positive_count": evidence.false_positives,
|
||||
"false_negative_count": evidence.false_negatives,
|
||||
},
|
||||
)
|
||||
return {
|
||||
@@ -338,15 +341,18 @@ class DetectionService:
|
||||
"reference_dataset_id": str(reference_dataset_id),
|
||||
"candidate_feature_count": len(candidate_geometries),
|
||||
"reference_feature_count": len(reference_geometries),
|
||||
"matches": matches,
|
||||
"false_positives": false_positives,
|
||||
"false_negatives": false_negatives,
|
||||
"matches": evidence.matches,
|
||||
"false_positives": evidence.false_positives,
|
||||
"false_negatives": evidence.false_negatives,
|
||||
"precision": precision,
|
||||
"recall": recall,
|
||||
"f1_score": f1_score,
|
||||
"mean_iou": mean_iou,
|
||||
"iou_threshold": iou_threshold,
|
||||
"warnings": warnings,
|
||||
"warnings": evidence.warnings,
|
||||
"match_evidence": evidence.match_evidence,
|
||||
"false_positive_evidence": evidence.false_positive_evidence,
|
||||
"false_negative_evidence": evidence.false_negative_evidence,
|
||||
}
|
||||
|
||||
@staticmethod
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any
|
||||
from uuid import UUID
|
||||
@@ -16,6 +17,19 @@ from app.schemas.qa import QaProviderComparisonResult
|
||||
from app.services.vector_operations_service import VectorOperationsService
|
||||
|
||||
|
||||
@dataclass
|
||||
class QaMatchEvidence:
|
||||
matches: int = 0
|
||||
false_positives: int = 0
|
||||
false_negatives: int = 0
|
||||
match_iou_values: list[float] = field(default_factory=list)
|
||||
warnings: list[str] = field(default_factory=list)
|
||||
unsupported: bool = False
|
||||
match_evidence: list[dict[str, Any]] = field(default_factory=list)
|
||||
false_positive_evidence: list[dict[str, Any]] = field(default_factory=list)
|
||||
false_negative_evidence: list[dict[str, Any]] = field(default_factory=list)
|
||||
|
||||
|
||||
def _extract_crs_warnings(source_dataset: Dataset, reference_dataset: Dataset) -> list[str]:
|
||||
warnings: list[str] = []
|
||||
for dataset, label in ((source_dataset, "candidate"), (reference_dataset, "reference")):
|
||||
@@ -92,17 +106,34 @@ class QaService:
|
||||
return area_geometry
|
||||
|
||||
@staticmethod
|
||||
def _match_io_u_metrics(
|
||||
def _feature_identifier(feature: dict[str, Any], fallback_prefix: str, index: int) -> str:
|
||||
feature_id = feature.get("id")
|
||||
if feature_id is not None:
|
||||
return str(feature_id)
|
||||
properties = feature.get("properties")
|
||||
if isinstance(properties, dict):
|
||||
for key in ("vector_feature_id", "source_feature_id", "detection_id", "segmentation_id", "id", "name"):
|
||||
value = properties.get(key)
|
||||
if value is not None:
|
||||
return str(value)
|
||||
for key in ("vector_feature_id", "source_feature_id", "detection_id", "segmentation_id", "class_name", "feature_class"):
|
||||
value = feature.get(key)
|
||||
if value is not None:
|
||||
return str(value)
|
||||
return f"{fallback_prefix}-{index + 1}"
|
||||
|
||||
@staticmethod
|
||||
def _match_io_u_evidence(
|
||||
source_geometries: list[tuple[dict[str, Any], BaseGeometry]],
|
||||
reference_geometries: list[tuple[dict[str, Any], BaseGeometry]],
|
||||
iou_threshold: float,
|
||||
) -> tuple[int, int, int, list[float], list[str], bool]:
|
||||
) -> QaMatchEvidence:
|
||||
source_supported = [
|
||||
(feature, geom) for feature, geom in source_geometries if geom.geom_type in QaService.SUPPORTED_GEOMETRY_TYPES
|
||||
(index, feature, geom) for index, (feature, geom) in enumerate(source_geometries) if geom.geom_type in QaService.SUPPORTED_GEOMETRY_TYPES
|
||||
]
|
||||
reference_supported = [
|
||||
(feature, geom)
|
||||
for feature, geom in reference_geometries
|
||||
(index, feature, geom)
|
||||
for index, (feature, geom) in enumerate(reference_geometries)
|
||||
if geom.geom_type in QaService.SUPPORTED_GEOMETRY_TYPES
|
||||
]
|
||||
|
||||
@@ -114,29 +145,35 @@ class QaService:
|
||||
}
|
||||
)
|
||||
if not source_supported or not reference_supported:
|
||||
return (
|
||||
0,
|
||||
len(source_supported),
|
||||
len(reference_supported),
|
||||
[],
|
||||
[f"Unsupported geometry types: {unsupported}"] if unsupported else [],
|
||||
True,
|
||||
return QaMatchEvidence(
|
||||
false_positives=len(source_supported),
|
||||
false_negatives=len(reference_supported),
|
||||
warnings=[f"Unsupported geometry types: {unsupported}"] if unsupported else [],
|
||||
unsupported=True,
|
||||
false_positive_evidence=[
|
||||
{"candidate_feature_id": QaService._feature_identifier(feature, "candidate", source_index)}
|
||||
for source_index, feature, _ in source_supported
|
||||
],
|
||||
false_negative_evidence=[
|
||||
{"reference_feature_id": QaService._feature_identifier(feature, "reference", reference_index)}
|
||||
for reference_index, feature, _ in reference_supported
|
||||
],
|
||||
)
|
||||
|
||||
unmatched_reference_indices = set(range(len(reference_supported)))
|
||||
matches = 0
|
||||
match_iou_values: list[float] = []
|
||||
false_positives = 0
|
||||
evidence = QaMatchEvidence(warnings=[f"Unsupported geometry types: {unsupported}"] if unsupported else [], unsupported=bool(unsupported))
|
||||
|
||||
for _, source_geom in source_supported:
|
||||
for source_index, source_feature, source_geom in source_supported:
|
||||
source_feature_id = QaService._feature_identifier(source_feature, "candidate", source_index)
|
||||
if source_geom.area <= 0:
|
||||
false_positives += 1
|
||||
evidence.false_positives += 1
|
||||
evidence.false_positive_evidence.append({"candidate_feature_id": source_feature_id})
|
||||
continue
|
||||
|
||||
best_iou = 0.0
|
||||
best_index = None
|
||||
for reference_index in list(unmatched_reference_indices):
|
||||
_, reference_geom = reference_supported[reference_index]
|
||||
_, _, reference_geom = reference_supported[reference_index]
|
||||
if reference_geom.area <= 0:
|
||||
unmatched_reference_indices.discard(reference_index)
|
||||
continue
|
||||
@@ -161,16 +198,45 @@ class QaService:
|
||||
best_index = reference_index
|
||||
|
||||
if best_index is not None and best_iou >= iou_threshold:
|
||||
matches += 1
|
||||
match_iou_values.append(best_iou)
|
||||
reference_original_index, reference_feature, _ = reference_supported[best_index]
|
||||
evidence.matches += 1
|
||||
evidence.match_iou_values.append(best_iou)
|
||||
evidence.match_evidence.append(
|
||||
{
|
||||
"candidate_feature_id": source_feature_id,
|
||||
"reference_feature_id": QaService._feature_identifier(reference_feature, "reference", reference_original_index),
|
||||
"iou": best_iou,
|
||||
}
|
||||
)
|
||||
unmatched_reference_indices.discard(best_index)
|
||||
else:
|
||||
false_positives += 1
|
||||
evidence.false_positives += 1
|
||||
evidence.false_positive_evidence.append({"candidate_feature_id": source_feature_id})
|
||||
|
||||
false_negatives = len(unmatched_reference_indices)
|
||||
warnings: list[str] = [f"Unsupported geometry types: {unsupported}"] if unsupported else []
|
||||
evidence.false_negatives = len(unmatched_reference_indices)
|
||||
for reference_index in sorted(unmatched_reference_indices):
|
||||
reference_original_index, reference_feature, _ = reference_supported[reference_index]
|
||||
evidence.false_negative_evidence.append(
|
||||
{"reference_feature_id": QaService._feature_identifier(reference_feature, "reference", reference_original_index)}
|
||||
)
|
||||
|
||||
return matches, false_positives, false_negatives, match_iou_values, warnings, bool(unsupported)
|
||||
return evidence
|
||||
|
||||
@staticmethod
|
||||
def _match_io_u_metrics(
|
||||
source_geometries: list[tuple[dict[str, Any], BaseGeometry]],
|
||||
reference_geometries: list[tuple[dict[str, Any], BaseGeometry]],
|
||||
iou_threshold: float,
|
||||
) -> tuple[int, int, int, list[float], list[str], bool]:
|
||||
evidence = QaService._match_io_u_evidence(source_geometries, reference_geometries, iou_threshold)
|
||||
return (
|
||||
evidence.matches,
|
||||
evidence.false_positives,
|
||||
evidence.false_negatives,
|
||||
evidence.match_iou_values,
|
||||
evidence.warnings,
|
||||
evidence.unsupported,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def compare_candidate_with_reference(
|
||||
@@ -206,7 +272,7 @@ class QaService:
|
||||
candidate_geometries = QaService._apply_area_filter(candidate_geometries, area_geometry, dataset_id=candidate_dataset.id)
|
||||
reference_geometries = QaService._apply_area_filter(reference_geometries, area_geometry, dataset_id=reference_dataset.id)
|
||||
|
||||
matches, false_positives, false_negatives, match_iou_values, warnings, unsupported = QaService._match_io_u_metrics(
|
||||
evidence = QaService._match_io_u_evidence(
|
||||
candidate_geometries,
|
||||
reference_geometries,
|
||||
iou_threshold,
|
||||
@@ -214,35 +280,38 @@ class QaService:
|
||||
|
||||
candidate_feature_count = len(candidate_payload.get("features", [])) if isinstance(candidate_payload, dict) else 0
|
||||
reference_feature_count = len(reference_payload.get("features", [])) if isinstance(reference_payload, dict) else 0
|
||||
mean_iou = None if not match_iou_values else sum(match_iou_values) / len(match_iou_values)
|
||||
mean_iou = None if not evidence.match_iou_values else sum(evidence.match_iou_values) / len(evidence.match_iou_values)
|
||||
|
||||
precision = None
|
||||
if matches + false_positives > 0:
|
||||
precision = matches / (matches + false_positives)
|
||||
if evidence.matches + evidence.false_positives > 0:
|
||||
precision = evidence.matches / (evidence.matches + evidence.false_positives)
|
||||
|
||||
recall = None
|
||||
if matches + false_negatives > 0:
|
||||
recall = matches / (matches + false_negatives)
|
||||
if evidence.matches + evidence.false_negatives > 0:
|
||||
recall = evidence.matches / (evidence.matches + evidence.false_negatives)
|
||||
|
||||
f1_score = None
|
||||
if precision is not None and recall is not None and precision + recall > 0:
|
||||
f1_score = (2 * precision * recall) / (precision + recall)
|
||||
|
||||
status = "unsupported" if unsupported else "ok"
|
||||
status = "unsupported" if evidence.unsupported else "ok"
|
||||
return QaProviderComparisonResult(
|
||||
status=status,
|
||||
warnings=_extract_crs_warnings(candidate_dataset, reference_dataset) + warnings,
|
||||
warnings=_extract_crs_warnings(candidate_dataset, reference_dataset) + evidence.warnings,
|
||||
candidate_feature_count=candidate_feature_count,
|
||||
reference_feature_count=reference_feature_count,
|
||||
matches=matches,
|
||||
false_positives=false_positives,
|
||||
false_negatives=false_negatives,
|
||||
matches=evidence.matches,
|
||||
false_positives=evidence.false_positives,
|
||||
false_negatives=evidence.false_negatives,
|
||||
precision=precision,
|
||||
recall=recall,
|
||||
f1_score=f1_score,
|
||||
mean_iou=mean_iou,
|
||||
iou_threshold=iou_threshold,
|
||||
unsupported_geometry=unsupported,
|
||||
unsupported_geometries=warnings,
|
||||
unsupported_geometry=evidence.unsupported,
|
||||
unsupported_geometries=evidence.warnings,
|
||||
match_evidence=evidence.match_evidence,
|
||||
false_positive_evidence=evidence.false_positive_evidence,
|
||||
false_negative_evidence=evidence.false_negative_evidence,
|
||||
generated_at=datetime.now(timezone.utc),
|
||||
)
|
||||
|
||||
@@ -251,18 +251,18 @@ class SegmentationService:
|
||||
|
||||
candidate_geometries = [({"id": str(row.id), "class_name": row.class_name}, to_shape(row.geometry)) for row in segmentations]
|
||||
reference_geometries = [({"id": str(row.id), "feature_class": row.feature_class}, to_shape(row.geometry)) for row in references]
|
||||
matches, false_positives, false_negatives, match_iou_values, warnings, unsupported = QaService._match_io_u_metrics(
|
||||
evidence = QaService._match_io_u_evidence(
|
||||
candidate_geometries,
|
||||
reference_geometries,
|
||||
iou_threshold,
|
||||
)
|
||||
mean_iou = None if not match_iou_values else sum(match_iou_values) / len(match_iou_values)
|
||||
precision = matches / (matches + false_positives) if matches + false_positives > 0 else None
|
||||
recall = matches / (matches + false_negatives) if matches + false_negatives > 0 else None
|
||||
mean_iou = None if not evidence.match_iou_values else sum(evidence.match_iou_values) / len(evidence.match_iou_values)
|
||||
precision = evidence.matches / (evidence.matches + evidence.false_positives) if evidence.matches + evidence.false_positives > 0 else None
|
||||
recall = evidence.matches / (evidence.matches + evidence.false_negatives) if evidence.matches + evidence.false_negatives > 0 else None
|
||||
f1_score = None
|
||||
if precision is not None and recall is not None:
|
||||
f1_score = (2 * precision * recall) / (precision + recall) if precision + recall > 0 else 0.0
|
||||
status = "unsupported" if unsupported else "ok"
|
||||
status = "unsupported" if evidence.unsupported else "ok"
|
||||
quality_check = QualityService.persist_quality_check(
|
||||
db=db,
|
||||
project_id=run.project_id,
|
||||
@@ -280,19 +280,22 @@ class SegmentationService:
|
||||
"min_confidence": min_confidence,
|
||||
},
|
||||
findings={
|
||||
"matches": matches,
|
||||
"false_positives": false_positives,
|
||||
"false_negatives": false_negatives,
|
||||
"warnings": warnings,
|
||||
"unsupported_geometry": unsupported,
|
||||
"matches": evidence.matches,
|
||||
"false_positives": evidence.false_positives,
|
||||
"false_negatives": evidence.false_negatives,
|
||||
"warnings": evidence.warnings,
|
||||
"unsupported_geometry": evidence.unsupported,
|
||||
"match_evidence": evidence.match_evidence,
|
||||
"false_positive_evidence": evidence.false_positive_evidence,
|
||||
"false_negative_evidence": evidence.false_negative_evidence,
|
||||
},
|
||||
metrics={
|
||||
"precision": precision,
|
||||
"recall": recall,
|
||||
"f1": f1_score,
|
||||
"mean_iou": mean_iou,
|
||||
"false_positive_count": false_positives,
|
||||
"false_negative_count": false_negatives,
|
||||
"false_positive_count": evidence.false_positives,
|
||||
"false_negative_count": evidence.false_negatives,
|
||||
},
|
||||
)
|
||||
return {
|
||||
@@ -302,15 +305,18 @@ class SegmentationService:
|
||||
"reference_dataset_id": str(reference_dataset_id),
|
||||
"candidate_feature_count": len(candidate_geometries),
|
||||
"reference_feature_count": len(reference_geometries),
|
||||
"matches": matches,
|
||||
"false_positives": false_positives,
|
||||
"false_negatives": false_negatives,
|
||||
"matches": evidence.matches,
|
||||
"false_positives": evidence.false_positives,
|
||||
"false_negatives": evidence.false_negatives,
|
||||
"precision": precision,
|
||||
"recall": recall,
|
||||
"f1_score": f1_score,
|
||||
"mean_iou": mean_iou,
|
||||
"iou_threshold": iou_threshold,
|
||||
"warnings": warnings,
|
||||
"warnings": evidence.warnings,
|
||||
"match_evidence": evidence.match_evidence,
|
||||
"false_positive_evidence": evidence.false_positive_evidence,
|
||||
"false_negative_evidence": evidence.false_negative_evidence,
|
||||
}
|
||||
|
||||
@staticmethod
|
||||
|
||||
@@ -22,23 +22,30 @@ class FakeSession:
|
||||
return None
|
||||
|
||||
|
||||
def _feature(feature_id: str, coordinates: list[list[list[float]]]) -> dict:
|
||||
return {
|
||||
"type": "Feature",
|
||||
"id": feature_id,
|
||||
"properties": {"source_feature_id": feature_id},
|
||||
"geometry": {
|
||||
"type": "Polygon",
|
||||
"coordinates": coordinates,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _write_dataset(path: Path, coordinates: list[list[list[float]]]) -> None:
|
||||
payload = {
|
||||
"type": "FeatureCollection",
|
||||
"features": [
|
||||
{
|
||||
"type": "Feature",
|
||||
"properties": {},
|
||||
"geometry": {
|
||||
"type": "Polygon",
|
||||
"coordinates": coordinates,
|
||||
},
|
||||
},
|
||||
],
|
||||
"features": [_feature("feature-1", coordinates)],
|
||||
}
|
||||
path.write_text(json.dumps(payload), encoding="utf-8")
|
||||
|
||||
|
||||
def _write_features(path: Path, features: list[dict]) -> None:
|
||||
path.write_text(json.dumps({"type": "FeatureCollection", "features": features}), encoding="utf-8")
|
||||
|
||||
|
||||
def test_qa_compare_candidate_with_reference_returns_metrics(tmp_path) -> None:
|
||||
project_id = uuid4()
|
||||
candidate_id = uuid4()
|
||||
@@ -87,6 +94,62 @@ def test_qa_compare_candidate_with_reference_returns_metrics(tmp_path) -> None:
|
||||
assert result.f1_score == 1.0
|
||||
|
||||
|
||||
def test_qa_compare_candidate_with_reference_returns_feature_level_evidence(tmp_path) -> None:
|
||||
project_id = uuid4()
|
||||
candidate_id = uuid4()
|
||||
reference_id = uuid4()
|
||||
candidate_path = tmp_path / "candidate.geojson"
|
||||
reference_path = tmp_path / "reference.geojson"
|
||||
matched_candidate = [[[4.0, 51.0], [4.1, 51.0], [4.1, 51.1], [4.0, 51.1], [4.0, 51.0]]]
|
||||
matched_reference = [[[4.0, 51.0], [4.1, 51.0], [4.1, 51.1], [4.0, 51.1], [4.0, 51.0]]]
|
||||
false_positive = [[[4.5, 51.5], [4.6, 51.5], [4.6, 51.6], [4.5, 51.6], [4.5, 51.5]]]
|
||||
false_negative = [[[4.8, 51.8], [4.9, 51.8], [4.9, 51.9], [4.8, 51.9], [4.8, 51.8]]]
|
||||
_write_features(candidate_path, [_feature("candidate-match", matched_candidate), _feature("candidate-extra", false_positive)])
|
||||
_write_features(reference_path, [_feature("reference-match", matched_reference), _feature("reference-missing", false_negative)])
|
||||
|
||||
candidate = Dataset(
|
||||
id=candidate_id,
|
||||
project_id=project_id,
|
||||
name="candidate.geojson",
|
||||
dataset_type="vector",
|
||||
source="test",
|
||||
storage_path=str(candidate_path),
|
||||
crs="EPSG:4326",
|
||||
metadata_json={"crs_assumed": False},
|
||||
)
|
||||
reference = Dataset(
|
||||
id=reference_id,
|
||||
project_id=project_id,
|
||||
name="reference.geojson",
|
||||
dataset_type="vector",
|
||||
source="test",
|
||||
storage_path=str(reference_path),
|
||||
crs="EPSG:4326",
|
||||
metadata_json={"crs_assumed": False},
|
||||
)
|
||||
|
||||
result = QaService.compare_candidate_with_reference(
|
||||
db=FakeSession([candidate, reference]),
|
||||
project_id=project_id,
|
||||
candidate_dataset_id=candidate_id,
|
||||
reference_dataset_id=reference_id,
|
||||
iou_threshold=0.5,
|
||||
)
|
||||
|
||||
assert result.matches == 1
|
||||
assert result.false_positives == 1
|
||||
assert result.false_negatives == 1
|
||||
assert result.match_evidence == [
|
||||
{
|
||||
"candidate_feature_id": "candidate-match",
|
||||
"reference_feature_id": "reference-match",
|
||||
"iou": 1.0,
|
||||
}
|
||||
]
|
||||
assert result.false_positive_evidence == [{"candidate_feature_id": "candidate-extra"}]
|
||||
assert result.false_negative_evidence == [{"reference_feature_id": "reference-missing"}]
|
||||
|
||||
|
||||
def test_dataset_reference_metadata_migration_declares_required_columns() -> None:
|
||||
migration_path = Path(__file__).parents[1] / "alembic" / "versions" / "202606120001_add_dataset_reference_metadata.py"
|
||||
migration_text = migration_path.read_text(encoding="utf-8")
|
||||
|
||||
@@ -0,0 +1,39 @@
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
|
||||
|
||||
def read_text(relative_path: str) -> str:
|
||||
return (ROOT / relative_path).read_text(encoding="utf-8")
|
||||
|
||||
|
||||
def test_qa_feature_evidence_contract_is_documented_and_rendered() -> None:
|
||||
schema = read_text("backend/app/schemas/qa.py")
|
||||
quality_panel = read_text("frontend/src/components/quality/QualityResultsPanel.tsx")
|
||||
api_contracts = read_text("docs/API_CONTRACTS.md")
|
||||
|
||||
for field_name in ("match_evidence", "false_positive_evidence", "false_negative_evidence"):
|
||||
assert field_name in schema
|
||||
assert field_name in quality_panel
|
||||
assert field_name in api_contracts
|
||||
|
||||
assert "Feature-level QA/QC evidence" in quality_panel
|
||||
assert "Matched feature ids" in quality_panel
|
||||
assert "False positive feature ids" in quality_panel
|
||||
assert "False negative feature ids" in quality_panel
|
||||
assert "evidenceLabel" in quality_panel
|
||||
|
||||
|
||||
def test_qa_services_persist_feature_evidence_without_new_migrations() -> None:
|
||||
qa_route = read_text("backend/app/api/routes/qa.py")
|
||||
detection_service = read_text("backend/app/services/detection_service.py")
|
||||
segmentation_service = read_text("backend/app/services/segmentation_service.py")
|
||||
migrations = "\n".join(path.name for path in (ROOT / "backend" / "alembic" / "versions").glob("*.py"))
|
||||
|
||||
for field_name in ("match_evidence", "false_positive_evidence", "false_negative_evidence"):
|
||||
assert field_name in qa_route
|
||||
assert field_name in detection_service
|
||||
assert field_name in segmentation_service
|
||||
|
||||
assert "quality_check_items" not in migrations
|
||||
@@ -267,6 +267,15 @@ def test_qa_route_persists_quality_check_domain_record(monkeypatch) -> None:
|
||||
"mean_iou": 1.0,
|
||||
"iou_threshold": 0.5,
|
||||
"warnings": [],
|
||||
"match_evidence": [
|
||||
{
|
||||
"candidate_feature_id": "candidate-1",
|
||||
"reference_feature_id": "reference-1",
|
||||
"iou": 1.0,
|
||||
}
|
||||
],
|
||||
"false_positive_evidence": [{"candidate_feature_id": "candidate-extra"}],
|
||||
"false_negative_evidence": [{"reference_feature_id": "reference-missing"}],
|
||||
}
|
||||
},
|
||||
)(),
|
||||
@@ -287,6 +296,9 @@ def test_qa_route_persists_quality_check_domain_record(monkeypatch) -> None:
|
||||
assert persisted_quality_checks[0].job_id == job_id
|
||||
assert persisted_quality_checks[0].candidate_dataset_id == candidate_dataset_id
|
||||
assert persisted_quality_checks[0].reference_dataset_id == reference_dataset_id
|
||||
assert persisted_quality_checks[0].findings_json["match_evidence"][0]["candidate_feature_id"] == "candidate-1"
|
||||
assert persisted_quality_checks[0].findings_json["false_positive_evidence"][0]["candidate_feature_id"] == "candidate-extra"
|
||||
assert persisted_quality_checks[0].findings_json["false_negative_evidence"][0]["reference_feature_id"] == "reference-missing"
|
||||
assert [metric.metric_key for metric in persisted_metrics] == [
|
||||
"precision",
|
||||
"recall",
|
||||
|
||||
+30
-1
@@ -988,6 +988,14 @@ Request:
|
||||
|
||||
Response is wrapped in the job envelope. On success, `result_json` includes precision, recall, F1, mean IoU, false positives, false negatives and `quality_check_id`.
|
||||
|
||||
Sprint 111 also includes feature-level evidence arrays for map/review handoff:
|
||||
|
||||
- `match_evidence`: matched candidate/reference feature ids with IoU.
|
||||
- `false_positive_evidence`: unmatched candidate feature ids.
|
||||
- `false_negative_evidence`: unmatched reference feature ids.
|
||||
|
||||
These arrays are derived from the same persisted/source geometries used for IoU matching. They are not separate QA records yet; they are persisted inside `quality_checks.findings_json`.
|
||||
|
||||
Sprint 7A persists the QA/QC result as:
|
||||
|
||||
- `jobs`: execution state.
|
||||
@@ -1016,7 +1024,28 @@ Response:
|
||||
"status": "ok",
|
||||
"score": 0.5,
|
||||
"parameters_json": {},
|
||||
"findings_json": {},
|
||||
"findings_json": {
|
||||
"matches": 1,
|
||||
"false_positives": 1,
|
||||
"false_negatives": 1,
|
||||
"match_evidence": [
|
||||
{
|
||||
"candidate_feature_id": "candidate-feature-id",
|
||||
"reference_feature_id": "reference-feature-id",
|
||||
"iou": 0.83
|
||||
}
|
||||
],
|
||||
"false_positive_evidence": [
|
||||
{
|
||||
"candidate_feature_id": "candidate-extra-id"
|
||||
}
|
||||
],
|
||||
"false_negative_evidence": [
|
||||
{
|
||||
"reference_feature_id": "reference-missing-id"
|
||||
}
|
||||
]
|
||||
},
|
||||
"metrics": [
|
||||
{
|
||||
"metric_key": "precision",
|
||||
|
||||
@@ -1,3 +1,37 @@
|
||||
## Sprint 111 QA feature evidence persistence (2026-06-25)
|
||||
|
||||
Changed:
|
||||
- Added feature-level evidence extraction to the shared QA IoU matcher.
|
||||
- Dataset QA now returns and persists `match_evidence`, `false_positive_evidence` and `false_negative_evidence`.
|
||||
- Detection QA and segmentation QA now use the same evidence-aware matcher and persist the same evidence keys in `quality_checks.findings_json`.
|
||||
- Extended the QA/QC drilldown with compact matched, false-positive and false-negative feature id lists before the raw findings JSON.
|
||||
- Updated `docs/API_CONTRACTS.md`, `frontend/README.md`, `CHANGELOG.md` and `docs/TODO.md`.
|
||||
- Added/extended regression coverage in `backend/tests/test_qa_service.py`, `backend/tests/test_sprint7a_persistence_foundation.py` and `backend/tests/test_sprint111_qa_feature_evidence.py`.
|
||||
|
||||
Validation:
|
||||
- RED: `python -m pytest backend\tests\test_qa_service.py -q` failed before implementation because `QaProviderComparisonResult` had no `match_evidence`.
|
||||
- RED: `python -m pytest backend\tests\test_sprint111_qa_feature_evidence.py -q` failed before docs were updated because `docs/API_CONTRACTS.md` did not document the evidence keys.
|
||||
- `python -m pytest backend\tests\test_qa_service.py -q` passed: 3 tests.
|
||||
- `python -m pytest backend\tests\test_sprint8c_detection_visualization_qa.py backend\tests\test_sprint9_segmentation_foundation.py -q` passed: 18 tests.
|
||||
- `python -m pytest backend\tests\test_sprint111_qa_feature_evidence.py -q` passed: 2 tests.
|
||||
- `python -m pytest backend\tests\test_qa_service.py backend\tests\test_sprint7a_persistence_foundation.py -q` passed: 10 tests.
|
||||
- `python -m compileall backend/app` passed.
|
||||
- `cd backend && python -m pytest -q` passed: 352 tests.
|
||||
- `cd frontend && npm run typecheck` passed.
|
||||
- `cd frontend && npm run build` passed.
|
||||
- `cd backend && python -m alembic heads` passed: `202606120900 (head)`.
|
||||
- `cd backend && python -m alembic upgrade head --sql` passed.
|
||||
- `bash -n scripts/live_migration_smoke.sh` passed.
|
||||
- `bash scripts/run_readiness_check.sh` passed.
|
||||
|
||||
Limitations:
|
||||
- Feature-level evidence is persisted as ids/IoU metadata in `quality_checks.findings_json`; no `quality_check_items` table or first-class evidence geometry table was introduced.
|
||||
- Evidence map overlays can now be built from persisted ids, but overlay generation remains future work.
|
||||
- No migration, provider fetching, AI dependency, real model behavior or new product domain was added.
|
||||
|
||||
Next recommended pass:
|
||||
- Add QA evidence overlay generation by resolving persisted evidence ids back to candidate/reference geometries and rendering false positives/false negatives as MapLibre layers.
|
||||
|
||||
## Sprint 110 Map QA evidence drilldown (2026-06-25)
|
||||
|
||||
Changed:
|
||||
|
||||
@@ -377,3 +377,4 @@ This file now starts with the current implementation status. Older preparation/b
|
||||
- [x] Persist map area selections as reusable derived vector datasets indexed into `vector_features`.
|
||||
- [x] Add Map workspace QA/QC shortcut for saved derived selection datasets.
|
||||
- [x] Add Map workspace QA/QC evidence drilldown handoff for saved selection comparisons.
|
||||
- [x] Persist QA/QC feature-level evidence for matches, false positives and false negatives.
|
||||
|
||||
@@ -287,6 +287,7 @@ AI Lab run controls explicitly explain when no raster dataset is available, inst
|
||||
- Raster controls show the latest generated tile manifest path from persisted `raster.tile` jobs and can hand that path directly to Detection Lab or Segmentation Lab with the selected raster dataset.
|
||||
- The QA/QC workspace shows candidate/reference handoff cards and resolves persisted quality-check dataset IDs back to dataset names when the datasets are loaded in the current project context.
|
||||
- The QA/QC workspace includes a selected-check evidence drilldown with candidate/reference provenance, false-positive/negative metric evidence, map handoff context and parameters/findings JSON.
|
||||
- QA/QC findings now persist feature-level evidence in `findings_json`: matched candidate/reference feature ids with IoU, false-positive candidate feature ids and false-negative reference feature ids. The QA/QC drilldown renders these as compact evidence lists before the raw JSON.
|
||||
|
||||
## Raster dependency visibility
|
||||
|
||||
|
||||
@@ -46,6 +46,20 @@ function metricByKey(check: QualityCheckRead | null, metricKey: string): MetricR
|
||||
return check?.metrics.find((metric) => metric.metric_key === metricKey)
|
||||
}
|
||||
|
||||
function findingEvidenceList(check: QualityCheckRead | null, key: string): Record<string, unknown>[] {
|
||||
const value = check?.findings_json?.[key]
|
||||
return Array.isArray(value) ? value.filter((item): item is Record<string, unknown> => Boolean(item) && typeof item === 'object' && !Array.isArray(item)) : []
|
||||
}
|
||||
|
||||
function evidenceLabel(item: Record<string, unknown>): string {
|
||||
const candidate = item['candidate_feature_id']
|
||||
const reference = item['reference_feature_id']
|
||||
const iou = item['iou']
|
||||
const pairLabel = [candidate ? `Candidate ${candidate}` : null, reference ? `Reference ${reference}` : null].filter(Boolean).join(' / ')
|
||||
const iouValue = Number(iou)
|
||||
return iou === null || iou === undefined || !Number.isFinite(iouValue) ? pairLabel || JSON.stringify(item) : `${pairLabel || 'Match'} / IoU ${iouValue.toFixed(3)}`
|
||||
}
|
||||
|
||||
function formatQualityTimestamp(value?: string | null): string {
|
||||
return value || 'n/a'
|
||||
}
|
||||
@@ -108,6 +122,9 @@ export function QualityResultsPanel({
|
||||
const selectedReferenceName = selectedQualityCheck
|
||||
? datasetNameById.get(selectedQualityCheck.reference_dataset_id) ?? selectedQualityCheck.reference_dataset_id
|
||||
: 'n/a'
|
||||
const selectedMatchEvidence = findingEvidenceList(selectedQualityCheck, 'match_evidence')
|
||||
const selectedFalsePositiveEvidence = findingEvidenceList(selectedQualityCheck, 'false_positive_evidence')
|
||||
const selectedFalseNegativeEvidence = findingEvidenceList(selectedQualityCheck, 'false_negative_evidence')
|
||||
const qualityStatuses = useMemo(() => Array.from(new Set(qualityChecks.map((check) => check.status))).sort(), [qualityChecks])
|
||||
const qualityTypes = useMemo(() => Array.from(new Set(qualityChecks.map((check) => check.check_type))).sort(), [qualityChecks])
|
||||
const filteredQualityChecks = useMemo(
|
||||
@@ -239,6 +256,44 @@ export function QualityResultsPanel({
|
||||
<p>Use the candidate and reference datasets as map layers for spatial review.</p>
|
||||
</div>
|
||||
</div>
|
||||
<div className="quality-feature-evidence-grid" aria-label="Feature-level QA/QC evidence">
|
||||
<div>
|
||||
<span>Matched feature ids</span>
|
||||
{selectedMatchEvidence.length > 0 ? (
|
||||
<ul>
|
||||
{selectedMatchEvidence.slice(0, 6).map((item, index) => (
|
||||
<li key={`${evidenceLabel(item)}-${index}`}>{evidenceLabel(item)}</li>
|
||||
))}
|
||||
</ul>
|
||||
) : (
|
||||
<p>No matched feature ids persisted for this check.</p>
|
||||
)}
|
||||
</div>
|
||||
<div>
|
||||
<span>False positive feature ids</span>
|
||||
{selectedFalsePositiveEvidence.length > 0 ? (
|
||||
<ul>
|
||||
{selectedFalsePositiveEvidence.slice(0, 6).map((item, index) => (
|
||||
<li key={`${evidenceLabel(item)}-${index}`}>{evidenceLabel(item)}</li>
|
||||
))}
|
||||
</ul>
|
||||
) : (
|
||||
<p>No false-positive feature ids persisted for this check.</p>
|
||||
)}
|
||||
</div>
|
||||
<div>
|
||||
<span>False negative feature ids</span>
|
||||
{selectedFalseNegativeEvidence.length > 0 ? (
|
||||
<ul>
|
||||
{selectedFalseNegativeEvidence.slice(0, 6).map((item, index) => (
|
||||
<li key={`${evidenceLabel(item)}-${index}`}>{evidenceLabel(item)}</li>
|
||||
))}
|
||||
</ul>
|
||||
) : (
|
||||
<p>No false-negative feature ids persisted for this check.</p>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
<div className="quality-provenance-grid">
|
||||
<div>
|
||||
<span>Parameters</span>
|
||||
|
||||
@@ -1618,6 +1618,7 @@ button.entity-card {
|
||||
|
||||
.quality-drilldown-grid,
|
||||
.quality-evidence-token-grid,
|
||||
.quality-feature-evidence-grid,
|
||||
.quality-provenance-grid {
|
||||
display: grid;
|
||||
grid-template-columns: repeat(auto-fit, minmax(12rem, 1fr));
|
||||
@@ -1641,6 +1642,7 @@ button.entity-card {
|
||||
|
||||
.quality-drilldown-grid > div,
|
||||
.quality-evidence-token-grid > div,
|
||||
.quality-feature-evidence-grid > div,
|
||||
.quality-provenance-grid > div {
|
||||
min-width: 0;
|
||||
border: 1px solid var(--line);
|
||||
@@ -1663,6 +1665,7 @@ button.entity-card {
|
||||
.quality-handoff-grid span,
|
||||
.quality-drilldown-grid span,
|
||||
.quality-evidence-token-grid span,
|
||||
.quality-feature-evidence-grid span,
|
||||
.quality-provenance-grid span,
|
||||
.quality-score-row span,
|
||||
.latest-export-card span,
|
||||
@@ -1703,7 +1706,8 @@ button.entity-card {
|
||||
}
|
||||
|
||||
.quality-drilldown-grid p,
|
||||
.quality-evidence-token-grid p {
|
||||
.quality-evidence-token-grid p,
|
||||
.quality-feature-evidence-grid p {
|
||||
margin: 0.24rem 0 0;
|
||||
color: var(--muted);
|
||||
font-size: 0.8rem;
|
||||
@@ -1711,6 +1715,17 @@ button.entity-card {
|
||||
overflow-wrap: anywhere;
|
||||
}
|
||||
|
||||
.quality-feature-evidence-grid ul {
|
||||
display: grid;
|
||||
gap: 0.28rem;
|
||||
margin: 0.38rem 0 0;
|
||||
padding-left: 1.1rem;
|
||||
color: var(--text);
|
||||
font-size: 0.78rem;
|
||||
line-height: 1.35;
|
||||
overflow-wrap: anywhere;
|
||||
}
|
||||
|
||||
.quality-provenance-pre {
|
||||
max-height: 11rem;
|
||||
margin: 0.38rem 0 0;
|
||||
|
||||
Reference in New Issue
Block a user