Files
geointel/backend/tests/test_accuracy_phase4_evaluator_hardening.py
T

398 lines
13 KiB
Python

from __future__ import annotations
import copy
import hashlib
import json
import math
import sys
from pathlib import Path
import pytest
ROOT = Path(__file__).resolve().parents[2]
SCRIPTS = ROOT / "scripts"
if str(SCRIPTS) not in sys.path:
sys.path.insert(0, str(SCRIPTS))
from accuracy_phase4_evaluator import ( # noqa: E402
TASKS,
canonical_hash,
count_metrics,
detection_ap,
evaluate_cases,
evaluate_object_detection,
evaluate_footprint_segmentation,
evaluate_raster_classification,
evaluate_terrain,
evaluate_validation,
evaluate_vector_comparison,
subgroup_report,
task_inventory,
)
METADATA = {
"region": "flanders",
"municipality": "Mol",
"urbanity": "urban",
"object_size": "medium",
"source": "synthetic-source",
"sensor": "synthetic-sensor",
"resolution_m": 0.25,
"season": "summer",
"date": "2026-01-01",
"vegetation": "partial",
"occlusion": "none",
"difficulty": "normal",
}
def lineage(sample_id: str) -> dict:
return {
"reference": {
"source_id": f"synthetic:{sample_id}:reference",
"source_version": "1",
"derivation": "hand_authored_contract_fixture",
},
"prediction": {
"source_id": f"synthetic:{sample_id}:prediction",
"source_version": "1",
"derivation": "hand_authored_fixed_output",
},
}
def detection_case(sample_id: str = "det-1") -> dict:
return {
"sample_id": sample_id,
"task": "object_detection",
"split": "test",
"metadata": copy.deepcopy(METADATA),
"config": {"confidence_threshold": 0.5, "match_iou": 0.5},
"lineage": lineage(sample_id),
"classes": ["building", "tank"],
"references": [{"id": "r-building", "class": "building", "bbox": [0, 0, 4, 4]}],
"predictions": [
{
"id": "p-building",
"class": "building",
"bbox": [0, 0, 4, 4],
"confidence": 0.8,
},
{
"id": "p-filtered",
"class": "building",
"bbox": [10, 10, 12, 12],
"confidence": 0.2,
},
],
}
def raster_case(sample_id: str = "raster-1") -> dict:
reference_side = {
"crs": "EPSG:31370",
"transform": [1, 0, 100000, 0, -1, 200000],
"shape": [2, 2],
"nodata": -9999,
"mask": [[True, True], [True, True]],
}
return {
"sample_id": sample_id,
"task": "raster_classification",
"split": "test",
"metadata": copy.deepcopy(METADATA),
"config": {},
"lineage": lineage(sample_id),
"classes": [0, 1],
"references": [[0, 1], [1, 0]],
"predictions": [[0, 1], [1, 0]],
"raster_context": {
"reference": reference_side,
"prediction": copy.deepcopy(reference_side),
},
}
def polygon_case(task: str = "vector_comparison") -> dict:
sample_id = f"{task}-1"
config = {"match_iou": 0.5}
if task == "footprint_segmentation":
config["boundary_tolerance_m"] = 1.0
polygon = [
[100000, 200000],
[100010, 200000],
[100010, 200010],
[100000, 200010],
[100000, 200000],
]
return {
"sample_id": sample_id,
"task": task,
"split": "test",
"metadata": copy.deepcopy(METADATA),
"config": config,
"lineage": lineage(sample_id),
"classes": ["building"],
"spatial_context": {
"crs": "EPSG:31370",
"coordinate_units": "m",
"metric": True,
},
"references": [{"id": "reference", "class": "building", "polygon": polygon}],
"predictions": [{"id": "prediction", "class": "building", "polygon": polygon}],
}
def test_raw_evidence_and_hashes_are_exact_and_recomputable(tmp_path: Path) -> None:
case = detection_case()
portfolio = {
"schema_version": 2,
"portfolio_id": "synthetic-hardening-test",
"portfolio_lineage": {
"origin": "repository_fixture",
"source_path": "synthetic.json",
"version": "1",
},
"split_roles": ["test"],
"selection_policy": "Fixed before evaluation; no selection.",
"claim_boundary": "Synthetic evaluator test; not product accuracy.",
"protected_policy": {"threshold_selection_allowed": False},
"cases": [case],
}
path = tmp_path / "portfolio.json"
path.write_text(
json.dumps(portfolio, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
report = evaluate_cases(path, {case["sample_id"]})
raw = report["results"][0]["raw"]
assert raw["references"] == case["references"]
assert raw["predictions_pre_filter"] == case["predictions"]
assert raw["predictions_post_filter"] == case["predictions"][:1]
assert raw["config"] == case["config"]
assert raw["split"] == "test"
assert raw["input_lineage"] == case["lineage"]
assert raw["portfolio_lineage"]["declared"] == portfolio["portfolio_lineage"]
assert raw["hashes"]["case_input_canonical_json_sha256"] == canonical_hash(case)
assert raw["hashes"]["references_canonical_json_sha256"] == canonical_hash(
case["references"]
)
assert (
report["portfolio_file_sha256"] == hashlib.sha256(path.read_bytes()).hexdigest()
)
assert report["portfolio_canonical_json_sha256"] == canonical_hash(portfolio)
assert report["results_canonical_json_sha256"] == canonical_hash(report["results"])
high_threshold = next(
row
for row in report["results"][0]["metrics"]["coverage_risk"]
if row["threshold"] == 0.9
)
assert high_threshold["retained_prediction_coverage"] == 0.0
assert high_threshold["reference_coverage"] == 0.0
assert high_threshold["false_negative_count"] == 1
assert high_threshold["risk"] == 1.0
challenge_exposed = copy.deepcopy(portfolio)
challenge_exposed["challenge_labels"] = []
path.write_text(json.dumps(challenge_exposed, ensure_ascii=False), encoding="utf-8")
with pytest.raises(ValueError, match="Challenge cases and labels"):
evaluate_cases(path, {case["sample_id"]})
def test_ap_ties_use_stable_ids_and_matching_is_class_aware() -> None:
references = [{"id": "r", "class": "building", "bbox": [0, 0, 4, 4]}]
predictions = [
{
"id": "z-true",
"class": "building",
"bbox": [0, 0, 4, 4],
"confidence": 0.8,
},
{
"id": "a-false",
"class": "building",
"bbox": [10, 10, 12, 12],
"confidence": 0.8,
},
]
forward = detection_ap(predictions, references, 0.5)
reverse = detection_ap(list(reversed(predictions)), references, 0.5)
assert forward == reverse == pytest.approx(0.5)
wrong_class = copy.deepcopy(predictions)
wrong_class[1] = {
"id": "a-tank",
"class": "tank",
"bbox": [0, 0, 4, 4],
"confidence": 0.95,
}
assert detection_ap(wrong_class, references, 0.5) == pytest.approx(0.5)
def test_raster_requires_exact_rectangular_alignment_masks_nodata_and_classes() -> None:
invalid_case = detection_case("invalid-class")
invalid_case["predictions"][0]["class"] = "road"
with pytest.raises(ValueError, match="outside the declared ontology"):
evaluate_object_detection(invalid_case)
valid = raster_case()
valid["predictions"][0][1] = -9999
valid["raster_context"]["prediction"]["mask"][0][1] = False
result = evaluate_raster_classification(valid)
assert result["metrics"]["prediction_coverage"] == pytest.approx(0.75)
assert result["metrics"]["per_class"]["1"]["false_negative"] == 1
jagged = raster_case("jagged")
jagged["predictions"][1].pop()
with pytest.raises(ValueError, match="exactly rectangular"):
evaluate_raster_classification(jagged)
missing_metadata = raster_case("missing-metadata")
del missing_metadata["raster_context"]["prediction"]["crs"]
with pytest.raises(ValueError, match="missing"):
evaluate_raster_classification(missing_metadata)
shifted = raster_case("shifted")
shifted["raster_context"]["prediction"]["transform"][2] += 1
with pytest.raises(ValueError, match="affine alignment differs"):
evaluate_raster_classification(shifted)
invalid_class = raster_case("invalid-class")
invalid_class["predictions"][0][0] = 3
with pytest.raises(ValueError, match="prediction class outside ontology"):
evaluate_raster_classification(invalid_class)
invalid_nodata = raster_case("invalid-nodata")
invalid_nodata["predictions"][0][0] = -9999
with pytest.raises(ValueError, match="marks nodata as valid"):
evaluate_raster_classification(invalid_nodata)
def test_polygon_metrics_require_valid_geometry_projected_crs_and_metres() -> None:
assert evaluate_vector_comparison(polygon_case())["metrics"]["f1"] == 1.0
assert (
evaluate_footprint_segmentation(polygon_case("footprint_segmentation"))[
"metrics"
]["mean_iou"]
== 1.0
)
geographic = polygon_case()
geographic["spatial_context"]["crs"] = "EPSG:4326"
with pytest.raises(ValueError, match="projected CRS"):
evaluate_vector_comparison(geographic)
wrong_units = polygon_case()
wrong_units["spatial_context"]["coordinate_units"] = "degree"
with pytest.raises(ValueError, match="must be 'm'"):
evaluate_vector_comparison(wrong_units)
bowtie = polygon_case()
bowtie["predictions"][0]["polygon"] = [
[100000, 200000],
[100010, 200010],
[100010, 200000],
[100000, 200010],
[100000, 200000],
]
with pytest.raises(ValueError, match="positive-area and valid"):
evaluate_vector_comparison(bowtie)
def test_terrain_rejects_non_finite_and_validation_counts_only_critical_misses() -> (
None
):
terrain = {
"sample_id": "terrain",
"task": "terrain_interpretation",
"split": "test",
"metadata": copy.deepcopy(METADATA),
"config": {},
"lineage": lineage("terrain"),
"units": "m_TAW",
"references": [1.0, 2.0],
"predictions": [1.1, None],
}
assert evaluate_terrain(terrain)["metrics"]["coverage"] == 0.5
for field, value in (("references", math.nan), ("predictions", math.inf)):
invalid = copy.deepcopy(terrain)
invalid[field][0] = value
with pytest.raises(ValueError, match="finite number"):
evaluate_terrain(invalid)
validation = {
"sample_id": "validation",
"task": "geospatial_data_validation",
"split": "test",
"metadata": copy.deepcopy(METADATA),
"config": {},
"lineage": lineage("validation"),
"expected_anomalies": [{"code": "D-MAJOR", "severity": "major"}],
"observed_anomalies": [],
}
assert (
evaluate_validation(validation)["metrics"]["blocker_or_critical_miss_count"]
== 0
)
validation["expected_anomalies"].append(
{"code": "D-CRITICAL", "severity": "critical"}
)
assert (
evaluate_validation(validation)["metrics"]["blocker_or_critical_miss_count"]
== 1
)
validation["expected_anomalies"] = ["D-NO-SEVERITY"]
with pytest.raises(ValueError, match="include code and severity"):
evaluate_validation(validation)
def _subgroup_result(region: str, tp: int, fp: int, fn: int) -> dict:
metadata = copy.deepcopy(METADATA)
metadata["region"] = region
return {
"task": "object_detection",
"metadata": metadata,
"metrics": {**count_metrics(tp, fp, fn), "ap50": 0.5, "ap50_95": 0.4},
"failures": [],
}
def test_subgroups_report_task_metrics_support_ci_and_worst_stratum() -> None:
results = [
*[_subgroup_result("strong", 10, 0, 0) for _ in range(5)],
*[_subgroup_result("weak", 1, 4, 4) for _ in range(5)],
]
report = subgroup_report(results)
region = report["dimensions"]["region"]
weak = region["strata"]["weak"]["task_metrics"]["object_detection"]
assert weak["status"] == "evaluable"
assert weak["case_support"] == 5
assert weak["micro"]["precision_ci95_wilson"]["status"] == "computed"
assert weak["macro"]["f1_case_support"] == 5
assert region["worst_stratum_by_task"]["object_detection"]["stratum"] == "weak"
insufficient = subgroup_report([_subgroup_result("thin", 1, 0, 0)])
thin = insufficient["dimensions"]["region"]["strata"]["thin"]
assert thin["task_metrics"]["object_detection"]["status"] == "insufficient_support"
assert thin["release_gate_status"] == "not_evaluable"
assert insufficient["overall_status"] == "not_evaluable"
def test_capability_inventory_is_comprehensive_and_honest() -> None:
inventory = task_inventory()
assert {item["task"] for item in inventory} == TASKS
assert len(inventory) >= 15
assert all(item["implementation_paths"] for item in inventory)
assert all(item["suitable_metrics"] for item in inventory)
assistant = next(
item
for item in inventory
if item["capability_id"] == "geo_assistant_orchestration"
)
assert assistant["evaluation_status"].startswith("no_independent_accuracy_score")