feat(accuracy): build hardened phase 4 evaluation harness
This commit is contained in:
@@ -25,6 +25,12 @@ build/
|
||||
/artifacts/evidence/accuracy/*
|
||||
!/artifacts/evidence/accuracy/P1/
|
||||
!/artifacts/evidence/accuracy/P1/**
|
||||
!/artifacts/evidence/accuracy/P3/
|
||||
!/artifacts/evidence/accuracy/P3/**
|
||||
!/artifacts/evidence/accuracy/P4/
|
||||
/artifacts/evidence/accuracy/P4/*
|
||||
!/artifacts/evidence/accuracy/P4/reference-harness-v2/
|
||||
!/artifacts/evidence/accuracy/P4/reference-harness-v2/**
|
||||
!/artifacts/evidence/accuracy/P2/
|
||||
!/artifacts/evidence/accuracy/P2/**
|
||||
/.cache/
|
||||
|
||||
@@ -0,0 +1,240 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
import hashlib
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
SCRIPTS = ROOT / "scripts"
|
||||
if str(SCRIPTS) not in sys.path:
|
||||
sys.path.insert(0, str(SCRIPTS))
|
||||
|
||||
from accuracy_phase4_evaluator import evaluate_cases # noqa: E402
|
||||
from generate_accuracy_phase4_splits import ( # noqa: E402
|
||||
LeakageError,
|
||||
assert_training_inputs_safe,
|
||||
build_manifests,
|
||||
)
|
||||
from run_accuracy_phase4_benchmark import ( # noqa: E402
|
||||
EvidenceConflictError,
|
||||
build_release_gate_report,
|
||||
canonical_golden_baseline,
|
||||
firewall_contract_checks,
|
||||
run_workflow,
|
||||
)
|
||||
|
||||
|
||||
SOURCE = ROOT / "fixtures/accuracy/p4/split-source-manifest.json"
|
||||
CASES = ROOT / "fixtures/accuracy/p4/protected-baseline-cases.json"
|
||||
|
||||
|
||||
def load_source() -> dict:
|
||||
return json.loads(SOURCE.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def evaluation_inputs() -> tuple[dict, dict, dict, dict]:
|
||||
source = load_source()
|
||||
development, protected, leakage = build_manifests(source)
|
||||
split_result = {
|
||||
"development": development,
|
||||
"protected": protected,
|
||||
"leakage": leakage,
|
||||
}
|
||||
allowed = {
|
||||
item["sample_id"]
|
||||
for item in protected["samples"]
|
||||
if item["split"] in {"test", "background-test"}
|
||||
}
|
||||
evaluation = evaluate_cases(CASES, allowed)
|
||||
portfolio = json.loads(CASES.read_text(encoding="utf-8"))
|
||||
firewall = firewall_contract_checks(split_result, CASES)
|
||||
return split_result, evaluation, portfolio, firewall
|
||||
|
||||
|
||||
def test_split_fixture_is_order_independent_and_has_all_roles() -> None:
|
||||
source = load_source()
|
||||
development, protected, leakage = build_manifests(source)
|
||||
reversed_source = copy.deepcopy(source)
|
||||
reversed_source["samples"].reverse()
|
||||
reversed_development, reversed_protected, reversed_leakage = build_manifests(
|
||||
reversed_source
|
||||
)
|
||||
|
||||
assert leakage["status"] == "pass"
|
||||
assert leakage["finding_count"] == 0
|
||||
assert leakage["split_counts"] == {
|
||||
"background-test": 2,
|
||||
"calibration": 2,
|
||||
"challenge": 4,
|
||||
"test": 7,
|
||||
"train": 3,
|
||||
"val": 3,
|
||||
}
|
||||
assert reversed_development["manifest_sha256"] == development["manifest_sha256"]
|
||||
assert reversed_protected["manifest_sha256"] == protected["manifest_sha256"]
|
||||
assert reversed_leakage == leakage
|
||||
|
||||
|
||||
def test_training_firewall_rejects_non_train_and_protected_lineage() -> None:
|
||||
development, protected, leakage = build_manifests(load_source())
|
||||
assert leakage["status"] == "pass"
|
||||
train = [item for item in development["samples"] if item["split"] == "train"]
|
||||
validation = next(item for item in development["samples"] if item["split"] == "val")
|
||||
|
||||
assert_training_inputs_safe([], train, protected)
|
||||
with pytest.raises(LeakageError, match="non_train_role"):
|
||||
assert_training_inputs_safe([], [validation], protected)
|
||||
with pytest.raises(LeakageError, match="protected_identity"):
|
||||
disguised = dict(train[0])
|
||||
disguised["source_family"] = protected["samples"][0]["source_family"]
|
||||
assert_training_inputs_safe([], [disguised], protected)
|
||||
with pytest.raises(LeakageError, match="protected_path"):
|
||||
assert_training_inputs_safe([CASES], [], protected)
|
||||
|
||||
|
||||
def test_task_evaluator_retains_exact_raw_inputs_metrics_and_failures() -> None:
|
||||
_split_result, report, _portfolio, _firewall = evaluation_inputs()
|
||||
|
||||
assert report["task_count"] == 7
|
||||
assert report["case_count"] == 9
|
||||
assert len(report["task_inventory"]) >= 15
|
||||
assert len(report["failures"]) == 11
|
||||
assert report["subgroups"]["overall_status"] == "not_evaluable"
|
||||
assert all(
|
||||
{
|
||||
"references",
|
||||
"predictions_pre_filter",
|
||||
"predictions_post_filter",
|
||||
"config",
|
||||
"input_lineage",
|
||||
"portfolio_lineage",
|
||||
}
|
||||
<= set(item["raw"])
|
||||
for item in report["results"]
|
||||
)
|
||||
detection = next(
|
||||
item
|
||||
for item in report["results"]
|
||||
if item["sample_id"] == "det-test-flanders-urban"
|
||||
)
|
||||
assert len(detection["raw"]["predictions_pre_filter"]) == 4
|
||||
assert len(detection["raw"]["predictions_post_filter"]) == 3
|
||||
assert detection["metrics"]["true_positive"] == 2
|
||||
assert detection["metrics"]["false_positive"] == 1
|
||||
assert detection["metrics"]["ap50"] is not None
|
||||
empty = next(
|
||||
item
|
||||
for item in report["results"]
|
||||
if item["sample_id"] == "background-test-pure-empty"
|
||||
)
|
||||
assert empty["metrics"]["precision"] is None
|
||||
assert empty["metrics"]["recall"] is None
|
||||
assert empty["metrics"]["f1"] is None
|
||||
|
||||
|
||||
def test_product_prerequisites_cannot_pass_without_executed_baseline() -> None:
|
||||
split_result, evaluation, portfolio, firewall = evaluation_inputs()
|
||||
product_gates = {
|
||||
"active_model_available_and_hash_verified": {"status": "pass"},
|
||||
"authoritative_reference_portfolio_available": {"status": "pass"},
|
||||
"human_review_complete": {"status": "pass"},
|
||||
"split_independence": {"status": "pass"},
|
||||
"phase3_leakage_resolved": {"status": "pass"},
|
||||
"protected_storage_isolation": {"status": "pass"},
|
||||
"executed_product_incumbent_baseline": {
|
||||
"status": "not_evaluable",
|
||||
"reason": "no raw active-model inference",
|
||||
},
|
||||
"representative_product_subgroup_support": {"status": "pass"},
|
||||
}
|
||||
report = build_release_gate_report(
|
||||
split_result,
|
||||
evaluation,
|
||||
portfolio,
|
||||
canonical_golden_baseline(),
|
||||
firewall,
|
||||
product_gates,
|
||||
)
|
||||
|
||||
assert report["local_harness_status"] == "pass"
|
||||
assert report["product_benchmark_status"] == "not_evaluable"
|
||||
assert report["status"] == "not_evaluable"
|
||||
assert report["phase_decision"] == "blocked"
|
||||
|
||||
|
||||
def test_one_workflow_is_byte_reproducible_complete_and_fail_closed(
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
output = tmp_path / "p4"
|
||||
first = run_workflow(ROOT, output)
|
||||
first_bytes = {path.name: path.read_bytes() for path in output.glob("*.json")}
|
||||
second = run_workflow(ROOT, output)
|
||||
second_bytes = {path.name: path.read_bytes() for path in output.glob("*.json")}
|
||||
|
||||
assert first == second
|
||||
assert first_bytes == second_bytes
|
||||
assert first["local_harness_status"] == "pass"
|
||||
assert first["product_benchmark_status"] == "not_evaluable"
|
||||
assert first["status"] == "not_evaluable"
|
||||
assert first["phase4_done"] is False
|
||||
assert first["phase5_ready"] is False
|
||||
required = {
|
||||
"acceptance-gates.json",
|
||||
"aoi-metrics.json",
|
||||
"baseline-raw-predictions.json",
|
||||
"benchmark-manifest.json",
|
||||
"calibration-metrics.json",
|
||||
"candidate-vs-incumbent.json",
|
||||
"development-split-manifest.json",
|
||||
"error-taxonomy.json",
|
||||
"evaluation-contract.json",
|
||||
"failure-gallery.json",
|
||||
"generation-status.json",
|
||||
"human-review-summary.json",
|
||||
"input-manifest.json",
|
||||
"latency-and-reliability.json",
|
||||
"leakage-gate-report.json",
|
||||
"metric-report.json",
|
||||
"object-metrics.json",
|
||||
"protected-split-manifest.json",
|
||||
"reference-implementation-baseline.json",
|
||||
"release-gate-report.json",
|
||||
"split-and-leakage-audit.json",
|
||||
"stratified-metrics.json",
|
||||
"tile-metrics.json",
|
||||
"workflow-summary.json",
|
||||
"evidence-manifest.json",
|
||||
}
|
||||
assert required == set(first_bytes)
|
||||
manifest = json.loads(
|
||||
(output / "evidence-manifest.json").read_text(encoding="utf-8")
|
||||
)
|
||||
assert manifest["artifact_count"] == len(required) - 1
|
||||
for item in manifest["artifacts"]:
|
||||
path = output / item["path"]
|
||||
assert path.stat().st_size == item["size_bytes"]
|
||||
assert hashlib.sha256(path.read_bytes()).hexdigest() == item["sha256"]
|
||||
gates = json.loads(
|
||||
(output / "release-gate-report.json").read_text(encoding="utf-8")
|
||||
)
|
||||
assert gates["promotion_allowed"] is False
|
||||
assert all(item["status"] == "pass" for item in gates["local_gates"].values())
|
||||
assert {item["status"] for item in gates["product_gates"].values()} <= {
|
||||
"pass",
|
||||
"fail",
|
||||
"not_evaluable",
|
||||
}
|
||||
|
||||
|
||||
def test_immutable_workflow_refuses_to_replace_changed_evidence(tmp_path: Path) -> None:
|
||||
output = tmp_path / "p4"
|
||||
run_workflow(ROOT, output)
|
||||
(output / "workflow-summary.json").write_text("{}\n", encoding="utf-8")
|
||||
|
||||
with pytest.raises(EvidenceConflictError, match="Refusing to overwrite"):
|
||||
run_workflow(ROOT, output)
|
||||
@@ -0,0 +1,397 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
import hashlib
|
||||
import json
|
||||
import math
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
SCRIPTS = ROOT / "scripts"
|
||||
if str(SCRIPTS) not in sys.path:
|
||||
sys.path.insert(0, str(SCRIPTS))
|
||||
|
||||
from accuracy_phase4_evaluator import ( # noqa: E402
|
||||
TASKS,
|
||||
canonical_hash,
|
||||
count_metrics,
|
||||
detection_ap,
|
||||
evaluate_cases,
|
||||
evaluate_object_detection,
|
||||
evaluate_footprint_segmentation,
|
||||
evaluate_raster_classification,
|
||||
evaluate_terrain,
|
||||
evaluate_validation,
|
||||
evaluate_vector_comparison,
|
||||
subgroup_report,
|
||||
task_inventory,
|
||||
)
|
||||
|
||||
|
||||
METADATA = {
|
||||
"region": "flanders",
|
||||
"municipality": "Mol",
|
||||
"urbanity": "urban",
|
||||
"object_size": "medium",
|
||||
"source": "synthetic-source",
|
||||
"sensor": "synthetic-sensor",
|
||||
"resolution_m": 0.25,
|
||||
"season": "summer",
|
||||
"date": "2026-01-01",
|
||||
"vegetation": "partial",
|
||||
"occlusion": "none",
|
||||
"difficulty": "normal",
|
||||
}
|
||||
|
||||
|
||||
def lineage(sample_id: str) -> dict:
|
||||
return {
|
||||
"reference": {
|
||||
"source_id": f"synthetic:{sample_id}:reference",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_contract_fixture",
|
||||
},
|
||||
"prediction": {
|
||||
"source_id": f"synthetic:{sample_id}:prediction",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_fixed_output",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def detection_case(sample_id: str = "det-1") -> dict:
|
||||
return {
|
||||
"sample_id": sample_id,
|
||||
"task": "object_detection",
|
||||
"split": "test",
|
||||
"metadata": copy.deepcopy(METADATA),
|
||||
"config": {"confidence_threshold": 0.5, "match_iou": 0.5},
|
||||
"lineage": lineage(sample_id),
|
||||
"classes": ["building", "tank"],
|
||||
"references": [{"id": "r-building", "class": "building", "bbox": [0, 0, 4, 4]}],
|
||||
"predictions": [
|
||||
{
|
||||
"id": "p-building",
|
||||
"class": "building",
|
||||
"bbox": [0, 0, 4, 4],
|
||||
"confidence": 0.8,
|
||||
},
|
||||
{
|
||||
"id": "p-filtered",
|
||||
"class": "building",
|
||||
"bbox": [10, 10, 12, 12],
|
||||
"confidence": 0.2,
|
||||
},
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def raster_case(sample_id: str = "raster-1") -> dict:
|
||||
reference_side = {
|
||||
"crs": "EPSG:31370",
|
||||
"transform": [1, 0, 100000, 0, -1, 200000],
|
||||
"shape": [2, 2],
|
||||
"nodata": -9999,
|
||||
"mask": [[True, True], [True, True]],
|
||||
}
|
||||
return {
|
||||
"sample_id": sample_id,
|
||||
"task": "raster_classification",
|
||||
"split": "test",
|
||||
"metadata": copy.deepcopy(METADATA),
|
||||
"config": {},
|
||||
"lineage": lineage(sample_id),
|
||||
"classes": [0, 1],
|
||||
"references": [[0, 1], [1, 0]],
|
||||
"predictions": [[0, 1], [1, 0]],
|
||||
"raster_context": {
|
||||
"reference": reference_side,
|
||||
"prediction": copy.deepcopy(reference_side),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def polygon_case(task: str = "vector_comparison") -> dict:
|
||||
sample_id = f"{task}-1"
|
||||
config = {"match_iou": 0.5}
|
||||
if task == "footprint_segmentation":
|
||||
config["boundary_tolerance_m"] = 1.0
|
||||
polygon = [
|
||||
[100000, 200000],
|
||||
[100010, 200000],
|
||||
[100010, 200010],
|
||||
[100000, 200010],
|
||||
[100000, 200000],
|
||||
]
|
||||
return {
|
||||
"sample_id": sample_id,
|
||||
"task": task,
|
||||
"split": "test",
|
||||
"metadata": copy.deepcopy(METADATA),
|
||||
"config": config,
|
||||
"lineage": lineage(sample_id),
|
||||
"classes": ["building"],
|
||||
"spatial_context": {
|
||||
"crs": "EPSG:31370",
|
||||
"coordinate_units": "m",
|
||||
"metric": True,
|
||||
},
|
||||
"references": [{"id": "reference", "class": "building", "polygon": polygon}],
|
||||
"predictions": [{"id": "prediction", "class": "building", "polygon": polygon}],
|
||||
}
|
||||
|
||||
|
||||
def test_raw_evidence_and_hashes_are_exact_and_recomputable(tmp_path: Path) -> None:
|
||||
case = detection_case()
|
||||
portfolio = {
|
||||
"schema_version": 2,
|
||||
"portfolio_id": "synthetic-hardening-test",
|
||||
"portfolio_lineage": {
|
||||
"origin": "repository_fixture",
|
||||
"source_path": "synthetic.json",
|
||||
"version": "1",
|
||||
},
|
||||
"split_roles": ["test"],
|
||||
"selection_policy": "Fixed before evaluation; no selection.",
|
||||
"claim_boundary": "Synthetic evaluator test; not product accuracy.",
|
||||
"protected_policy": {"threshold_selection_allowed": False},
|
||||
"cases": [case],
|
||||
}
|
||||
path = tmp_path / "portfolio.json"
|
||||
path.write_text(
|
||||
json.dumps(portfolio, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
report = evaluate_cases(path, {case["sample_id"]})
|
||||
raw = report["results"][0]["raw"]
|
||||
|
||||
assert raw["references"] == case["references"]
|
||||
assert raw["predictions_pre_filter"] == case["predictions"]
|
||||
assert raw["predictions_post_filter"] == case["predictions"][:1]
|
||||
assert raw["config"] == case["config"]
|
||||
assert raw["split"] == "test"
|
||||
assert raw["input_lineage"] == case["lineage"]
|
||||
assert raw["portfolio_lineage"]["declared"] == portfolio["portfolio_lineage"]
|
||||
assert raw["hashes"]["case_input_canonical_json_sha256"] == canonical_hash(case)
|
||||
assert raw["hashes"]["references_canonical_json_sha256"] == canonical_hash(
|
||||
case["references"]
|
||||
)
|
||||
assert (
|
||||
report["portfolio_file_sha256"] == hashlib.sha256(path.read_bytes()).hexdigest()
|
||||
)
|
||||
assert report["portfolio_canonical_json_sha256"] == canonical_hash(portfolio)
|
||||
assert report["results_canonical_json_sha256"] == canonical_hash(report["results"])
|
||||
high_threshold = next(
|
||||
row
|
||||
for row in report["results"][0]["metrics"]["coverage_risk"]
|
||||
if row["threshold"] == 0.9
|
||||
)
|
||||
assert high_threshold["retained_prediction_coverage"] == 0.0
|
||||
assert high_threshold["reference_coverage"] == 0.0
|
||||
assert high_threshold["false_negative_count"] == 1
|
||||
assert high_threshold["risk"] == 1.0
|
||||
|
||||
challenge_exposed = copy.deepcopy(portfolio)
|
||||
challenge_exposed["challenge_labels"] = []
|
||||
path.write_text(json.dumps(challenge_exposed, ensure_ascii=False), encoding="utf-8")
|
||||
with pytest.raises(ValueError, match="Challenge cases and labels"):
|
||||
evaluate_cases(path, {case["sample_id"]})
|
||||
|
||||
|
||||
def test_ap_ties_use_stable_ids_and_matching_is_class_aware() -> None:
|
||||
references = [{"id": "r", "class": "building", "bbox": [0, 0, 4, 4]}]
|
||||
predictions = [
|
||||
{
|
||||
"id": "z-true",
|
||||
"class": "building",
|
||||
"bbox": [0, 0, 4, 4],
|
||||
"confidence": 0.8,
|
||||
},
|
||||
{
|
||||
"id": "a-false",
|
||||
"class": "building",
|
||||
"bbox": [10, 10, 12, 12],
|
||||
"confidence": 0.8,
|
||||
},
|
||||
]
|
||||
forward = detection_ap(predictions, references, 0.5)
|
||||
reverse = detection_ap(list(reversed(predictions)), references, 0.5)
|
||||
assert forward == reverse == pytest.approx(0.5)
|
||||
|
||||
wrong_class = copy.deepcopy(predictions)
|
||||
wrong_class[1] = {
|
||||
"id": "a-tank",
|
||||
"class": "tank",
|
||||
"bbox": [0, 0, 4, 4],
|
||||
"confidence": 0.95,
|
||||
}
|
||||
assert detection_ap(wrong_class, references, 0.5) == pytest.approx(0.5)
|
||||
|
||||
|
||||
def test_raster_requires_exact_rectangular_alignment_masks_nodata_and_classes() -> None:
|
||||
invalid_case = detection_case("invalid-class")
|
||||
invalid_case["predictions"][0]["class"] = "road"
|
||||
with pytest.raises(ValueError, match="outside the declared ontology"):
|
||||
evaluate_object_detection(invalid_case)
|
||||
|
||||
valid = raster_case()
|
||||
valid["predictions"][0][1] = -9999
|
||||
valid["raster_context"]["prediction"]["mask"][0][1] = False
|
||||
result = evaluate_raster_classification(valid)
|
||||
assert result["metrics"]["prediction_coverage"] == pytest.approx(0.75)
|
||||
assert result["metrics"]["per_class"]["1"]["false_negative"] == 1
|
||||
|
||||
jagged = raster_case("jagged")
|
||||
jagged["predictions"][1].pop()
|
||||
with pytest.raises(ValueError, match="exactly rectangular"):
|
||||
evaluate_raster_classification(jagged)
|
||||
|
||||
missing_metadata = raster_case("missing-metadata")
|
||||
del missing_metadata["raster_context"]["prediction"]["crs"]
|
||||
with pytest.raises(ValueError, match="missing"):
|
||||
evaluate_raster_classification(missing_metadata)
|
||||
|
||||
shifted = raster_case("shifted")
|
||||
shifted["raster_context"]["prediction"]["transform"][2] += 1
|
||||
with pytest.raises(ValueError, match="affine alignment differs"):
|
||||
evaluate_raster_classification(shifted)
|
||||
|
||||
invalid_class = raster_case("invalid-class")
|
||||
invalid_class["predictions"][0][0] = 3
|
||||
with pytest.raises(ValueError, match="prediction class outside ontology"):
|
||||
evaluate_raster_classification(invalid_class)
|
||||
|
||||
invalid_nodata = raster_case("invalid-nodata")
|
||||
invalid_nodata["predictions"][0][0] = -9999
|
||||
with pytest.raises(ValueError, match="marks nodata as valid"):
|
||||
evaluate_raster_classification(invalid_nodata)
|
||||
|
||||
|
||||
def test_polygon_metrics_require_valid_geometry_projected_crs_and_metres() -> None:
|
||||
assert evaluate_vector_comparison(polygon_case())["metrics"]["f1"] == 1.0
|
||||
assert (
|
||||
evaluate_footprint_segmentation(polygon_case("footprint_segmentation"))[
|
||||
"metrics"
|
||||
]["mean_iou"]
|
||||
== 1.0
|
||||
)
|
||||
|
||||
geographic = polygon_case()
|
||||
geographic["spatial_context"]["crs"] = "EPSG:4326"
|
||||
with pytest.raises(ValueError, match="projected CRS"):
|
||||
evaluate_vector_comparison(geographic)
|
||||
|
||||
wrong_units = polygon_case()
|
||||
wrong_units["spatial_context"]["coordinate_units"] = "degree"
|
||||
with pytest.raises(ValueError, match="must be 'm'"):
|
||||
evaluate_vector_comparison(wrong_units)
|
||||
|
||||
bowtie = polygon_case()
|
||||
bowtie["predictions"][0]["polygon"] = [
|
||||
[100000, 200000],
|
||||
[100010, 200010],
|
||||
[100010, 200000],
|
||||
[100000, 200010],
|
||||
[100000, 200000],
|
||||
]
|
||||
with pytest.raises(ValueError, match="positive-area and valid"):
|
||||
evaluate_vector_comparison(bowtie)
|
||||
|
||||
|
||||
def test_terrain_rejects_non_finite_and_validation_counts_only_critical_misses() -> (
|
||||
None
|
||||
):
|
||||
terrain = {
|
||||
"sample_id": "terrain",
|
||||
"task": "terrain_interpretation",
|
||||
"split": "test",
|
||||
"metadata": copy.deepcopy(METADATA),
|
||||
"config": {},
|
||||
"lineage": lineage("terrain"),
|
||||
"units": "m_TAW",
|
||||
"references": [1.0, 2.0],
|
||||
"predictions": [1.1, None],
|
||||
}
|
||||
assert evaluate_terrain(terrain)["metrics"]["coverage"] == 0.5
|
||||
for field, value in (("references", math.nan), ("predictions", math.inf)):
|
||||
invalid = copy.deepcopy(terrain)
|
||||
invalid[field][0] = value
|
||||
with pytest.raises(ValueError, match="finite number"):
|
||||
evaluate_terrain(invalid)
|
||||
|
||||
validation = {
|
||||
"sample_id": "validation",
|
||||
"task": "geospatial_data_validation",
|
||||
"split": "test",
|
||||
"metadata": copy.deepcopy(METADATA),
|
||||
"config": {},
|
||||
"lineage": lineage("validation"),
|
||||
"expected_anomalies": [{"code": "D-MAJOR", "severity": "major"}],
|
||||
"observed_anomalies": [],
|
||||
}
|
||||
assert (
|
||||
evaluate_validation(validation)["metrics"]["blocker_or_critical_miss_count"]
|
||||
== 0
|
||||
)
|
||||
validation["expected_anomalies"].append(
|
||||
{"code": "D-CRITICAL", "severity": "critical"}
|
||||
)
|
||||
assert (
|
||||
evaluate_validation(validation)["metrics"]["blocker_or_critical_miss_count"]
|
||||
== 1
|
||||
)
|
||||
validation["expected_anomalies"] = ["D-NO-SEVERITY"]
|
||||
with pytest.raises(ValueError, match="include code and severity"):
|
||||
evaluate_validation(validation)
|
||||
|
||||
|
||||
def _subgroup_result(region: str, tp: int, fp: int, fn: int) -> dict:
|
||||
metadata = copy.deepcopy(METADATA)
|
||||
metadata["region"] = region
|
||||
return {
|
||||
"task": "object_detection",
|
||||
"metadata": metadata,
|
||||
"metrics": {**count_metrics(tp, fp, fn), "ap50": 0.5, "ap50_95": 0.4},
|
||||
"failures": [],
|
||||
}
|
||||
|
||||
|
||||
def test_subgroups_report_task_metrics_support_ci_and_worst_stratum() -> None:
|
||||
results = [
|
||||
*[_subgroup_result("strong", 10, 0, 0) for _ in range(5)],
|
||||
*[_subgroup_result("weak", 1, 4, 4) for _ in range(5)],
|
||||
]
|
||||
report = subgroup_report(results)
|
||||
region = report["dimensions"]["region"]
|
||||
weak = region["strata"]["weak"]["task_metrics"]["object_detection"]
|
||||
|
||||
assert weak["status"] == "evaluable"
|
||||
assert weak["case_support"] == 5
|
||||
assert weak["micro"]["precision_ci95_wilson"]["status"] == "computed"
|
||||
assert weak["macro"]["f1_case_support"] == 5
|
||||
assert region["worst_stratum_by_task"]["object_detection"]["stratum"] == "weak"
|
||||
|
||||
insufficient = subgroup_report([_subgroup_result("thin", 1, 0, 0)])
|
||||
thin = insufficient["dimensions"]["region"]["strata"]["thin"]
|
||||
assert thin["task_metrics"]["object_detection"]["status"] == "insufficient_support"
|
||||
assert thin["release_gate_status"] == "not_evaluable"
|
||||
assert insufficient["overall_status"] == "not_evaluable"
|
||||
|
||||
|
||||
def test_capability_inventory_is_comprehensive_and_honest() -> None:
|
||||
inventory = task_inventory()
|
||||
assert {item["task"] for item in inventory} == TASKS
|
||||
assert len(inventory) >= 15
|
||||
assert all(item["implementation_paths"] for item in inventory)
|
||||
assert all(item["suitable_metrics"] for item in inventory)
|
||||
assistant = next(
|
||||
item
|
||||
for item in inventory
|
||||
if item["capability_id"] == "geo_assistant_orchestration"
|
||||
)
|
||||
assert assistant["evaluation_status"].startswith("no_independent_accuracy_score")
|
||||
@@ -0,0 +1,222 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
SCRIPTS = ROOT / "scripts"
|
||||
if str(SCRIPTS) not in sys.path:
|
||||
sys.path.insert(0, str(SCRIPTS))
|
||||
|
||||
from generate_accuracy_phase4_splits import ( # noqa: E402
|
||||
LeakageError,
|
||||
assert_training_inputs_safe,
|
||||
build_manifests,
|
||||
generate,
|
||||
)
|
||||
|
||||
|
||||
SOURCE = ROOT / "fixtures/accuracy/p4/split-source-manifest.json"
|
||||
|
||||
|
||||
def load_source() -> dict:
|
||||
return json.loads(SOURCE.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def test_normative_roles_hashes_and_source_order_are_enforced() -> None:
|
||||
source = load_source()
|
||||
development, protected, leakage = build_manifests(source)
|
||||
reversed_source = copy.deepcopy(source)
|
||||
reversed_source["samples"].reverse()
|
||||
reversed_development, reversed_protected, reversed_leakage = build_manifests(
|
||||
reversed_source
|
||||
)
|
||||
|
||||
assert leakage["status"] == "pass"
|
||||
assert leakage["finding_count"] == 0
|
||||
assert leakage["split_counts"] == {
|
||||
"background-test": 2,
|
||||
"calibration": 2,
|
||||
"challenge": 4,
|
||||
"test": 7,
|
||||
"train": 3,
|
||||
"val": 3,
|
||||
}
|
||||
assert leakage["crs_validation"] == {
|
||||
"status": "pass",
|
||||
"crs": "EPSG:31370",
|
||||
"distance_units": "m",
|
||||
}
|
||||
assert development["training_access_allowed_by_split"] == {
|
||||
"train": True,
|
||||
"val": False,
|
||||
"calibration": False,
|
||||
}
|
||||
assert protected["labels_available_by_split"]["challenge"] == "sealed_external"
|
||||
assert reversed_development["manifest_sha256"] == development["manifest_sha256"]
|
||||
assert reversed_protected["manifest_sha256"] == protected["manifest_sha256"]
|
||||
assert reversed_leakage == leakage
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("field", "expected_code"),
|
||||
[
|
||||
("group_id", "S-SPATIAL-GROUP"),
|
||||
("source_family", "S-SOURCE-FAMILY"),
|
||||
("temporal_family", "S-TEMPORAL-FAMILY"),
|
||||
("raw_image_sha256", "S-RAW-IMAGE-DUPLICATE"),
|
||||
("processed_image_sha256", "S-PROCESSED-IMAGE-DUPLICATE"),
|
||||
("label_sha256", "S-LABEL-DUPLICATE"),
|
||||
("label_geometry_hash", "S-LABEL-GEOMETRY-DUPLICATE"),
|
||||
("parent_raster_id", "S-PARENT-RASTER"),
|
||||
("acquisition_id", "S-ACQUISITION"),
|
||||
],
|
||||
)
|
||||
def test_cross_split_lineage_and_content_collisions_fail(
|
||||
field: str, expected_code: str
|
||||
) -> None:
|
||||
source = load_source()
|
||||
source["samples"][8][field] = source["samples"][0][field]
|
||||
_development, _protected, leakage = build_manifests(source)
|
||||
|
||||
assert leakage["status"] == "fail"
|
||||
assert expected_code in {item["code"] for item in leakage["findings"]}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("field", "expected_code"),
|
||||
[
|
||||
("perceptual_image_hash", "S-PERCEPTUAL-IMAGE-NEAR-DUPLICATE"),
|
||||
("label_geometry_fingerprint", "S-LABEL-GEOMETRY-NEAR-DUPLICATE"),
|
||||
],
|
||||
)
|
||||
def test_near_duplicate_fingerprints_fail(field: str, expected_code: str) -> None:
|
||||
source = load_source()
|
||||
source["samples"][8][field] = source["samples"][0][field]
|
||||
_development, _protected, leakage = build_manifests(source)
|
||||
|
||||
assert leakage["status"] == "fail"
|
||||
assert expected_code in {item["code"] for item in leakage["findings"]}
|
||||
|
||||
|
||||
def test_object_native_feature_and_spatial_collisions_fail() -> None:
|
||||
source = load_source()
|
||||
source["samples"][8]["object_ids"] = source["samples"][0]["object_ids"]
|
||||
source["samples"][9]["native_feature_ids"] = source["samples"][1][
|
||||
"native_feature_ids"
|
||||
]
|
||||
source["samples"][10]["bbox"] = source["samples"][2]["bbox"]
|
||||
_development, _protected, leakage = build_manifests(source)
|
||||
codes = {item["code"] for item in leakage["findings"]}
|
||||
|
||||
assert {"S-OBJECT-INSTANCE", "S-NATIVE-FEATURE", "S-SPATIAL-OVERLAP"} <= codes
|
||||
|
||||
|
||||
def test_non_metric_crs_and_missing_normative_role_fail_closed() -> None:
|
||||
geographic = load_source()
|
||||
geographic["crs"] = "EPSG:4326"
|
||||
with pytest.raises(LeakageError, match="projected in metres"):
|
||||
build_manifests(geographic)
|
||||
|
||||
missing = load_source()
|
||||
missing["samples"] = [
|
||||
item for item in missing["samples"] if item["split"] != "calibration"
|
||||
]
|
||||
with pytest.raises(LeakageError, match="Required splits are absent"):
|
||||
build_manifests(missing)
|
||||
|
||||
|
||||
def test_training_firewall_only_allows_train_and_binds_protected_lineage() -> None:
|
||||
development, protected, leakage = build_manifests(load_source())
|
||||
assert leakage["status"] == "pass"
|
||||
train = [item for item in development["samples"] if item["split"] == "train"]
|
||||
validation = next(item for item in development["samples"] if item["split"] == "val")
|
||||
protected_item = protected["samples"][0]
|
||||
|
||||
assert_training_inputs_safe([], train, protected)
|
||||
with pytest.raises(LeakageError, match="non_train_role"):
|
||||
assert_training_inputs_safe([], [validation], protected)
|
||||
with pytest.raises(LeakageError, match="protected_identity"):
|
||||
disguised = copy.deepcopy(train[0])
|
||||
disguised["source_family"] = protected_item["source_family"]
|
||||
assert_training_inputs_safe([], [disguised], protected)
|
||||
with pytest.raises(LeakageError, match="protected_path"):
|
||||
assert_training_inputs_safe([Path("vault/protected/test.json")], [], protected)
|
||||
|
||||
|
||||
def test_failed_generation_writes_status_but_no_consumable_manifests(
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
source = load_source()
|
||||
source["samples"][8]["group_id"] = source["samples"][0]["group_id"]
|
||||
source_path = tmp_path / "source.json"
|
||||
source_path.write_text(json.dumps(source), encoding="utf-8")
|
||||
output = tmp_path / "out"
|
||||
|
||||
with pytest.raises(LeakageError, match="Leakage gate failed"):
|
||||
generate(source_path, output)
|
||||
|
||||
status = json.loads((output / "generation-status.json").read_text(encoding="utf-8"))
|
||||
assert status["status"] == "fail"
|
||||
assert not (output / "development-split-manifest.json").exists()
|
||||
assert not (output / "protected-split-manifest.json").exists()
|
||||
|
||||
|
||||
def test_deterministic_grouped_assignment_is_stable_and_keeps_relatives_together() -> (
|
||||
None
|
||||
):
|
||||
source = load_source()
|
||||
source["assignment_mode"] = "deterministic_grouped"
|
||||
source["split_assignment"] = {
|
||||
"seed": "fixed-phase4-test-seed",
|
||||
"roles": [
|
||||
"train",
|
||||
"val",
|
||||
"calibration",
|
||||
"test",
|
||||
"background-test",
|
||||
"challenge",
|
||||
],
|
||||
"weights": {
|
||||
"train": 6,
|
||||
"val": 2,
|
||||
"calibration": 1,
|
||||
"test": 2,
|
||||
"background-test": 1,
|
||||
"challenge": 1,
|
||||
},
|
||||
"stratify_by": ["task"],
|
||||
}
|
||||
for item in source["samples"]:
|
||||
item.pop("split")
|
||||
source["samples"][1]["group_id"] = source["samples"][0]["group_id"]
|
||||
|
||||
development, protected, leakage = build_manifests(source)
|
||||
reversed_source = copy.deepcopy(source)
|
||||
reversed_source["samples"].reverse()
|
||||
reversed_development, reversed_protected, reversed_leakage = build_manifests(
|
||||
reversed_source
|
||||
)
|
||||
|
||||
assigned = {
|
||||
item["sample_id"]: item["split"]
|
||||
for item in development["samples"] + protected["samples"]
|
||||
}
|
||||
assert assigned["det-train-a"] == assigned["seg-train-a"]
|
||||
assert set(leakage["split_counts"]) == {
|
||||
"train",
|
||||
"val",
|
||||
"calibration",
|
||||
"test",
|
||||
"background-test",
|
||||
"challenge",
|
||||
}
|
||||
assert leakage["status"] == "pass"
|
||||
assert reversed_development["manifest_sha256"] == development["manifest_sha256"]
|
||||
assert reversed_protected["manifest_sha256"] == protected["manifest_sha256"]
|
||||
assert reversed_leakage == leakage
|
||||
@@ -0,0 +1,881 @@
|
||||
{
|
||||
"schema_version": 2,
|
||||
"portfolio_id": "geointel-p4-reference-harness-v2",
|
||||
"split_roles": [
|
||||
"test",
|
||||
"background-test"
|
||||
],
|
||||
"selection_policy": "No data-dependent threshold, model or gate selection is performed from protected cases.",
|
||||
"protected_policy": {
|
||||
"operating_point_selection_allowed": false,
|
||||
"diagnostic_curves_select_operating_point": false,
|
||||
"test_feedback_allowed": false,
|
||||
"threshold_selection_source": "pre_registered_configuration_only"
|
||||
},
|
||||
"portfolio_lineage": {
|
||||
"origin": "repository_fixture",
|
||||
"source_path": "fixtures/accuracy/p4/protected-baseline-cases.json",
|
||||
"version": "1"
|
||||
},
|
||||
"claim_boundary": "Synthetic deterministic reference cases validate evaluator behavior, not production model accuracy.",
|
||||
"cases": [
|
||||
{
|
||||
"sample_id": "det-test-flanders-urban",
|
||||
"split": "test",
|
||||
"task": "object_detection",
|
||||
"metadata": {
|
||||
"region": "flanders",
|
||||
"municipality": "Antwerpen",
|
||||
"urbanity": "urban",
|
||||
"object_size": "mixed",
|
||||
"source": "fixture-orthophoto",
|
||||
"sensor": "synthetic-rgb",
|
||||
"resolution_m": 0.25,
|
||||
"season": "summer",
|
||||
"occlusion": "partial",
|
||||
"vegetation": "moderate",
|
||||
"difficulty": "hard",
|
||||
"context": "dense_urban",
|
||||
"tile_edge": false,
|
||||
"label_review_state": "fixture_reviewed",
|
||||
"ood": false,
|
||||
"date": "2025-06-15"
|
||||
},
|
||||
"config": {
|
||||
"confidence_threshold": 0.5,
|
||||
"match_iou": 0.5,
|
||||
"fixed_diagnostic_risk_thresholds": [
|
||||
0,
|
||||
0.5,
|
||||
0.7,
|
||||
0.9
|
||||
]
|
||||
},
|
||||
"lineage": {
|
||||
"reference": {
|
||||
"source_id": "synthetic:det-test-flanders-urban:reference",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_contract_fixture"
|
||||
},
|
||||
"prediction": {
|
||||
"source_id": "synthetic:det-test-flanders-urban:prediction",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_fixed_output"
|
||||
}
|
||||
},
|
||||
"classes": [
|
||||
"building"
|
||||
],
|
||||
"references": [
|
||||
{
|
||||
"id": "r1",
|
||||
"class": "building",
|
||||
"bbox": [
|
||||
10,
|
||||
10,
|
||||
30,
|
||||
30
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "r2",
|
||||
"class": "building",
|
||||
"bbox": [
|
||||
50,
|
||||
50,
|
||||
80,
|
||||
80
|
||||
]
|
||||
}
|
||||
],
|
||||
"predictions": [
|
||||
{
|
||||
"id": "p1",
|
||||
"class": "building",
|
||||
"bbox": [
|
||||
10,
|
||||
10,
|
||||
30,
|
||||
30
|
||||
],
|
||||
"confidence": 0.9
|
||||
},
|
||||
{
|
||||
"id": "p2",
|
||||
"class": "building",
|
||||
"bbox": [
|
||||
52,
|
||||
52,
|
||||
80,
|
||||
80
|
||||
],
|
||||
"confidence": 0.7
|
||||
},
|
||||
{
|
||||
"id": "p3",
|
||||
"class": "building",
|
||||
"bbox": [
|
||||
85,
|
||||
85,
|
||||
95,
|
||||
95
|
||||
],
|
||||
"confidence": 0.6
|
||||
},
|
||||
{
|
||||
"id": "p4-filtered",
|
||||
"class": "building",
|
||||
"bbox": [
|
||||
2,
|
||||
80,
|
||||
7,
|
||||
86
|
||||
],
|
||||
"confidence": 0.2
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"sample_id": "seg-test-wallonia-rural",
|
||||
"split": "test",
|
||||
"task": "footprint_segmentation",
|
||||
"metadata": {
|
||||
"region": "wallonia",
|
||||
"municipality": "Namur",
|
||||
"urbanity": "rural",
|
||||
"object_size": "large",
|
||||
"source": "fixture-picc",
|
||||
"sensor": "synthetic-rgb",
|
||||
"resolution_m": 0.5,
|
||||
"season": "spring",
|
||||
"occlusion": "none",
|
||||
"vegetation": "low",
|
||||
"difficulty": "normal",
|
||||
"context": "rural_buildings",
|
||||
"tile_edge": false,
|
||||
"label_review_state": "fixture_reviewed",
|
||||
"ood": false,
|
||||
"date": "2025-04-20"
|
||||
},
|
||||
"config": {
|
||||
"match_iou": 0.5,
|
||||
"boundary_tolerance_m": 1
|
||||
},
|
||||
"lineage": {
|
||||
"reference": {
|
||||
"source_id": "synthetic:seg-test-wallonia-rural:reference",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_contract_fixture"
|
||||
},
|
||||
"prediction": {
|
||||
"source_id": "synthetic:seg-test-wallonia-rural:prediction",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_fixed_output"
|
||||
}
|
||||
},
|
||||
"spatial_context": {
|
||||
"crs": "EPSG:31370",
|
||||
"coordinate_units": "m",
|
||||
"metric": true
|
||||
},
|
||||
"classes": [
|
||||
"building"
|
||||
],
|
||||
"references": [
|
||||
{
|
||||
"id": "r1",
|
||||
"class": "building",
|
||||
"polygon": [
|
||||
[
|
||||
0,
|
||||
0
|
||||
],
|
||||
[
|
||||
10,
|
||||
0
|
||||
],
|
||||
[
|
||||
10,
|
||||
8
|
||||
],
|
||||
[
|
||||
0,
|
||||
8
|
||||
],
|
||||
[
|
||||
0,
|
||||
0
|
||||
]
|
||||
]
|
||||
}
|
||||
],
|
||||
"predictions": [
|
||||
{
|
||||
"id": "p1",
|
||||
"class": "building",
|
||||
"polygon": [
|
||||
[
|
||||
0.5,
|
||||
0.5
|
||||
],
|
||||
[
|
||||
9.5,
|
||||
0.5
|
||||
],
|
||||
[
|
||||
9.5,
|
||||
8.5
|
||||
],
|
||||
[
|
||||
0.5,
|
||||
8.5
|
||||
],
|
||||
[
|
||||
0.5,
|
||||
0.5
|
||||
]
|
||||
],
|
||||
"confidence": 0.8
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"sample_id": "raster-test-brussels-urban",
|
||||
"split": "test",
|
||||
"task": "raster_classification",
|
||||
"metadata": {
|
||||
"region": "brussels",
|
||||
"municipality": "Brussel",
|
||||
"urbanity": "urban",
|
||||
"object_size": "not_applicable",
|
||||
"source": "fixture-thematic-raster",
|
||||
"sensor": "synthetic-multispectral",
|
||||
"resolution_m": 1,
|
||||
"season": "autumn",
|
||||
"occlusion": "not_applicable",
|
||||
"vegetation": "moderate",
|
||||
"difficulty": "hard",
|
||||
"context": "dense_urban_raster",
|
||||
"tile_edge": true,
|
||||
"label_review_state": "fixture_reviewed",
|
||||
"ood": false,
|
||||
"date": "2025-10-05"
|
||||
},
|
||||
"config": {
|
||||
"class_order": [
|
||||
0,
|
||||
1,
|
||||
2
|
||||
],
|
||||
"masked_pixels_excluded": true
|
||||
},
|
||||
"lineage": {
|
||||
"reference": {
|
||||
"source_id": "synthetic:raster-test-brussels-urban:reference",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_contract_fixture"
|
||||
},
|
||||
"prediction": {
|
||||
"source_id": "synthetic:raster-test-brussels-urban:prediction",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_fixed_output"
|
||||
}
|
||||
},
|
||||
"classes": [
|
||||
0,
|
||||
1,
|
||||
2
|
||||
],
|
||||
"raster_context": {
|
||||
"reference": {
|
||||
"crs": "EPSG:31370",
|
||||
"transform": [
|
||||
1,
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
-1,
|
||||
3
|
||||
],
|
||||
"shape": [
|
||||
3,
|
||||
4
|
||||
],
|
||||
"nodata": null,
|
||||
"mask": [
|
||||
[
|
||||
true,
|
||||
true,
|
||||
true,
|
||||
true
|
||||
],
|
||||
[
|
||||
true,
|
||||
true,
|
||||
true,
|
||||
true
|
||||
],
|
||||
[
|
||||
true,
|
||||
true,
|
||||
true,
|
||||
true
|
||||
]
|
||||
]
|
||||
},
|
||||
"prediction": {
|
||||
"crs": "EPSG:31370",
|
||||
"transform": [
|
||||
1,
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
-1,
|
||||
3
|
||||
],
|
||||
"shape": [
|
||||
3,
|
||||
4
|
||||
],
|
||||
"nodata": null,
|
||||
"mask": [
|
||||
[
|
||||
true,
|
||||
true,
|
||||
true,
|
||||
true
|
||||
],
|
||||
[
|
||||
true,
|
||||
true,
|
||||
true,
|
||||
true
|
||||
],
|
||||
[
|
||||
true,
|
||||
true,
|
||||
true,
|
||||
true
|
||||
]
|
||||
]
|
||||
}
|
||||
},
|
||||
"references": [
|
||||
[
|
||||
0,
|
||||
0,
|
||||
1,
|
||||
1
|
||||
],
|
||||
[
|
||||
0,
|
||||
1,
|
||||
1,
|
||||
2
|
||||
],
|
||||
[
|
||||
2,
|
||||
2,
|
||||
1,
|
||||
0
|
||||
]
|
||||
],
|
||||
"predictions": [
|
||||
[
|
||||
0,
|
||||
0,
|
||||
1,
|
||||
2
|
||||
],
|
||||
[
|
||||
0,
|
||||
1,
|
||||
1,
|
||||
2
|
||||
],
|
||||
[
|
||||
2,
|
||||
1,
|
||||
1,
|
||||
0
|
||||
]
|
||||
]
|
||||
},
|
||||
{
|
||||
"sample_id": "vector-test-flanders-suburban",
|
||||
"split": "test",
|
||||
"task": "vector_comparison",
|
||||
"metadata": {
|
||||
"region": "flanders",
|
||||
"municipality": "Mol",
|
||||
"urbanity": "suburban",
|
||||
"object_size": "mixed",
|
||||
"source": "fixture-grb",
|
||||
"sensor": "vector",
|
||||
"resolution_m": 0.1,
|
||||
"season": "not_applicable",
|
||||
"occlusion": "not_applicable",
|
||||
"vegetation": "low",
|
||||
"difficulty": "normal",
|
||||
"context": "suburban_buildings",
|
||||
"tile_edge": false,
|
||||
"label_review_state": "fixture_reviewed",
|
||||
"ood": false,
|
||||
"date": "2025-01-15"
|
||||
},
|
||||
"config": {
|
||||
"match_iou": 0.5
|
||||
},
|
||||
"lineage": {
|
||||
"reference": {
|
||||
"source_id": "synthetic:vector-test-flanders-suburban:reference",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_contract_fixture"
|
||||
},
|
||||
"prediction": {
|
||||
"source_id": "synthetic:vector-test-flanders-suburban:prediction",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_fixed_output"
|
||||
}
|
||||
},
|
||||
"spatial_context": {
|
||||
"crs": "EPSG:31370",
|
||||
"coordinate_units": "m",
|
||||
"metric": true
|
||||
},
|
||||
"classes": [
|
||||
"building"
|
||||
],
|
||||
"references": [
|
||||
{
|
||||
"id": "r1",
|
||||
"class": "building",
|
||||
"polygon": [
|
||||
[
|
||||
0,
|
||||
0
|
||||
],
|
||||
[
|
||||
4,
|
||||
0
|
||||
],
|
||||
[
|
||||
4,
|
||||
4
|
||||
],
|
||||
[
|
||||
0,
|
||||
4
|
||||
],
|
||||
[
|
||||
0,
|
||||
0
|
||||
]
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "r2",
|
||||
"class": "building",
|
||||
"polygon": [
|
||||
[
|
||||
10,
|
||||
0
|
||||
],
|
||||
[
|
||||
14,
|
||||
0
|
||||
],
|
||||
[
|
||||
14,
|
||||
4
|
||||
],
|
||||
[
|
||||
10,
|
||||
4
|
||||
],
|
||||
[
|
||||
10,
|
||||
0
|
||||
]
|
||||
]
|
||||
}
|
||||
],
|
||||
"predictions": [
|
||||
{
|
||||
"id": "p1",
|
||||
"class": "building",
|
||||
"polygon": [
|
||||
[
|
||||
0,
|
||||
0
|
||||
],
|
||||
[
|
||||
4,
|
||||
0
|
||||
],
|
||||
[
|
||||
4,
|
||||
4
|
||||
],
|
||||
[
|
||||
0,
|
||||
4
|
||||
],
|
||||
[
|
||||
0,
|
||||
0
|
||||
]
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "p2",
|
||||
"class": "building",
|
||||
"polygon": [
|
||||
[
|
||||
20,
|
||||
0
|
||||
],
|
||||
[
|
||||
24,
|
||||
0
|
||||
],
|
||||
[
|
||||
24,
|
||||
4
|
||||
],
|
||||
[
|
||||
20,
|
||||
4
|
||||
],
|
||||
[
|
||||
20,
|
||||
0
|
||||
]
|
||||
]
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"sample_id": "change-test-wallonia-industrial",
|
||||
"split": "test",
|
||||
"task": "change_detection",
|
||||
"metadata": {
|
||||
"region": "wallonia",
|
||||
"municipality": "Liège",
|
||||
"urbanity": "industrial",
|
||||
"object_size": "mixed",
|
||||
"source": "fixture-picc-temporal",
|
||||
"sensor": "vector",
|
||||
"resolution_m": 0.5,
|
||||
"season": "multi-date",
|
||||
"occlusion": "none",
|
||||
"vegetation": "low",
|
||||
"difficulty": "hard",
|
||||
"context": "industrial_change",
|
||||
"tile_edge": false,
|
||||
"label_review_state": "fixture_reviewed",
|
||||
"ood": false,
|
||||
"date": "2024-05-01/2025-05-01"
|
||||
},
|
||||
"config": {
|
||||
"match_iou": 0.5
|
||||
},
|
||||
"lineage": {
|
||||
"reference": {
|
||||
"source_id": "synthetic:change-test-wallonia-industrial:reference",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_contract_fixture"
|
||||
},
|
||||
"prediction": {
|
||||
"source_id": "synthetic:change-test-wallonia-industrial:prediction",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_fixed_output"
|
||||
}
|
||||
},
|
||||
"classes": [
|
||||
"added",
|
||||
"removed"
|
||||
],
|
||||
"references": [
|
||||
{
|
||||
"id": "r1",
|
||||
"class": "added",
|
||||
"bbox": [
|
||||
0,
|
||||
0,
|
||||
5,
|
||||
5
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "r2",
|
||||
"class": "removed",
|
||||
"bbox": [
|
||||
10,
|
||||
0,
|
||||
15,
|
||||
5
|
||||
]
|
||||
}
|
||||
],
|
||||
"predictions": [
|
||||
{
|
||||
"id": "p1",
|
||||
"class": "added",
|
||||
"bbox": [
|
||||
0,
|
||||
0,
|
||||
5,
|
||||
5
|
||||
],
|
||||
"confidence": 0.85
|
||||
},
|
||||
{
|
||||
"id": "p2",
|
||||
"class": "removed",
|
||||
"bbox": [
|
||||
20,
|
||||
0,
|
||||
25,
|
||||
5
|
||||
],
|
||||
"confidence": 0.65
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"sample_id": "terrain-test-flanders-rural",
|
||||
"split": "test",
|
||||
"task": "terrain_interpretation",
|
||||
"metadata": {
|
||||
"region": "flanders",
|
||||
"municipality": "Hasselt",
|
||||
"urbanity": "rural",
|
||||
"object_size": "not_applicable",
|
||||
"source": "fixture-dhmv",
|
||||
"sensor": "elevation-raster",
|
||||
"resolution_m": 5,
|
||||
"season": "not_applicable",
|
||||
"occlusion": "not_applicable",
|
||||
"vegetation": "moderate",
|
||||
"difficulty": "normal",
|
||||
"context": "terrain_profile",
|
||||
"tile_edge": false,
|
||||
"label_review_state": "fixture_reviewed",
|
||||
"ood": false,
|
||||
"date": "2023-01-01"
|
||||
},
|
||||
"config": {
|
||||
"missing_prediction_policy": "exclude_and_reduce_coverage"
|
||||
},
|
||||
"lineage": {
|
||||
"reference": {
|
||||
"source_id": "synthetic:terrain-test-flanders-rural:reference",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_contract_fixture"
|
||||
},
|
||||
"prediction": {
|
||||
"source_id": "synthetic:terrain-test-flanders-rural:prediction",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_fixed_output"
|
||||
}
|
||||
},
|
||||
"units": "m_TAW",
|
||||
"references": [
|
||||
12,
|
||||
13,
|
||||
14,
|
||||
15,
|
||||
16
|
||||
],
|
||||
"predictions": [
|
||||
12.2,
|
||||
12.8,
|
||||
14.4,
|
||||
null,
|
||||
15.7
|
||||
]
|
||||
},
|
||||
{
|
||||
"sample_id": "validation-test-national",
|
||||
"split": "test",
|
||||
"task": "geospatial_data_validation",
|
||||
"metadata": {
|
||||
"region": "national",
|
||||
"municipality": "not_applicable",
|
||||
"urbanity": "mixed",
|
||||
"object_size": "not_applicable",
|
||||
"source": "fixture-contracts",
|
||||
"sensor": "mixed",
|
||||
"resolution_m": null,
|
||||
"season": "not_applicable",
|
||||
"occlusion": "not_applicable",
|
||||
"vegetation": "mixed",
|
||||
"difficulty": "hard",
|
||||
"context": "data_contract",
|
||||
"tile_edge": false,
|
||||
"label_review_state": "fixture_reviewed",
|
||||
"ood": false,
|
||||
"date": "2025-01-01"
|
||||
},
|
||||
"config": {
|
||||
"critical_severities": [
|
||||
"blocker",
|
||||
"critical"
|
||||
]
|
||||
},
|
||||
"lineage": {
|
||||
"reference": {
|
||||
"source_id": "synthetic:validation-test-national:reference",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_contract_fixture"
|
||||
},
|
||||
"prediction": {
|
||||
"source_id": "synthetic:validation-test-national:prediction",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_fixed_output"
|
||||
}
|
||||
},
|
||||
"expected_anomalies": [
|
||||
{
|
||||
"code": "D-CRS-WRONG",
|
||||
"severity": "blocker"
|
||||
},
|
||||
{
|
||||
"code": "D-INVALID-GEOMETRY",
|
||||
"severity": "critical"
|
||||
},
|
||||
{
|
||||
"code": "D-MISSING-PROVENANCE",
|
||||
"severity": "critical"
|
||||
}
|
||||
],
|
||||
"observed_anomalies": [
|
||||
{
|
||||
"code": "D-CRS-WRONG",
|
||||
"severity": "blocker"
|
||||
},
|
||||
{
|
||||
"code": "D-INVALID-GEOMETRY",
|
||||
"severity": "critical"
|
||||
},
|
||||
{
|
||||
"code": "D-RESOLUTION",
|
||||
"severity": "major"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"sample_id": "background-test-pure-empty",
|
||||
"split": "background-test",
|
||||
"task": "object_detection",
|
||||
"metadata": {
|
||||
"region": "brussels",
|
||||
"municipality": "Brussel",
|
||||
"urbanity": "urban",
|
||||
"object_size": "empty",
|
||||
"source": "fixture-orthophoto",
|
||||
"sensor": "synthetic-rgb",
|
||||
"resolution_m": 0.25,
|
||||
"season": "winter",
|
||||
"occlusion": "none",
|
||||
"vegetation": "low",
|
||||
"difficulty": "hard",
|
||||
"context": "pure_background",
|
||||
"tile_edge": false,
|
||||
"label_review_state": "fixture_reviewed",
|
||||
"ood": false,
|
||||
"date": "2025-02-01"
|
||||
},
|
||||
"config": {
|
||||
"confidence_threshold": 0.5,
|
||||
"match_iou": 0.5,
|
||||
"fixed_diagnostic_risk_thresholds": [
|
||||
0,
|
||||
0.5,
|
||||
0.7,
|
||||
0.9
|
||||
]
|
||||
},
|
||||
"lineage": {
|
||||
"reference": {
|
||||
"source_id": "synthetic:background-test-pure-empty:reference",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_contract_fixture"
|
||||
},
|
||||
"prediction": {
|
||||
"source_id": "synthetic:background-test-pure-empty:prediction",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_fixed_output"
|
||||
}
|
||||
},
|
||||
"classes": [
|
||||
"building"
|
||||
],
|
||||
"references": [],
|
||||
"predictions": []
|
||||
},
|
||||
{
|
||||
"sample_id": "background-test-hard-negative",
|
||||
"split": "background-test",
|
||||
"task": "object_detection",
|
||||
"metadata": {
|
||||
"region": "flanders",
|
||||
"municipality": "Oostende",
|
||||
"urbanity": "urban",
|
||||
"object_size": "empty",
|
||||
"source": "fixture-orthophoto",
|
||||
"sensor": "synthetic-rgb",
|
||||
"resolution_m": 0.25,
|
||||
"season": "summer",
|
||||
"occlusion": "none",
|
||||
"vegetation": "moderate",
|
||||
"difficulty": "hard",
|
||||
"context": "coastal_hard_negative",
|
||||
"tile_edge": true,
|
||||
"label_review_state": "fixture_reviewed",
|
||||
"ood": false,
|
||||
"date": "2025-07-01"
|
||||
},
|
||||
"config": {
|
||||
"confidence_threshold": 0.5,
|
||||
"match_iou": 0.5,
|
||||
"fixed_diagnostic_risk_thresholds": [
|
||||
0,
|
||||
0.5,
|
||||
0.7,
|
||||
0.9
|
||||
]
|
||||
},
|
||||
"lineage": {
|
||||
"reference": {
|
||||
"source_id": "synthetic:background-test-hard-negative:reference",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_contract_fixture"
|
||||
},
|
||||
"prediction": {
|
||||
"source_id": "synthetic:background-test-hard-negative:prediction",
|
||||
"source_version": "1",
|
||||
"derivation": "hand_authored_fixed_output"
|
||||
}
|
||||
},
|
||||
"classes": [
|
||||
"building"
|
||||
],
|
||||
"references": [],
|
||||
"predictions": [
|
||||
{
|
||||
"id": "p-hard-fp",
|
||||
"class": "building",
|
||||
"bbox": [
|
||||
20,
|
||||
20,
|
||||
35,
|
||||
35
|
||||
],
|
||||
"confidence": 0.92
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
{
|
||||
"schema_version": 1,
|
||||
"dataset_version": "geointel-p4-harness-fixture-v2",
|
||||
"crs": "EPSG:31370",
|
||||
"independence_buffer_m": 2000.0,
|
||||
"perceptual_hamming_threshold": 4,
|
||||
"label_geometry_hamming_threshold": 2,
|
||||
"required_splits": ["train", "val", "calibration", "test", "background-test", "challenge"],
|
||||
"assignment_mode": "preassigned",
|
||||
"claim_boundary": "Synthetic contract fixtures for evaluator regression only; never production accuracy evidence.",
|
||||
"samples": [
|
||||
{"sample_id":"det-train-a","task":"object_detection","split":"train","group_id":"g01","source_family":"scene-01","temporal_family":"temporal-01","object_ids":["o01"],"bbox":[10000,10000,10100,10100],"raw_image_sha256":"0101010101010101010101010101010101010101010101010101010101010101","processed_image_sha256":"2121212121212121212121212121212121212121212121212121212121212121","label_sha256":"4141414141414141414141414141414141414141414141414141414141414141","perceptual_image_hash":"dcef24412401d4f9","label_geometry_hash":"6161616161616161616161616161616161616161616161616161616161616161","label_geometry_fingerprint":"c0f62e16b4033ee9","native_feature_ids":["native-01-1"],"parent_raster_id":"parent-raster-01","acquisition_id":"acquisition-01","acquisition_date":"2025-01-01"},
|
||||
{"sample_id":"seg-train-a","task":"footprint_segmentation","split":"train","group_id":"g02","source_family":"scene-02","temporal_family":"temporal-02","object_ids":["o02"],"bbox":[20000,10000,20100,10100],"raw_image_sha256":"0202020202020202020202020202020202020202020202020202020202020202","processed_image_sha256":"2222222222222222222222222222222222222222222222222222222222222222","label_sha256":"4242424242424242424242424242424242424242424242424242424242424242","perceptual_image_hash":"030bcdd3bc87412e","label_geometry_hash":"6262626262626262626262626262626262626262626262626262626262626262","label_geometry_fingerprint":"666a0fc2d26265ee","native_feature_ids":["native-02-1"],"parent_raster_id":"parent-raster-02","acquisition_id":"acquisition-02","acquisition_date":"2025-01-02"},
|
||||
{"sample_id":"raster-train-a","task":"raster_classification","split":"train","group_id":"g03","source_family":"scene-03","temporal_family":"temporal-03","object_ids":["o03"],"bbox":[30000,10000,30100,10100],"raw_image_sha256":"0303030303030303030303030303030303030303030303030303030303030303","processed_image_sha256":"2323232323232323232323232323232323232323232323232323232323232323","label_sha256":"4343434343434343434343434343434343434343434343434343434343434343","perceptual_image_hash":"75e36e008b4a6702","label_geometry_hash":"6363636363636363636363636363636363636363636363636363636363636363","label_geometry_fingerprint":"30820f2c124a3ff0","native_feature_ids":["native-03-1"],"parent_raster_id":"parent-raster-03","acquisition_id":"acquisition-03","acquisition_date":"2025-01-03"},
|
||||
{"sample_id":"val-vector-a","task":"vector_comparison","split":"val","group_id":"g04","source_family":"scene-04","temporal_family":"temporal-04","object_ids":["o04"],"bbox":[10000,30000,10100,30100],"raw_image_sha256":"0404040404040404040404040404040404040404040404040404040404040404","processed_image_sha256":"2424242424242424242424242424242424242424242424242424242424242424","label_sha256":"4444444444444444444444444444444444444444444444444444444444444444","perceptual_image_hash":"2e3b331cc01536c9","label_geometry_hash":"6464646464646464646464646464646464646464646464646464646464646464","label_geometry_fingerprint":"a559b46cba64e501","native_feature_ids":["native-04-1"],"parent_raster_id":"parent-raster-04","acquisition_id":"acquisition-04","acquisition_date":"2025-01-04"},
|
||||
{"sample_id":"val-change-b","task":"change_detection","split":"val","group_id":"g05","source_family":"scene-05","temporal_family":"temporal-05","object_ids":["o05"],"bbox":[30000,30000,30100,30100],"raw_image_sha256":"0505050505050505050505050505050505050505050505050505050505050505","processed_image_sha256":"2525252525252525252525252525252525252525252525252525252525252525","label_sha256":"4545454545454545454545454545454545454545454545454545454545454545","perceptual_image_hash":"828b99671c3f2f39","label_geometry_hash":"6565656565656565656565656565656565656565656565656565656565656565","label_geometry_fingerprint":"14ceff3640a013eb","native_feature_ids":["native-05-1"],"parent_raster_id":"parent-raster-05","acquisition_id":"acquisition-05","acquisition_date":"2025-01-05"},
|
||||
{"sample_id":"val-terrain-c","task":"terrain_interpretation","split":"val","group_id":"g06","source_family":"scene-06","temporal_family":"temporal-06","object_ids":["o06"],"bbox":[50000,30000,50100,30100],"raw_image_sha256":"0606060606060606060606060606060606060606060606060606060606060606","processed_image_sha256":"2626262626262626262626262626262626262626262626262626262626262626","label_sha256":"4646464646464646464646464646464646464646464646464646464646464646","perceptual_image_hash":"11c11105b0d10635","label_geometry_hash":"6666666666666666666666666666666666666666666666666666666666666666","label_geometry_fingerprint":"a1d9bbb93ba6d6da","native_feature_ids":["native-06-1"],"parent_raster_id":"parent-raster-06","acquisition_id":"acquisition-06","acquisition_date":"2025-01-06"},
|
||||
{"sample_id":"calibration-det-a","task":"object_detection","split":"calibration","group_id":"g07","source_family":"scene-07","temporal_family":"temporal-07","object_ids":["o21"],"bbox":[70000,30000,70100,30100],"raw_image_sha256":"0707070707070707070707070707070707070707070707070707070707070707","processed_image_sha256":"2727272727272727272727272727272727272727272727272727272727272727","label_sha256":"4747474747474747474747474747474747474747474747474747474747474747","perceptual_image_hash":"c4316514fed98f6b","label_geometry_hash":"6767676767676767676767676767676767676767676767676767676767676767","label_geometry_fingerprint":"dba609f34cb9826b","native_feature_ids":["native-07-1"],"parent_raster_id":"parent-raster-07","acquisition_id":"acquisition-07","acquisition_date":"2025-01-07"},
|
||||
{"sample_id":"calibration-seg-b","task":"footprint_segmentation","split":"calibration","group_id":"g08","source_family":"scene-08","temporal_family":"temporal-08","object_ids":["o22"],"bbox":[90000,30000,90100,30100],"raw_image_sha256":"0808080808080808080808080808080808080808080808080808080808080808","processed_image_sha256":"2828282828282828282828282828282828282828282828282828282828282828","label_sha256":"4848484848484848484848484848484848484848484848484848484848484848","perceptual_image_hash":"518ba3145333c08c","label_geometry_hash":"6868686868686868686868686868686868686868686868686868686868686868","label_geometry_fingerprint":"627df87eea48aec0","native_feature_ids":["native-08-1"],"parent_raster_id":"parent-raster-08","acquisition_id":"acquisition-08","acquisition_date":"2025-01-08"},
|
||||
{"sample_id":"det-test-flanders-urban","task":"object_detection","split":"test","group_id":"g09","source_family":"scene-09","temporal_family":"temporal-09","object_ids":["o07","o08"],"bbox":[10000,50000,10100,50100],"raw_image_sha256":"0909090909090909090909090909090909090909090909090909090909090909","processed_image_sha256":"2929292929292929292929292929292929292929292929292929292929292929","label_sha256":"4949494949494949494949494949494949494949494949494949494949494949","perceptual_image_hash":"d6d20ed34050b89e","label_geometry_hash":"6969696969696969696969696969696969696969696969696969696969696969","label_geometry_fingerprint":"7ed221e11cc6828d","native_feature_ids":["native-09-1","native-09-2"],"parent_raster_id":"parent-raster-09","acquisition_id":"acquisition-09","acquisition_date":"2025-01-09"},
|
||||
{"sample_id":"seg-test-wallonia-rural","task":"footprint_segmentation","split":"test","group_id":"g10","source_family":"scene-10","temporal_family":"temporal-10","object_ids":["o09"],"bbox":[20000,50000,20100,50100],"raw_image_sha256":"0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a0a","processed_image_sha256":"2a2a2a2a2a2a2a2a2a2a2a2a2a2a2a2a2a2a2a2a2a2a2a2a2a2a2a2a2a2a2a2a","label_sha256":"4a4a4a4a4a4a4a4a4a4a4a4a4a4a4a4a4a4a4a4a4a4a4a4a4a4a4a4a4a4a4a4a","perceptual_image_hash":"6c476d18e0539687","label_geometry_hash":"6a6a6a6a6a6a6a6a6a6a6a6a6a6a6a6a6a6a6a6a6a6a6a6a6a6a6a6a6a6a6a6a","label_geometry_fingerprint":"e9fb3c93de82d0be","native_feature_ids":["native-10-1"],"parent_raster_id":"parent-raster-10","acquisition_id":"acquisition-10","acquisition_date":"2025-01-10"},
|
||||
{"sample_id":"raster-test-brussels-urban","task":"raster_classification","split":"test","group_id":"g11","source_family":"scene-11","temporal_family":"temporal-11","object_ids":["o10"],"bbox":[30000,50000,30100,50100],"raw_image_sha256":"0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b","processed_image_sha256":"2b2b2b2b2b2b2b2b2b2b2b2b2b2b2b2b2b2b2b2b2b2b2b2b2b2b2b2b2b2b2b2b","label_sha256":"4b4b4b4b4b4b4b4b4b4b4b4b4b4b4b4b4b4b4b4b4b4b4b4b4b4b4b4b4b4b4b4b","perceptual_image_hash":"4383fa06c328f326","label_geometry_hash":"6b6b6b6b6b6b6b6b6b6b6b6b6b6b6b6b6b6b6b6b6b6b6b6b6b6b6b6b6b6b6b6b","label_geometry_fingerprint":"f4269ac17b100571","native_feature_ids":["native-11-1"],"parent_raster_id":"parent-raster-11","acquisition_id":"acquisition-11","acquisition_date":"2025-01-11"},
|
||||
{"sample_id":"vector-test-flanders-suburban","task":"vector_comparison","split":"test","group_id":"g12","source_family":"scene-12","temporal_family":"temporal-12","object_ids":["o11","o12"],"bbox":[40000,50000,40100,50100],"raw_image_sha256":"0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c0c","processed_image_sha256":"2c2c2c2c2c2c2c2c2c2c2c2c2c2c2c2c2c2c2c2c2c2c2c2c2c2c2c2c2c2c2c2c","label_sha256":"4c4c4c4c4c4c4c4c4c4c4c4c4c4c4c4c4c4c4c4c4c4c4c4c4c4c4c4c4c4c4c4c","perceptual_image_hash":"1236e1aa148c25d5","label_geometry_hash":"6c6c6c6c6c6c6c6c6c6c6c6c6c6c6c6c6c6c6c6c6c6c6c6c6c6c6c6c6c6c6c6c","label_geometry_fingerprint":"2db7d069691934ed","native_feature_ids":["native-12-1","native-12-2"],"parent_raster_id":"parent-raster-12","acquisition_id":"acquisition-12","acquisition_date":"2025-01-12"},
|
||||
{"sample_id":"change-test-wallonia-industrial","task":"change_detection","split":"test","group_id":"g13","source_family":"scene-13","temporal_family":"temporal-13","object_ids":["o13","o14"],"bbox":[50000,50000,50100,50100],"raw_image_sha256":"0d0d0d0d0d0d0d0d0d0d0d0d0d0d0d0d0d0d0d0d0d0d0d0d0d0d0d0d0d0d0d0d","processed_image_sha256":"2d2d2d2d2d2d2d2d2d2d2d2d2d2d2d2d2d2d2d2d2d2d2d2d2d2d2d2d2d2d2d2d","label_sha256":"4d4d4d4d4d4d4d4d4d4d4d4d4d4d4d4d4d4d4d4d4d4d4d4d4d4d4d4d4d4d4d4d","perceptual_image_hash":"14e347040ef16eac","label_geometry_hash":"6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d6d","label_geometry_fingerprint":"6c5b5f9db0d73c31","native_feature_ids":["native-13-1","native-13-2"],"parent_raster_id":"parent-raster-13","acquisition_id":"acquisition-13","acquisition_date":"2025-01-13"},
|
||||
{"sample_id":"terrain-test-flanders-rural","task":"terrain_interpretation","split":"test","group_id":"g14","source_family":"scene-14","temporal_family":"temporal-14","object_ids":["o15"],"bbox":[60000,50000,60100,50100],"raw_image_sha256":"0e0e0e0e0e0e0e0e0e0e0e0e0e0e0e0e0e0e0e0e0e0e0e0e0e0e0e0e0e0e0e0e","processed_image_sha256":"2e2e2e2e2e2e2e2e2e2e2e2e2e2e2e2e2e2e2e2e2e2e2e2e2e2e2e2e2e2e2e2e","label_sha256":"4e4e4e4e4e4e4e4e4e4e4e4e4e4e4e4e4e4e4e4e4e4e4e4e4e4e4e4e4e4e4e4e","perceptual_image_hash":"c808940b0a325e9c","label_geometry_hash":"6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e6e","label_geometry_fingerprint":"4ec9f1dd6c799ec7","native_feature_ids":["native-14-1"],"parent_raster_id":"parent-raster-14","acquisition_id":"acquisition-14","acquisition_date":"2025-01-14"},
|
||||
{"sample_id":"validation-test-national","task":"geospatial_data_validation","split":"test","group_id":"g15","source_family":"scene-15","temporal_family":"temporal-15","object_ids":["o16"],"bbox":[70000,50000,70100,50100],"raw_image_sha256":"0f0f0f0f0f0f0f0f0f0f0f0f0f0f0f0f0f0f0f0f0f0f0f0f0f0f0f0f0f0f0f0f","processed_image_sha256":"2f2f2f2f2f2f2f2f2f2f2f2f2f2f2f2f2f2f2f2f2f2f2f2f2f2f2f2f2f2f2f2f","label_sha256":"4f4f4f4f4f4f4f4f4f4f4f4f4f4f4f4f4f4f4f4f4f4f4f4f4f4f4f4f4f4f4f4f","perceptual_image_hash":"4365e6b7df6acf3d","label_geometry_hash":"6f6f6f6f6f6f6f6f6f6f6f6f6f6f6f6f6f6f6f6f6f6f6f6f6f6f6f6f6f6f6f6f","label_geometry_fingerprint":"616691796f62dbbb","native_feature_ids":["native-15-1"],"parent_raster_id":"parent-raster-15","acquisition_id":"acquisition-15","acquisition_date":"2025-01-15"},
|
||||
{"sample_id":"background-test-pure-empty","task":"object_detection","split":"background-test","group_id":"g16","source_family":"scene-16","temporal_family":"temporal-16","object_ids":[],"bbox":[20000,70000,20100,70100],"raw_image_sha256":"1010101010101010101010101010101010101010101010101010101010101010","processed_image_sha256":"3030303030303030303030303030303030303030303030303030303030303030","label_sha256":"5050505050505050505050505050505050505050505050505050505050505050","perceptual_image_hash":"be4078e1ef9fa296","label_geometry_hash":"7070707070707070707070707070707070707070707070707070707070707070","label_geometry_fingerprint":"004e762f898c4fad","native_feature_ids":[],"parent_raster_id":"parent-raster-16","acquisition_id":"acquisition-16","acquisition_date":"2025-01-16"},
|
||||
{"sample_id":"background-test-hard-negative","task":"object_detection","split":"background-test","group_id":"g17","source_family":"scene-17","temporal_family":"temporal-17","object_ids":[],"bbox":[60000,70000,60100,70100],"raw_image_sha256":"1111111111111111111111111111111111111111111111111111111111111111","processed_image_sha256":"3131313131313131313131313131313131313131313131313131313131313131","label_sha256":"5151515151515151515151515151515151515151515151515151515151515151","perceptual_image_hash":"6ada1b5aa46cc261","label_geometry_hash":"7171717171717171717171717171717171717171717171717171717171717171","label_geometry_fingerprint":"2b405bcc9c95acfb","native_feature_ids":[],"parent_raster_id":"parent-raster-17","acquisition_id":"acquisition-17","acquisition_date":"2025-01-17"},
|
||||
{"sample_id":"det-challenge-coast","task":"object_detection","split":"challenge","group_id":"g18","source_family":"scene-18","temporal_family":"temporal-18","object_ids":["o17"],"bbox":[10000,90000,10100,90100],"raw_image_sha256":"1212121212121212121212121212121212121212121212121212121212121212","processed_image_sha256":"3232323232323232323232323232323232323232323232323232323232323232","label_sha256":"5252525252525252525252525252525252525252525252525252525252525252","perceptual_image_hash":"15910b3a2d056648","label_geometry_hash":"7272727272727272727272727272727272727272727272727272727272727272","label_geometry_fingerprint":"95518e2583b3a7d3","native_feature_ids":["native-18-1"],"parent_raster_id":"parent-raster-18","acquisition_id":"acquisition-18","acquisition_date":"2025-01-18"},
|
||||
{"sample_id":"seg-challenge-brussels","task":"footprint_segmentation","split":"challenge","group_id":"g19","source_family":"scene-19","temporal_family":"temporal-19","object_ids":["o18"],"bbox":[30000,90000,30100,90100],"raw_image_sha256":"1313131313131313131313131313131313131313131313131313131313131313","processed_image_sha256":"3333333333333333333333333333333333333333333333333333333333333333","label_sha256":"5353535353535353535353535353535353535353535353535353535353535353","perceptual_image_hash":"b5c6e201d687cf9e","label_geometry_hash":"7373737373737373737373737373737373737373737373737373737373737373","label_geometry_fingerprint":"2625fe765bce389e","native_feature_ids":["native-19-1"],"parent_raster_id":"parent-raster-19","acquisition_id":"acquisition-19","acquisition_date":"2025-01-19"},
|
||||
{"sample_id":"raster-challenge-seasonal","task":"raster_classification","split":"challenge","group_id":"g20","source_family":"scene-20","temporal_family":"temporal-20","object_ids":["o19"],"bbox":[50000,90000,50100,90100],"raw_image_sha256":"1414141414141414141414141414141414141414141414141414141414141414","processed_image_sha256":"3434343434343434343434343434343434343434343434343434343434343434","label_sha256":"5454545454545454545454545454545454545454545454545454545454545454","perceptual_image_hash":"85f82f2f8eb2c2f5","label_geometry_hash":"7474747474747474747474747474747474747474747474747474747474747474","label_geometry_fingerprint":"e91bcaae1021dccc","native_feature_ids":["native-20-1"],"parent_raster_id":"parent-raster-20","acquisition_id":"acquisition-20","acquisition_date":"2025-01-20"},
|
||||
{"sample_id":"validation-challenge-corrupt","task":"geospatial_data_validation","split":"challenge","group_id":"g21","source_family":"scene-21","temporal_family":"temporal-21","object_ids":["o20"],"bbox":[70000,90000,70100,90100],"raw_image_sha256":"1515151515151515151515151515151515151515151515151515151515151515","processed_image_sha256":"3535353535353535353535353535353535353535353535353535353535353535","label_sha256":"5555555555555555555555555555555555555555555555555555555555555555","perceptual_image_hash":"c71055ef6baeee8b","label_geometry_hash":"7575757575757575757575757575757575757575757575757575757575757575","label_geometry_fingerprint":"686f1d8a92703709","native_feature_ids":["native-21-1"],"parent_raster_id":"parent-raster-21","acquisition_id":"acquisition-21","acquisition_date":"2025-01-21"}
|
||||
]
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -28,7 +28,7 @@ from training_release_manifest import ( # noqa: E402
|
||||
)
|
||||
|
||||
|
||||
PROTECTED_SPLITS = {"calibration", "test", "background-test"}
|
||||
PROTECTED_SPLITS = {"calibration", "test", "background-test", "challenge"}
|
||||
PROPOSAL_DATASET_PROVENANCE_NAME = "proposal-dataset-provenance.json"
|
||||
PROPOSAL_DATASET_EVIDENCE_NAME = "proposal-dataset-evidence.json"
|
||||
PROPOSAL_DATASET_SCHEMA_VERSION = 1
|
||||
|
||||
@@ -26,7 +26,7 @@ from training_release_manifest import ( # noqa: E402
|
||||
)
|
||||
|
||||
|
||||
PROTECTED_SPLITS = {"calibration", "test", "background-test"}
|
||||
PROTECTED_SPLITS = {"calibration", "test", "background-test", "challenge"}
|
||||
|
||||
|
||||
def sha256(path: Path) -> str:
|
||||
|
||||
@@ -0,0 +1,701 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Generate immutable P4 split manifests and fail closed on leakage."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import math
|
||||
import re
|
||||
from collections import Counter, defaultdict
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
from typing import Any, Iterable
|
||||
|
||||
from pyproj import CRS
|
||||
|
||||
SCHEMA_VERSION = 1
|
||||
GENERATOR_VERSION = "1.1.0"
|
||||
TRAIN_SPLITS = {"train"}
|
||||
SELECTION_SPLITS = {"val", "calibration"}
|
||||
DEVELOPMENT_SPLITS = TRAIN_SPLITS | SELECTION_SPLITS
|
||||
PROTECTED_SPLITS = {"test", "background-test", "challenge"}
|
||||
ALL_SPLITS = DEVELOPMENT_SPLITS | PROTECTED_SPLITS
|
||||
NORMATIVE_SPLITS = {"train", "val", "calibration", "test", "background-test"}
|
||||
SHA256_PATTERN = re.compile(r"^[0-9a-f]{64}$")
|
||||
FINGERPRINT_PATTERN = re.compile(r"^[0-9a-f]{16}$")
|
||||
|
||||
|
||||
class LeakageError(ValueError):
|
||||
"""Raised when split isolation is not demonstrably safe."""
|
||||
|
||||
|
||||
def canonical_bytes(value: Any) -> bytes:
|
||||
return json.dumps(
|
||||
value, ensure_ascii=False, sort_keys=True, separators=(",", ":")
|
||||
).encode("utf-8")
|
||||
|
||||
|
||||
def canonical_hash(value: Any) -> str:
|
||||
return hashlib.sha256(canonical_bytes(value)).hexdigest()
|
||||
|
||||
|
||||
def write_json(path: Path, payload: Any) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
content = json.dumps(payload, ensure_ascii=False, indent=2, sort_keys=True) + "\n"
|
||||
temporary = path.with_name(f".{path.name}.tmp")
|
||||
temporary.write_text(content, encoding="utf-8")
|
||||
temporary.replace(path)
|
||||
|
||||
|
||||
def validate_metric_crs(value: Any) -> str:
|
||||
try:
|
||||
crs = CRS.from_user_input(value)
|
||||
except Exception as exc: # noqa: BLE001 - pyproj exposes multiple parser errors
|
||||
raise LeakageError(f"Invalid CRS: {value!r}") from exc
|
||||
axes = crs.axis_info
|
||||
if crs.is_geographic or not axes:
|
||||
raise LeakageError(f"CRS must be projected in metres: {value!r}")
|
||||
if any(
|
||||
abs(float(axis.unit_conversion_factor or 0.0) - 1.0) > 1e-12
|
||||
for axis in axes[:2]
|
||||
):
|
||||
raise LeakageError(f"CRS axes must use metres: {value!r}")
|
||||
return crs.to_string()
|
||||
|
||||
|
||||
def require_hex(
|
||||
sample_id: str, field: str, value: Any, pattern: re.Pattern[str]
|
||||
) -> str:
|
||||
normalized = str(value or "").lower()
|
||||
if not pattern.fullmatch(normalized):
|
||||
raise LeakageError(f"{sample_id}: {field} has an invalid fingerprint")
|
||||
return normalized
|
||||
|
||||
|
||||
def fingerprint_distance(left: str, right: str) -> int:
|
||||
return (int(left, 16) ^ int(right, 16)).bit_count()
|
||||
|
||||
|
||||
def bbox_distance(left: list[float], right: list[float]) -> float:
|
||||
dx = max(left[0] - right[2], right[0] - left[2], 0.0)
|
||||
dy = max(left[1] - right[3], right[1] - left[3], 0.0)
|
||||
return math.hypot(dx, dy)
|
||||
|
||||
|
||||
def normalized_sample(sample: dict[str, Any]) -> dict[str, Any]:
|
||||
required = {
|
||||
"sample_id",
|
||||
"task",
|
||||
"split",
|
||||
"group_id",
|
||||
"source_family",
|
||||
"temporal_family",
|
||||
"object_ids",
|
||||
"bbox",
|
||||
"raw_image_sha256",
|
||||
"processed_image_sha256",
|
||||
"label_sha256",
|
||||
"perceptual_image_hash",
|
||||
"label_geometry_hash",
|
||||
"label_geometry_fingerprint",
|
||||
"native_feature_ids",
|
||||
"parent_raster_id",
|
||||
"acquisition_id",
|
||||
"acquisition_date",
|
||||
}
|
||||
missing = sorted(required - set(sample))
|
||||
if missing:
|
||||
raise LeakageError(
|
||||
f"{sample.get('sample_id', '<unknown>')}: missing fields {missing}"
|
||||
)
|
||||
split = str(sample["split"])
|
||||
if split not in ALL_SPLITS:
|
||||
raise LeakageError(f"{sample['sample_id']}: unsupported split {split!r}")
|
||||
bbox = sample["bbox"]
|
||||
if not isinstance(bbox, list) or len(bbox) != 4:
|
||||
raise LeakageError(f"{sample['sample_id']}: bbox must contain four values")
|
||||
values = [float(value) for value in bbox]
|
||||
if (
|
||||
not all(math.isfinite(value) for value in values)
|
||||
or values[0] >= values[2]
|
||||
or values[1] >= values[3]
|
||||
):
|
||||
raise LeakageError(f"{sample['sample_id']}: invalid bbox")
|
||||
result = dict(sample)
|
||||
result["bbox"] = values
|
||||
for field in (
|
||||
"raw_image_sha256",
|
||||
"processed_image_sha256",
|
||||
"label_sha256",
|
||||
"label_geometry_hash",
|
||||
):
|
||||
result[field] = require_hex(
|
||||
str(sample["sample_id"]), field, sample[field], SHA256_PATTERN
|
||||
)
|
||||
for field in ("perceptual_image_hash", "label_geometry_fingerprint"):
|
||||
result[field] = require_hex(
|
||||
str(sample["sample_id"]), field, sample[field], FINGERPRINT_PATTERN
|
||||
)
|
||||
for field in ("object_ids", "native_feature_ids"):
|
||||
if not isinstance(sample[field], list) or any(
|
||||
not str(value) for value in sample[field]
|
||||
):
|
||||
raise LeakageError(
|
||||
f"{sample['sample_id']}: {field} must be a list of non-empty identities"
|
||||
)
|
||||
result[field] = sorted({str(value) for value in sample[field]})
|
||||
for field in (
|
||||
"group_id",
|
||||
"source_family",
|
||||
"temporal_family",
|
||||
"parent_raster_id",
|
||||
"acquisition_id",
|
||||
):
|
||||
if not str(sample[field]).strip():
|
||||
raise LeakageError(f"{sample['sample_id']}: {field} must be non-empty")
|
||||
result[field] = str(sample[field])
|
||||
try:
|
||||
result["acquisition_date"] = date.fromisoformat(
|
||||
str(sample["acquisition_date"])
|
||||
).isoformat()
|
||||
except ValueError as exc:
|
||||
raise LeakageError(
|
||||
f"{sample['sample_id']}: acquisition_date must be ISO-8601"
|
||||
) from exc
|
||||
record_without_split = {
|
||||
key: value for key, value in result.items() if key != "split"
|
||||
}
|
||||
result["record_sha256"] = canonical_hash(record_without_split)
|
||||
return result
|
||||
|
||||
|
||||
def assign_split_roles(source: dict[str, Any]) -> list[dict[str, Any]]:
|
||||
"""Assign complete leakage groups deterministically, or validate a frozen assignment."""
|
||||
raw_samples = [dict(item) for item in source.get("samples", [])]
|
||||
mode = str(source.get("assignment_mode") or "preassigned")
|
||||
split_presence = [bool(item.get("split")) for item in raw_samples]
|
||||
if mode == "preassigned":
|
||||
if raw_samples and not all(split_presence):
|
||||
raise LeakageError("Preassigned mode requires a split on every sample")
|
||||
return raw_samples
|
||||
if mode != "deterministic_grouped":
|
||||
raise LeakageError(f"Unsupported assignment_mode: {mode!r}")
|
||||
if any(split_presence):
|
||||
raise LeakageError(
|
||||
"deterministic_grouped mode refuses partially or fully preassigned splits"
|
||||
)
|
||||
config = source.get("split_assignment") or {}
|
||||
roles = list(
|
||||
config.get("roles")
|
||||
or ["train", "val", "calibration", "test", "background-test"]
|
||||
)
|
||||
if not roles or len(set(roles)) != len(roles) or set(roles) - ALL_SPLITS:
|
||||
raise LeakageError("split_assignment.roles must be unique supported roles")
|
||||
weights = {
|
||||
role: float((config.get("weights") or {}).get(role, 1.0)) for role in roles
|
||||
}
|
||||
if any(not math.isfinite(value) or value <= 0 for value in weights.values()):
|
||||
raise LeakageError("split_assignment weights must be positive finite values")
|
||||
placeholder = [
|
||||
normalized_sample({**item, "split": "train"}) for item in raw_samples
|
||||
]
|
||||
parent = list(range(len(placeholder)))
|
||||
|
||||
def find(index: int) -> int:
|
||||
while parent[index] != index:
|
||||
parent[index] = parent[parent[index]]
|
||||
index = parent[index]
|
||||
return index
|
||||
|
||||
def union(left: int, right: int) -> None:
|
||||
left_root = find(left)
|
||||
right_root = find(right)
|
||||
if left_root != right_root:
|
||||
parent[max(left_root, right_root)] = min(left_root, right_root)
|
||||
|
||||
identity_fields = (
|
||||
"group_id",
|
||||
"source_family",
|
||||
"temporal_family",
|
||||
"raw_image_sha256",
|
||||
"processed_image_sha256",
|
||||
"label_sha256",
|
||||
"label_geometry_hash",
|
||||
"parent_raster_id",
|
||||
"acquisition_id",
|
||||
)
|
||||
seen: dict[tuple[str, str], int] = {}
|
||||
for index, sample in enumerate(placeholder):
|
||||
identities = [(field, str(sample[field])) for field in identity_fields]
|
||||
identities.extend(("object_id", str(value)) for value in sample["object_ids"])
|
||||
identities.extend(
|
||||
("native_feature_id", str(value)) for value in sample["native_feature_ids"]
|
||||
)
|
||||
for identity in identities:
|
||||
if identity in seen:
|
||||
union(index, seen[identity])
|
||||
else:
|
||||
seen[identity] = index
|
||||
image_threshold = int(source.get("perceptual_hamming_threshold", 4))
|
||||
geometry_threshold = int(source.get("label_geometry_hamming_threshold", 2))
|
||||
buffer_m = float(source.get("independence_buffer_m") or 0)
|
||||
for index, left in enumerate(placeholder):
|
||||
for right_index, right in enumerate(placeholder[index + 1 :], start=index + 1):
|
||||
if (
|
||||
fingerprint_distance(
|
||||
left["perceptual_image_hash"], right["perceptual_image_hash"]
|
||||
)
|
||||
<= image_threshold
|
||||
or fingerprint_distance(
|
||||
left["label_geometry_fingerprint"],
|
||||
right["label_geometry_fingerprint"],
|
||||
)
|
||||
<= geometry_threshold
|
||||
or bbox_distance(left["bbox"], right["bbox"]) < buffer_m
|
||||
):
|
||||
union(index, right_index)
|
||||
components: dict[int, list[int]] = defaultdict(list)
|
||||
for index in range(len(placeholder)):
|
||||
components[find(index)].append(index)
|
||||
if len(components) < len(roles):
|
||||
raise LeakageError(
|
||||
f"Not enough independent groups for required roles: {len(components)} < {len(roles)}"
|
||||
)
|
||||
seed = str(config.get("seed") or "geointel-p4-group-split-v1")
|
||||
groups = sorted(
|
||||
components.values(),
|
||||
key=lambda indices: canonical_hash(
|
||||
{
|
||||
"seed": seed,
|
||||
"sample_ids": sorted(
|
||||
placeholder[index]["sample_id"] for index in indices
|
||||
),
|
||||
}
|
||||
),
|
||||
)
|
||||
stratify_by = tuple(config.get("stratify_by") or ["task"])
|
||||
assigned_counts: Counter[str] = Counter()
|
||||
stratum_counts: dict[str, Counter[str]] = defaultdict(Counter)
|
||||
assignment: dict[int, str] = {}
|
||||
for group_index, indices in enumerate(groups):
|
||||
strata = {
|
||||
f"{field}={placeholder[index].get(field) or (placeholder[index].get('metadata') or {}).get(field)}"
|
||||
for index in indices
|
||||
for field in stratify_by
|
||||
}
|
||||
candidates = (
|
||||
roles[group_index : group_index + 1] if group_index < len(roles) else roles
|
||||
)
|
||||
role = min(
|
||||
candidates,
|
||||
key=lambda candidate: (
|
||||
assigned_counts[candidate] / weights[candidate]
|
||||
+ sum(
|
||||
stratum_counts[stratum][candidate] / weights[candidate]
|
||||
for stratum in strata
|
||||
),
|
||||
candidate,
|
||||
),
|
||||
)
|
||||
for index in indices:
|
||||
assignment[index] = role
|
||||
assigned_counts[role] += len(indices)
|
||||
for stratum in strata:
|
||||
stratum_counts[stratum][role] += len(indices)
|
||||
return [
|
||||
{**sample, "split": assignment[index]}
|
||||
for index, sample in enumerate(raw_samples)
|
||||
]
|
||||
|
||||
|
||||
def leakage_findings(
|
||||
samples: list[dict[str, Any]],
|
||||
independence_buffer_m: float,
|
||||
perceptual_hamming_threshold: int,
|
||||
label_geometry_hamming_threshold: int,
|
||||
) -> list[dict[str, Any]]:
|
||||
findings: list[dict[str, Any]] = []
|
||||
identity_fields = (
|
||||
"sample_id",
|
||||
"group_id",
|
||||
"source_family",
|
||||
"temporal_family",
|
||||
"raw_image_sha256",
|
||||
"processed_image_sha256",
|
||||
"label_sha256",
|
||||
"label_geometry_hash",
|
||||
"parent_raster_id",
|
||||
"acquisition_id",
|
||||
"record_sha256",
|
||||
)
|
||||
identities: dict[str, dict[str, set[str]]] = {
|
||||
key: defaultdict(set) for key in identity_fields
|
||||
}
|
||||
object_splits: dict[str, set[str]] = defaultdict(set)
|
||||
native_feature_splits: dict[str, set[str]] = defaultdict(set)
|
||||
for sample in samples:
|
||||
for key, index in identities.items():
|
||||
index[str(sample[key])].add(sample["split"])
|
||||
for object_id in sample["object_ids"]:
|
||||
object_splits[str(object_id)].add(sample["split"])
|
||||
for feature_id in sample["native_feature_ids"]:
|
||||
native_feature_splits[str(feature_id)].add(sample["split"])
|
||||
code_by_key = {
|
||||
"sample_id": "S-SAMPLE-IDENTITY",
|
||||
"group_id": "S-SPATIAL-GROUP",
|
||||
"source_family": "S-SOURCE-FAMILY",
|
||||
"temporal_family": "S-TEMPORAL-FAMILY",
|
||||
"raw_image_sha256": "S-RAW-IMAGE-DUPLICATE",
|
||||
"processed_image_sha256": "S-PROCESSED-IMAGE-DUPLICATE",
|
||||
"label_sha256": "S-LABEL-DUPLICATE",
|
||||
"label_geometry_hash": "S-LABEL-GEOMETRY-DUPLICATE",
|
||||
"parent_raster_id": "S-PARENT-RASTER",
|
||||
"acquisition_id": "S-ACQUISITION",
|
||||
"record_sha256": "S-EXACT-DUPLICATE",
|
||||
}
|
||||
for key, index in identities.items():
|
||||
for value, splits in sorted(index.items()):
|
||||
if len(splits) > 1:
|
||||
findings.append(
|
||||
{
|
||||
"code": code_by_key[key],
|
||||
"identity": value,
|
||||
"splits": sorted(splits),
|
||||
}
|
||||
)
|
||||
for object_id, splits in sorted(object_splits.items()):
|
||||
if len(splits) > 1:
|
||||
findings.append(
|
||||
{
|
||||
"code": "S-OBJECT-INSTANCE",
|
||||
"identity": object_id,
|
||||
"splits": sorted(splits),
|
||||
}
|
||||
)
|
||||
for feature_id, splits in sorted(native_feature_splits.items()):
|
||||
if len(splits) > 1:
|
||||
findings.append(
|
||||
{
|
||||
"code": "S-NATIVE-FEATURE",
|
||||
"identity": feature_id,
|
||||
"splits": sorted(splits),
|
||||
}
|
||||
)
|
||||
for index, left in enumerate(samples):
|
||||
for right in samples[index + 1 :]:
|
||||
if left["split"] == right["split"]:
|
||||
continue
|
||||
image_distance = fingerprint_distance(
|
||||
left["perceptual_image_hash"], right["perceptual_image_hash"]
|
||||
)
|
||||
if image_distance <= perceptual_hamming_threshold:
|
||||
findings.append(
|
||||
{
|
||||
"code": "S-PERCEPTUAL-IMAGE-NEAR-DUPLICATE",
|
||||
"left": left["sample_id"],
|
||||
"right": right["sample_id"],
|
||||
"distance": image_distance,
|
||||
"maximum_allowed_distance": perceptual_hamming_threshold,
|
||||
}
|
||||
)
|
||||
geometry_distance = fingerprint_distance(
|
||||
left["label_geometry_fingerprint"], right["label_geometry_fingerprint"]
|
||||
)
|
||||
if geometry_distance <= label_geometry_hamming_threshold:
|
||||
findings.append(
|
||||
{
|
||||
"code": "S-LABEL-GEOMETRY-NEAR-DUPLICATE",
|
||||
"left": left["sample_id"],
|
||||
"right": right["sample_id"],
|
||||
"distance": geometry_distance,
|
||||
"maximum_allowed_distance": label_geometry_hamming_threshold,
|
||||
}
|
||||
)
|
||||
distance = bbox_distance(left["bbox"], right["bbox"])
|
||||
if distance < independence_buffer_m:
|
||||
findings.append(
|
||||
{
|
||||
"code": "S-SPATIAL-OVERLAP",
|
||||
"left": left["sample_id"],
|
||||
"right": right["sample_id"],
|
||||
"left_split": left["split"],
|
||||
"right_split": right["split"],
|
||||
"distance_m": distance,
|
||||
"required_distance_m": independence_buffer_m,
|
||||
}
|
||||
)
|
||||
return findings
|
||||
|
||||
|
||||
def build_manifests(
|
||||
source: dict[str, Any],
|
||||
) -> tuple[dict[str, Any], dict[str, Any], dict[str, Any]]:
|
||||
if source.get("schema_version") != 1:
|
||||
raise LeakageError("Unsupported source manifest schema_version")
|
||||
metric_crs = validate_metric_crs(source.get("crs"))
|
||||
assigned_source_samples = assign_split_roles(source)
|
||||
samples = [normalized_sample(item) for item in assigned_source_samples]
|
||||
if not samples:
|
||||
raise LeakageError("Source manifest has no samples")
|
||||
if len({item["sample_id"] for item in samples}) != len(samples):
|
||||
raise LeakageError("Duplicate sample_id in source manifest")
|
||||
buffer_m = float(source.get("independence_buffer_m") or 0)
|
||||
if not math.isfinite(buffer_m) or buffer_m <= 0:
|
||||
raise LeakageError("independence_buffer_m must be a positive finite number")
|
||||
split_counts = Counter(item["split"] for item in samples)
|
||||
required_splits = set(source.get("required_splits") or NORMATIVE_SPLITS)
|
||||
unsupported_required = sorted(required_splits - ALL_SPLITS)
|
||||
if unsupported_required:
|
||||
raise LeakageError(f"Unsupported required splits: {unsupported_required}")
|
||||
missing_splits = sorted(required_splits - set(split_counts))
|
||||
if missing_splits:
|
||||
raise LeakageError(f"Required splits are absent: {missing_splits}")
|
||||
perceptual_threshold = int(source.get("perceptual_hamming_threshold", 4))
|
||||
geometry_threshold = int(source.get("label_geometry_hamming_threshold", 2))
|
||||
if not 0 <= perceptual_threshold < 64 or not 0 <= geometry_threshold < 64:
|
||||
raise LeakageError("Near-duplicate Hamming thresholds must be between 0 and 63")
|
||||
findings = leakage_findings(
|
||||
samples, buffer_m, perceptual_threshold, geometry_threshold
|
||||
)
|
||||
canonical_source = dict(source)
|
||||
canonical_source["samples"] = sorted(
|
||||
source.get("samples", []), key=lambda item: str(item.get("sample_id", ""))
|
||||
)
|
||||
source_hash = canonical_hash(canonical_source)
|
||||
|
||||
development_samples = sorted(
|
||||
(item for item in samples if item["split"] in DEVELOPMENT_SPLITS),
|
||||
key=lambda item: item["sample_id"],
|
||||
)
|
||||
protected_samples = sorted(
|
||||
(item for item in samples if item["split"] in PROTECTED_SPLITS),
|
||||
key=lambda item: item["sample_id"],
|
||||
)
|
||||
common = {
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"generator_version": GENERATOR_VERSION,
|
||||
"dataset_version": source["dataset_version"],
|
||||
"source_manifest_sha256": source_hash,
|
||||
"crs": metric_crs,
|
||||
"independence_buffer_m": buffer_m,
|
||||
"perceptual_hamming_threshold": perceptual_threshold,
|
||||
"label_geometry_hamming_threshold": geometry_threshold,
|
||||
"assignment_mode": str(source.get("assignment_mode") or "preassigned"),
|
||||
"assignment_algorithm": "group-before-split-deficit-balancer-v1",
|
||||
"assignment_seed": str(
|
||||
(source.get("split_assignment") or {}).get("seed")
|
||||
or "geointel-p4-group-split-v1"
|
||||
),
|
||||
"claim_boundary": source.get("claim_boundary"),
|
||||
}
|
||||
development = {
|
||||
**common,
|
||||
"manifest_role": "development_and_calibration",
|
||||
"allowed_splits": sorted(DEVELOPMENT_SPLITS),
|
||||
"training_access_allowed_by_split": {
|
||||
"train": True,
|
||||
"val": False,
|
||||
"calibration": False,
|
||||
},
|
||||
"selection_access_allowed_by_split": {
|
||||
"train": False,
|
||||
"val": True,
|
||||
"calibration": True,
|
||||
},
|
||||
"samples": development_samples,
|
||||
}
|
||||
development["manifest_sha256"] = canonical_hash(development)
|
||||
protected = {
|
||||
**common,
|
||||
"manifest_role": "protected_release_only",
|
||||
"allowed_splits": sorted(PROTECTED_SPLITS),
|
||||
"training_access_allowed": False,
|
||||
"selection_use_allowed": False,
|
||||
"labels_available_by_split": {
|
||||
"test": "frozen_evaluator_only",
|
||||
"background-test": "frozen_evaluator_only",
|
||||
"challenge": "sealed_external",
|
||||
},
|
||||
"access_policy": (
|
||||
"Only the frozen release evaluator may consume test/background-test labels; "
|
||||
"challenge labels remain external/sealed."
|
||||
),
|
||||
"samples": protected_samples,
|
||||
}
|
||||
protected["manifest_sha256"] = canonical_hash(protected)
|
||||
leakage = {
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"generator_version": GENERATOR_VERSION,
|
||||
"status": "pass" if not findings else "fail",
|
||||
"source_manifest_sha256": source_hash,
|
||||
"development_manifest_sha256": development["manifest_sha256"],
|
||||
"protected_manifest_sha256": protected["manifest_sha256"],
|
||||
"inventory_total": len(samples),
|
||||
"split_counts": dict(sorted(split_counts.items())),
|
||||
"checked_identity_fields": [
|
||||
"sample_id",
|
||||
"group_id",
|
||||
"source_family",
|
||||
"temporal_family",
|
||||
"object_ids",
|
||||
"native_feature_ids",
|
||||
"raw_image_sha256",
|
||||
"processed_image_sha256",
|
||||
"label_sha256",
|
||||
"record_sha256",
|
||||
"perceptual_image_hash",
|
||||
"label_geometry_hash",
|
||||
"label_geometry_fingerprint",
|
||||
"parent_raster_id",
|
||||
"acquisition_id",
|
||||
"bbox_distance",
|
||||
],
|
||||
"crs_validation": {"status": "pass", "crs": metric_crs, "distance_units": "m"},
|
||||
"independence_buffer_m": buffer_m,
|
||||
"perceptual_hamming_threshold": perceptual_threshold,
|
||||
"label_geometry_hamming_threshold": geometry_threshold,
|
||||
"finding_count": len(findings),
|
||||
"findings": findings,
|
||||
}
|
||||
return development, protected, leakage
|
||||
|
||||
|
||||
def protected_identities(protected_manifest: dict[str, Any]) -> set[str]:
|
||||
identities: set[str] = set()
|
||||
scalar_fields = (
|
||||
"sample_id",
|
||||
"group_id",
|
||||
"record_sha256",
|
||||
"source_family",
|
||||
"temporal_family",
|
||||
"raw_image_sha256",
|
||||
"processed_image_sha256",
|
||||
"label_sha256",
|
||||
"label_geometry_hash",
|
||||
"perceptual_image_hash",
|
||||
"label_geometry_fingerprint",
|
||||
"parent_raster_id",
|
||||
"acquisition_id",
|
||||
)
|
||||
for item in protected_manifest.get("samples", []):
|
||||
identities.update(str(item[field]) for field in scalar_fields)
|
||||
identities.update(str(value) for value in item.get("object_ids", []))
|
||||
identities.update(str(value) for value in item.get("native_feature_ids", []))
|
||||
return identities
|
||||
|
||||
|
||||
def assert_training_inputs_safe(
|
||||
input_paths: Iterable[Path],
|
||||
input_records: Iterable[dict[str, Any]],
|
||||
protected_manifest: dict[str, Any],
|
||||
) -> None:
|
||||
"""Refuse non-train roles and protected identities at a fitting boundary."""
|
||||
forbidden = protected_identities(protected_manifest)
|
||||
violations: list[str] = []
|
||||
for path in input_paths:
|
||||
lowered = path.as_posix().lower()
|
||||
if any(
|
||||
token in lowered
|
||||
for token in ("protected", "holdout", "challenge", "background-test")
|
||||
):
|
||||
violations.append(f"protected_path:{path}")
|
||||
scalar_fields = (
|
||||
"sample_id",
|
||||
"group_id",
|
||||
"record_sha256",
|
||||
"source_family",
|
||||
"temporal_family",
|
||||
"raw_image_sha256",
|
||||
"processed_image_sha256",
|
||||
"label_sha256",
|
||||
"label_geometry_hash",
|
||||
"perceptual_image_hash",
|
||||
"label_geometry_fingerprint",
|
||||
"parent_raster_id",
|
||||
"acquisition_id",
|
||||
)
|
||||
for record in input_records:
|
||||
if record.get("split") not in TRAIN_SPLITS:
|
||||
violations.append(
|
||||
f"non_train_role:{record.get('sample_id')}:{record.get('split')}"
|
||||
)
|
||||
values = {str(record.get(field, "")) for field in scalar_fields}
|
||||
values.update(str(value) for value in record.get("object_ids", []))
|
||||
values.update(str(value) for value in record.get("native_feature_ids", []))
|
||||
overlap = sorted((values - {""}) & forbidden)
|
||||
if overlap:
|
||||
violations.append(f"protected_identity:{','.join(overlap)}")
|
||||
if violations:
|
||||
raise LeakageError(
|
||||
"Training input firewall blocked: " + "; ".join(sorted(violations))
|
||||
)
|
||||
|
||||
|
||||
def generate(source_path: Path, output_dir: Path) -> dict[str, Any]:
|
||||
source = json.loads(source_path.read_text(encoding="utf-8"))
|
||||
development, protected, leakage = build_manifests(source)
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
generation_status = {
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"generator_version": GENERATOR_VERSION,
|
||||
"status": leakage["status"],
|
||||
"source_manifest_sha256": leakage["source_manifest_sha256"],
|
||||
"development_manifest_sha256": (
|
||||
development["manifest_sha256"] if leakage["status"] == "pass" else None
|
||||
),
|
||||
"protected_manifest_sha256": (
|
||||
protected["manifest_sha256"] if leakage["status"] == "pass" else None
|
||||
),
|
||||
}
|
||||
write_json(output_dir / "generation-status.json", generation_status)
|
||||
write_json(output_dir / "leakage-gate-report.json", leakage)
|
||||
if leakage["status"] != "pass":
|
||||
raise LeakageError(
|
||||
f"Leakage gate failed with {leakage['finding_count']} findings"
|
||||
)
|
||||
train_samples = [
|
||||
item for item in development["samples"] if item["split"] == "train"
|
||||
]
|
||||
assert_training_inputs_safe([], train_samples, protected)
|
||||
write_json(output_dir / "development-split-manifest.json", development)
|
||||
write_json(output_dir / "protected-split-manifest.json", protected)
|
||||
return {
|
||||
"development": development,
|
||||
"protected": protected,
|
||||
"leakage": leakage,
|
||||
"generation_status": generation_status,
|
||||
}
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--source", type=Path, required=True)
|
||||
parser.add_argument("--output-dir", type=Path, required=True)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
try:
|
||||
result = generate(args.source, args.output_dir)
|
||||
except (OSError, json.JSONDecodeError, LeakageError) as exc:
|
||||
print(json.dumps({"status": "fail", "error": str(exc)}, indent=2))
|
||||
return 2
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"status": "pass",
|
||||
"development_manifest_sha256": result["development"]["manifest_sha256"],
|
||||
"protected_manifest_sha256": result["protected"]["manifest_sha256"],
|
||||
"split_counts": result["leakage"]["split_counts"],
|
||||
},
|
||||
indent=2,
|
||||
)
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,996 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run the complete local Phase 4 evaluation workflow from frozen inputs."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import platform
|
||||
import subprocess
|
||||
import sys
|
||||
from importlib import metadata as importlib_metadata
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
BACKEND_ROOT = ROOT / "backend"
|
||||
for entry in (str(ROOT), str(BACKEND_ROOT), str(ROOT / "scripts")):
|
||||
if entry not in sys.path:
|
||||
sys.path.insert(0, entry)
|
||||
|
||||
from accuracy_phase4_evaluator import EVALUATOR_VERSION, canonical_hash, evaluate_cases # noqa: E402
|
||||
from generate_accuracy_phase4_splits import ( # noqa: E402
|
||||
GENERATOR_VERSION,
|
||||
LeakageError,
|
||||
assert_training_inputs_safe,
|
||||
build_manifests,
|
||||
)
|
||||
from run_golden_qa_benchmark import run_benchmark # noqa: E402
|
||||
|
||||
WORKFLOW_VERSION = "2.0.0"
|
||||
BENCHMARK_ID = "geointel-p4-reference-harness-v2"
|
||||
GATE_STATES = {"pass", "fail", "not_evaluable"}
|
||||
|
||||
|
||||
class EvidenceConflictError(RuntimeError):
|
||||
"""Raised when an immutable evidence path already contains different bytes."""
|
||||
|
||||
|
||||
def sha256(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as stream:
|
||||
for chunk in iter(lambda: stream.read(1024 * 1024), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def json_bytes(payload: Any) -> bytes:
|
||||
return (
|
||||
json.dumps(payload, ensure_ascii=False, indent=2, sort_keys=True) + "\n"
|
||||
).encode("utf-8")
|
||||
|
||||
|
||||
def write_json_immutable(path: Path, payload: Any) -> None:
|
||||
content = json_bytes(payload)
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
if path.exists():
|
||||
if path.read_bytes() != content:
|
||||
raise EvidenceConflictError(
|
||||
f"Refusing to overwrite immutable evidence with different content: {path}"
|
||||
)
|
||||
return
|
||||
temporary = path.with_name(f".{path.name}.tmp")
|
||||
temporary.write_bytes(content)
|
||||
temporary.replace(path)
|
||||
|
||||
|
||||
def repository_commit(repo_root: Path) -> str | None:
|
||||
try:
|
||||
return subprocess.run(
|
||||
["git", "rev-parse", "HEAD"],
|
||||
cwd=repo_root,
|
||||
check=True,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
).stdout.strip()
|
||||
except (OSError, subprocess.CalledProcessError):
|
||||
return None
|
||||
|
||||
|
||||
def dependency_version(distribution: str) -> str | None:
|
||||
try:
|
||||
return importlib_metadata.version(distribution)
|
||||
except importlib_metadata.PackageNotFoundError:
|
||||
return None
|
||||
|
||||
|
||||
def repository_file(repo_root: Path, relative_path: str) -> dict[str, Any]:
|
||||
path = repo_root / relative_path
|
||||
return {
|
||||
"path": relative_path,
|
||||
"sha256": sha256(path),
|
||||
"size_bytes": path.stat().st_size,
|
||||
}
|
||||
|
||||
|
||||
def canonical_golden_baseline() -> dict[str, Any]:
|
||||
result = run_benchmark()
|
||||
scenarios = []
|
||||
for item in result["scenarios"]:
|
||||
normalized = dict(item)
|
||||
normalized.pop("quality_check_id", None)
|
||||
scenarios.append(normalized)
|
||||
return {
|
||||
"status": result["status"],
|
||||
"version": result["version"],
|
||||
"scenario_count": result["scenario_count"],
|
||||
"scenarios": scenarios,
|
||||
"persistence": result["persistence"],
|
||||
"implementation": "backend/app/services/qa_service.py via scripts/run_golden_qa_benchmark.py",
|
||||
"claim_boundary": "Reference implementation regression evidence; not production model accuracy.",
|
||||
"content_sha256": canonical_hash(scenarios),
|
||||
}
|
||||
|
||||
|
||||
def product_baseline_manifest_gate(
|
||||
repo_root: Path,
|
||||
manifest_path: Path,
|
||||
active_model: dict[str, Any],
|
||||
) -> dict[str, Any]:
|
||||
relative = manifest_path
|
||||
try:
|
||||
relative = manifest_path.resolve().relative_to(repo_root.resolve())
|
||||
except (OSError, ValueError):
|
||||
return {
|
||||
"status": "fail",
|
||||
"reason": "Product baseline manifest must reside inside the governed repository evidence root.",
|
||||
"path": str(manifest_path),
|
||||
}
|
||||
if not manifest_path.is_file():
|
||||
return {
|
||||
"status": "not_evaluable",
|
||||
"reason": "No executed, hash-bound product incumbent baseline manifest is available.",
|
||||
"expected_path": relative.as_posix(),
|
||||
}
|
||||
try:
|
||||
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
|
||||
except (OSError, json.JSONDecodeError) as exc:
|
||||
return {
|
||||
"status": "fail",
|
||||
"reason": f"Unreadable product baseline manifest: {exc}",
|
||||
}
|
||||
required_keys = {
|
||||
"schema_version",
|
||||
"status",
|
||||
"synthetic",
|
||||
"active_model_sha256",
|
||||
"evaluator_sha256",
|
||||
"configuration_sha256",
|
||||
"protected_split_manifest",
|
||||
"authoritative_reference_manifest",
|
||||
"raw_predictions",
|
||||
"metric_report",
|
||||
"inference",
|
||||
}
|
||||
missing = sorted(required_keys - set(manifest))
|
||||
violations: list[str] = []
|
||||
if missing:
|
||||
violations.append(f"missing_fields:{','.join(missing)}")
|
||||
if manifest.get("status") != "pass":
|
||||
violations.append("manifest_status_not_pass")
|
||||
if manifest.get("synthetic") is not False:
|
||||
violations.append("synthetic_or_unspecified")
|
||||
if manifest.get("active_model_sha256") != active_model.get("sha256"):
|
||||
violations.append("active_model_hash_mismatch")
|
||||
evaluator_path = repo_root / "scripts/accuracy_phase4_evaluator.py"
|
||||
if manifest.get("evaluator_sha256") != sha256(evaluator_path):
|
||||
violations.append("evaluator_hash_mismatch")
|
||||
inference = manifest.get("inference") or {}
|
||||
if inference.get("executed") is not True:
|
||||
violations.append("inference_not_executed")
|
||||
if inference.get("test_used_for_selection") is not False:
|
||||
violations.append("protected_test_selection_policy_invalid")
|
||||
if not str(inference.get("device") or "").lower().startswith("cuda"):
|
||||
violations.append("governed_cuda_execution_not_proven")
|
||||
checked_artifacts: list[dict[str, Any]] = []
|
||||
for key in (
|
||||
"protected_split_manifest",
|
||||
"authoritative_reference_manifest",
|
||||
"raw_predictions",
|
||||
"metric_report",
|
||||
):
|
||||
item = manifest.get(key) or {}
|
||||
item_path = repo_root / str(item.get("path") or "")
|
||||
try:
|
||||
item_path.resolve().relative_to(repo_root.resolve())
|
||||
except (OSError, ValueError):
|
||||
violations.append(f"{key}_outside_repository")
|
||||
continue
|
||||
if not item_path.is_file():
|
||||
violations.append(f"{key}_missing")
|
||||
continue
|
||||
observed_hash = sha256(item_path)
|
||||
checked_artifacts.append(
|
||||
{
|
||||
"role": key,
|
||||
"path": item_path.relative_to(repo_root).as_posix(),
|
||||
"sha256": observed_hash,
|
||||
}
|
||||
)
|
||||
if observed_hash != item.get("sha256"):
|
||||
violations.append(f"{key}_hash_mismatch")
|
||||
return {
|
||||
"status": "fail" if violations else "pass",
|
||||
"path": relative.as_posix(),
|
||||
"manifest_sha256": sha256(manifest_path),
|
||||
"violations": sorted(violations),
|
||||
"checked_artifacts": checked_artifacts,
|
||||
"evidence": (
|
||||
"A non-synthetic active-model inference, protected split, authority reference, "
|
||||
"raw predictions and metric report are all checksum-bound."
|
||||
if not violations
|
||||
else None
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def readiness_snapshot(repo_root: Path) -> dict[str, Any]:
|
||||
status_path = repo_root / "docs/accuracy-program/status.json"
|
||||
p3_path = repo_root / "artifacts/evidence/accuracy/P3/full-scan-manifest.json"
|
||||
leakage_path = repo_root / "artifacts/evidence/accuracy/P3/leakage-report.json"
|
||||
status = json.loads(status_path.read_text(encoding="utf-8"))
|
||||
p3 = json.loads(p3_path.read_text(encoding="utf-8"))
|
||||
leakage = json.loads(leakage_path.read_text(encoding="utf-8"))
|
||||
ml_data = status.get("ml_data") or {}
|
||||
return {
|
||||
"schema_version": 1,
|
||||
"source_paths": {
|
||||
"phase3_full_scan": repository_file(
|
||||
repo_root, "artifacts/evidence/accuracy/P3/full-scan-manifest.json"
|
||||
),
|
||||
"phase3_leakage": repository_file(
|
||||
repo_root, "artifacts/evidence/accuracy/P3/leakage-report.json"
|
||||
),
|
||||
},
|
||||
"active_model": (status.get("runtime") or {}).get("active_model"),
|
||||
"v56_review_and_split": (ml_data.get("v56") or {}),
|
||||
"protected_test_isolation": ml_data.get("protected_test_isolation"),
|
||||
"phase3_scan": {
|
||||
"scan_id": p3.get("scan_id"),
|
||||
"content_hash": p3.get("content_hash"),
|
||||
"grb_consistency": p3.get("grb_consistency"),
|
||||
},
|
||||
"phase3_leakage_status": leakage.get("status"),
|
||||
"authority_requirements": [
|
||||
{"task": "building_validation", "zone": "flanders", "primary": "grb"},
|
||||
{"task": "building_validation", "zone": "wallonia", "primary": "picc"},
|
||||
{"task": "building_validation", "zone": "brussels", "primary": "urbis"},
|
||||
{"task": "terrain_height", "zone": "flanders", "primary": "dhmv"},
|
||||
{"task": "terrain_height", "zone": "wallonia", "primary": "spw_terrain"},
|
||||
{
|
||||
"task": "north_sea_bathymetry",
|
||||
"zone": "belgian_north_sea",
|
||||
"primary": "mdk",
|
||||
},
|
||||
{
|
||||
"task": "imagery_corroboration",
|
||||
"zone": "belgium",
|
||||
"primary": "official_orthophoto",
|
||||
"contextual": "sentinel-2",
|
||||
},
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def product_gate_evidence(
|
||||
repo_root: Path,
|
||||
snapshot: dict[str, Any],
|
||||
product_baseline_manifest: Path,
|
||||
) -> dict[str, Any]:
|
||||
v56 = snapshot["v56_review_and_split"]
|
||||
active_model = snapshot["active_model"] or {}
|
||||
configured_path = Path(str(active_model.get("path") or ""))
|
||||
p3_grb = snapshot["phase3_scan"].get("grb_consistency") or {}
|
||||
baseline_gate = product_baseline_manifest_gate(
|
||||
repo_root,
|
||||
product_baseline_manifest,
|
||||
active_model,
|
||||
)
|
||||
return {
|
||||
"active_model_available_and_hash_verified": {
|
||||
"status": "pass"
|
||||
if configured_path.is_file()
|
||||
and sha256(configured_path) == active_model.get("sha256")
|
||||
else "not_evaluable",
|
||||
"configured_path": str(configured_path),
|
||||
"configured_sha256": active_model.get("sha256"),
|
||||
"reason": None
|
||||
if configured_path.is_file()
|
||||
else "Configured active model is not locally accessible.",
|
||||
},
|
||||
"authoritative_reference_portfolio_available": {
|
||||
"status": "not_evaluable",
|
||||
"requirements": snapshot["authority_requirements"],
|
||||
"observed_grb": p3_grb,
|
||||
"reason": "A task- and zone-complete governed GRB/PICC/UrbIS/DHMV/SPW/MDK reference portfolio is not locally accessible.",
|
||||
},
|
||||
"human_review_complete": {
|
||||
"status": "pass" if bool(v56.get("review_complete")) else "fail",
|
||||
"observed": v56.get("reviewed_sample_count"),
|
||||
"required": v56.get("sample_count"),
|
||||
},
|
||||
"split_independence": {
|
||||
"status": "pass" if bool(v56.get("split_independence_proven")) else "fail",
|
||||
"cross_split_pairs_below_2000_m": v56.get("cross_split_pairs_below_2000_m"),
|
||||
},
|
||||
"phase3_leakage_resolved": {
|
||||
"status": "pass"
|
||||
if snapshot.get("phase3_leakage_status") == "pass"
|
||||
else "not_evaluable",
|
||||
"observed": snapshot.get("phase3_leakage_status"),
|
||||
},
|
||||
"protected_storage_isolation": {
|
||||
"status": "pass"
|
||||
if snapshot.get("protected_test_isolation") is True
|
||||
else "not_evaluable",
|
||||
"reason": "A physically isolated vault, scoped credentials and immutable access log are not proven.",
|
||||
},
|
||||
"executed_product_incumbent_baseline": baseline_gate,
|
||||
"representative_product_subgroup_support": {
|
||||
"status": "not_evaluable",
|
||||
"reason": "No real protected raw-prediction portfolio is available for AOI/region/context subgroup support.",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def build_release_gate_report(
|
||||
split_result: dict[str, Any],
|
||||
evaluation: dict[str, Any],
|
||||
portfolio: dict[str, Any],
|
||||
golden: dict[str, Any],
|
||||
firewall_checks: dict[str, bool],
|
||||
product_gates: dict[str, Any],
|
||||
) -> dict[str, Any]:
|
||||
declared_families = {item["task"] for item in evaluation["task_inventory"]}
|
||||
observed_families = {item["task"] for item in evaluation["results"]}
|
||||
required_split_roles = {"train", "val", "calibration", "test", "background-test"}
|
||||
observed_split_roles = set(split_result["leakage"]["split_counts"])
|
||||
required_raw_fields = {
|
||||
"references",
|
||||
"predictions_pre_filter",
|
||||
"predictions_post_filter",
|
||||
"config",
|
||||
"input_lineage",
|
||||
"portfolio_lineage",
|
||||
}
|
||||
raw_violations = [
|
||||
item["sample_id"]
|
||||
for item in evaluation["results"]
|
||||
if not required_raw_fields <= set(item.get("raw") or {})
|
||||
]
|
||||
protected_policy = portfolio.get("protected_policy") or {}
|
||||
selection_contract_valid = (
|
||||
protected_policy.get("operating_point_selection_allowed") is False
|
||||
and protected_policy.get("diagnostic_curves_select_operating_point") is False
|
||||
and protected_policy.get("test_feedback_allowed") is False
|
||||
and protected_policy.get("threshold_selection_source")
|
||||
== "pre_registered_configuration_only"
|
||||
and all(
|
||||
isinstance((item.get("raw") or {}).get("config"), dict)
|
||||
for item in evaluation["results"]
|
||||
)
|
||||
)
|
||||
empty_case = next(
|
||||
(
|
||||
item
|
||||
for item in evaluation["results"]
|
||||
if item["sample_id"] == "background-test-pure-empty"
|
||||
),
|
||||
None,
|
||||
)
|
||||
empty_metrics = (empty_case or {}).get("metrics") or {}
|
||||
null_semantics_valid = (
|
||||
empty_case is not None
|
||||
and empty_metrics.get("reference_count") == 0
|
||||
and empty_metrics.get("prediction_count") == 0
|
||||
and empty_metrics.get("precision") is None
|
||||
and empty_metrics.get("recall") is None
|
||||
and empty_metrics.get("f1") is None
|
||||
)
|
||||
subgroup_report = evaluation.get("subgroups") or {}
|
||||
subgroup_contract_valid = (
|
||||
subgroup_report.get("overall_status")
|
||||
in {"not_evaluable", "evaluable_no_release_target"}
|
||||
and isinstance(subgroup_report.get("dimensions"), dict)
|
||||
and bool(subgroup_report.get("dimensions"))
|
||||
and all(
|
||||
isinstance(dimension.get("strata"), dict)
|
||||
and isinstance(dimension.get("worst_stratum_by_task"), dict)
|
||||
for dimension in subgroup_report["dimensions"].values()
|
||||
)
|
||||
)
|
||||
capability_inventory = evaluation.get("task_inventory") or []
|
||||
capability_contract_valid = bool(capability_inventory) and all(
|
||||
item.get("capability_id")
|
||||
and item.get("implementation_paths")
|
||||
and item.get("suitable_metrics")
|
||||
and item.get("evaluation_status")
|
||||
in {
|
||||
"synthetic_contract_case_only",
|
||||
"covered_by_family_not_separately_benchmarked",
|
||||
"not_separately_benchmarked",
|
||||
"no_independent_accuracy_score_underlying_tool_results_are_authoritative",
|
||||
"synthetic_metric_contract_only_no_generic_learned_classifier_claim",
|
||||
}
|
||||
for item in capability_inventory
|
||||
)
|
||||
local_gates = {
|
||||
"all_declared_evaluator_families_exercised": {
|
||||
"status": "pass" if declared_families == observed_families else "fail",
|
||||
"declared": sorted(declared_families),
|
||||
"observed": sorted(observed_families),
|
||||
},
|
||||
"implemented_capability_inventory": {
|
||||
"status": "pass" if capability_contract_valid else "fail",
|
||||
"capability_count": len(capability_inventory),
|
||||
},
|
||||
"normative_split_roles_and_leakage": {
|
||||
"status": (
|
||||
"pass"
|
||||
if required_split_roles <= observed_split_roles
|
||||
and split_result["leakage"]["status"] == "pass"
|
||||
else "fail"
|
||||
),
|
||||
"required_roles": sorted(required_split_roles),
|
||||
"observed_roles": sorted(observed_split_roles),
|
||||
"leakage_status": split_result["leakage"]["status"],
|
||||
},
|
||||
"manifest_training_firewall_contract": {
|
||||
"status": "pass"
|
||||
if firewall_checks and all(firewall_checks.values())
|
||||
else "fail",
|
||||
"checks": firewall_checks,
|
||||
},
|
||||
"protected_operating_point_contract": {
|
||||
"status": "pass" if selection_contract_valid else "fail",
|
||||
"evidence": (
|
||||
"Protected cases carry pre-registered configurations. Fixed AP/risk-coverage "
|
||||
"diagnostics cannot select an operating point or feed back into training."
|
||||
),
|
||||
},
|
||||
"complete_raw_predictions_retained": {
|
||||
"status": "pass" if not raw_violations else "fail",
|
||||
"violating_samples": raw_violations,
|
||||
},
|
||||
"reference_implementation_baseline": {
|
||||
"status": "pass" if golden.get("status") == "passed" else "fail",
|
||||
},
|
||||
"stratified_metric_contract": {
|
||||
"status": "pass" if subgroup_contract_valid else "fail",
|
||||
"observed_overall_status": subgroup_report.get("overall_status"),
|
||||
},
|
||||
"undefined_metric_truth_table": {
|
||||
"status": "pass" if null_semantics_valid else "fail",
|
||||
"sample_id": "background-test-pure-empty",
|
||||
},
|
||||
}
|
||||
invalid_gate_states = {
|
||||
f"{family}.{name}": item.get("status")
|
||||
for family, gates in (("local", local_gates), ("product", product_gates))
|
||||
for name, item in gates.items()
|
||||
if item.get("status") not in GATE_STATES
|
||||
}
|
||||
local_green = not invalid_gate_states and all(
|
||||
item["status"] == "pass" for item in local_gates.values()
|
||||
)
|
||||
product_green = not invalid_gate_states and all(
|
||||
item["status"] == "pass" for item in product_gates.values()
|
||||
)
|
||||
if not local_green:
|
||||
overall_status = "fail"
|
||||
elif not product_green:
|
||||
overall_status = (
|
||||
"not_evaluable"
|
||||
if any(item["status"] == "not_evaluable" for item in product_gates.values())
|
||||
else "fail"
|
||||
)
|
||||
else:
|
||||
overall_status = "pass"
|
||||
return {
|
||||
"schema_version": 2,
|
||||
"gate_policy": "geointel-p4-evaluation-harness-v2",
|
||||
"status": overall_status,
|
||||
"phase_decision": "ready_for_phase5" if overall_status == "pass" else "blocked",
|
||||
"local_harness_status": "pass" if local_green else "fail",
|
||||
"product_benchmark_status": "pass" if product_green else "not_evaluable",
|
||||
"promotion_allowed": False,
|
||||
"numeric_model_release_targets": "not_frozen_without_reviewed_representative_incumbent_baseline",
|
||||
"local_gates": local_gates,
|
||||
"product_gates": product_gates,
|
||||
"invalid_gate_states": invalid_gate_states,
|
||||
"critical_subgroup_policy": (
|
||||
"Any required subgroup with insufficient support, missing metrics, a failed "
|
||||
"non-inferiority comparison or regression blocks promotion; averages cannot override it."
|
||||
),
|
||||
"decision": (
|
||||
"The local contract harness passes, but Phase 4 remains in progress and Phase 5 "
|
||||
"is not ready until a real protected product incumbent baseline is evaluable."
|
||||
if local_green and not product_green
|
||||
else "All Phase 4 completion gates pass."
|
||||
if local_green and product_green
|
||||
else "The local Phase 4 harness has failing contract gates."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def build_evidence_manifest(
|
||||
artifacts: dict[str, Any],
|
||||
benchmark_manifest: dict[str, Any],
|
||||
) -> dict[str, Any]:
|
||||
retained = []
|
||||
for name, payload in sorted(artifacts.items()):
|
||||
content = json_bytes(payload)
|
||||
retained.append(
|
||||
{
|
||||
"path": name,
|
||||
"sha256": hashlib.sha256(content).hexdigest(),
|
||||
"size_bytes": len(content),
|
||||
}
|
||||
)
|
||||
return {
|
||||
"schema_version": 2,
|
||||
"phase": "P4",
|
||||
"benchmark_id": benchmark_manifest["benchmark_id"],
|
||||
"artifacts": retained,
|
||||
"artifact_count": len(retained),
|
||||
"claim_boundary": (
|
||||
"Immutable local reference-harness evidence only; product accuracy and release "
|
||||
"remain blocked while product gates are not evaluable or fail."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def firewall_contract_checks(
|
||||
split_result: dict[str, Any],
|
||||
protected_cases_path: Path,
|
||||
) -> dict[str, bool]:
|
||||
development = split_result["development"]["samples"]
|
||||
protected = split_result["protected"]
|
||||
train = [item for item in development if item["split"] == "train"]
|
||||
validation = next(item for item in development if item["split"] == "val")
|
||||
protected_item = protected["samples"][0]
|
||||
checks: dict[str, bool] = {}
|
||||
try:
|
||||
assert_training_inputs_safe([], train, protected)
|
||||
except LeakageError:
|
||||
checks["clean_train_allowed"] = False
|
||||
else:
|
||||
checks["clean_train_allowed"] = True
|
||||
for name, paths, records in (
|
||||
("non_train_role_blocked", [], [validation]),
|
||||
("protected_path_blocked", [protected_cases_path], []),
|
||||
(
|
||||
"renamed_protected_lineage_blocked",
|
||||
[],
|
||||
[{**train[0], "source_family": protected_item["source_family"]}],
|
||||
),
|
||||
):
|
||||
try:
|
||||
assert_training_inputs_safe(paths, records, protected)
|
||||
except LeakageError:
|
||||
checks[name] = True
|
||||
else:
|
||||
checks[name] = False
|
||||
return checks
|
||||
|
||||
|
||||
def runtime_identity() -> dict[str, Any]:
|
||||
return {
|
||||
"python": platform.python_version(),
|
||||
"python_implementation": platform.python_implementation(),
|
||||
"platform": platform.platform(),
|
||||
"dependencies": {
|
||||
"numpy": dependency_version("numpy"),
|
||||
"pyproj": dependency_version("pyproj"),
|
||||
"shapely": dependency_version("shapely"),
|
||||
},
|
||||
"execution_device": "CPU deterministic evaluator arithmetic; no production model inference",
|
||||
"cuda_used_for_reference_harness": False,
|
||||
}
|
||||
|
||||
|
||||
def metric_results_without_raw(evaluation: dict[str, Any]) -> list[dict[str, Any]]:
|
||||
return [
|
||||
{key: value for key, value in item.items() if key not in {"raw", "failures"}}
|
||||
for item in evaluation["results"]
|
||||
]
|
||||
|
||||
|
||||
def build_input_manifest(
|
||||
repo_root: Path,
|
||||
snapshot: dict[str, Any],
|
||||
) -> dict[str, Any]:
|
||||
code_paths = [
|
||||
"scripts/accuracy_phase4_evaluator.py",
|
||||
"scripts/generate_accuracy_phase4_splits.py",
|
||||
"scripts/run_accuracy_phase4_benchmark.py",
|
||||
"scripts/run_golden_qa_benchmark.py",
|
||||
"backend/app/services/qa_service.py",
|
||||
]
|
||||
input_paths = [
|
||||
"fixtures/accuracy/p4/split-source-manifest.json",
|
||||
"fixtures/accuracy/p4/protected-baseline-cases.json",
|
||||
"fixtures/golden/golden_qa_benchmarks.json",
|
||||
"docs/accuracy-program/05-metric-framework.md",
|
||||
"docs/accuracy-program/07-source-authority-matrix.md",
|
||||
"artifacts/evidence/accuracy/P3/full-scan-manifest.json",
|
||||
"artifacts/evidence/accuracy/P3/leakage-report.json",
|
||||
]
|
||||
return {
|
||||
"schema_version": 2,
|
||||
"repository_commit": repository_commit(repo_root),
|
||||
"code": [repository_file(repo_root, path) for path in code_paths],
|
||||
"inputs": [repository_file(repo_root, path) for path in input_paths],
|
||||
"readiness_snapshot": snapshot,
|
||||
"readiness_snapshot_sha256": canonical_hash(snapshot),
|
||||
"runtime": runtime_identity(),
|
||||
"model_execution": {
|
||||
"status": "not_evaluable",
|
||||
"reason": "The configured active model and governed protected product inputs are not locally accessible.",
|
||||
"configured_active_model": snapshot.get("active_model"),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def run_workflow(
|
||||
repo_root: Path,
|
||||
output_dir: Path,
|
||||
product_baseline_manifest: Path | None = None,
|
||||
) -> dict[str, Any]:
|
||||
source_path = repo_root / "fixtures/accuracy/p4/split-source-manifest.json"
|
||||
cases_path = repo_root / "fixtures/accuracy/p4/protected-baseline-cases.json"
|
||||
source = json.loads(source_path.read_text(encoding="utf-8"))
|
||||
development, protected, leakage = build_manifests(source)
|
||||
if leakage["status"] != "pass":
|
||||
raise LeakageError(
|
||||
f"Leakage gate failed with {leakage['finding_count']} findings"
|
||||
)
|
||||
split_result = {
|
||||
"development": development,
|
||||
"protected": protected,
|
||||
"leakage": leakage,
|
||||
"generation_status": {
|
||||
"schema_version": 1,
|
||||
"generator_version": GENERATOR_VERSION,
|
||||
"status": "pass",
|
||||
"source_manifest_sha256": leakage["source_manifest_sha256"],
|
||||
"development_manifest_sha256": development["manifest_sha256"],
|
||||
"protected_manifest_sha256": protected["manifest_sha256"],
|
||||
},
|
||||
}
|
||||
protected_evaluation_ids = {
|
||||
item["sample_id"]
|
||||
for item in protected["samples"]
|
||||
if item["split"] in {"test", "background-test"}
|
||||
}
|
||||
portfolio = json.loads(cases_path.read_text(encoding="utf-8"))
|
||||
evaluation = evaluate_cases(cases_path, protected_evaluation_ids)
|
||||
firewall_checks = firewall_contract_checks(split_result, cases_path)
|
||||
golden = canonical_golden_baseline()
|
||||
snapshot = readiness_snapshot(repo_root)
|
||||
baseline_path = (
|
||||
product_baseline_manifest
|
||||
if product_baseline_manifest is not None
|
||||
else repo_root / "artifacts/evidence/accuracy/P4/product-baseline-manifest.json"
|
||||
)
|
||||
product_gates = product_gate_evidence(repo_root, snapshot, baseline_path)
|
||||
gate_report = build_release_gate_report(
|
||||
split_result,
|
||||
evaluation,
|
||||
portfolio,
|
||||
golden,
|
||||
firewall_checks,
|
||||
product_gates,
|
||||
)
|
||||
input_manifest = build_input_manifest(repo_root, snapshot)
|
||||
evaluation_contract = {
|
||||
"schema_version": 2,
|
||||
"benchmark_id": BENCHMARK_ID,
|
||||
"workflow_version": WORKFLOW_VERSION,
|
||||
"evaluator_version": EVALUATOR_VERSION,
|
||||
"split_generator_version": GENERATOR_VERSION,
|
||||
"evaluator_families": sorted(evaluation["evaluated_task_families"]),
|
||||
"implemented_capabilities": evaluation["task_inventory"],
|
||||
"metric_contract": repository_file(
|
||||
repo_root, "docs/accuracy-program/05-metric-framework.md"
|
||||
),
|
||||
"gate_states": sorted(GATE_STATES),
|
||||
"protected_policy": evaluation["protected_policy"],
|
||||
"undefined_value_policy": (
|
||||
"Undefined denominators are null with numerator, denominator and support; "
|
||||
"they are never coerced to a perfect score."
|
||||
),
|
||||
"claim_boundary": evaluation["claim_boundary"],
|
||||
}
|
||||
raw_items = [item["raw"] for item in evaluation["results"]]
|
||||
raw_predictions = {
|
||||
"schema_version": 2,
|
||||
"evaluator_version": EVALUATOR_VERSION,
|
||||
"portfolio_file_sha256": evaluation["portfolio_file_sha256"],
|
||||
"items": raw_items,
|
||||
"items_canonical_json_sha256": canonical_hash(raw_items),
|
||||
"hash_specification": evaluation["hash_specification"],
|
||||
}
|
||||
metric_results = metric_results_without_raw(evaluation)
|
||||
metric_report = {
|
||||
key: value
|
||||
for key, value in evaluation.items()
|
||||
if key not in {"results", "failures"}
|
||||
}
|
||||
metric_report["results"] = metric_results
|
||||
metric_report["metric_results_canonical_json_sha256"] = canonical_hash(
|
||||
metric_results
|
||||
)
|
||||
metric_report["full_results_canonical_json_sha256"] = evaluation[
|
||||
"results_canonical_json_sha256"
|
||||
]
|
||||
failure_gallery = {
|
||||
"schema_version": 2,
|
||||
"taxonomy": "docs/accuracy-program/05-metric-framework.md section 4",
|
||||
"failure_count": len(evaluation["failures"]),
|
||||
"items": evaluation["failures"],
|
||||
"items_canonical_json_sha256": canonical_hash(evaluation["failures"]),
|
||||
"rendering_status": (
|
||||
"machine_readable_examples_retained; a visual production gallery requires "
|
||||
"controlled access to protected imagery"
|
||||
),
|
||||
}
|
||||
taxonomy_entries = sorted(
|
||||
{(item["error_code"], item["kind"]) for item in evaluation["failures"]}
|
||||
)
|
||||
error_taxonomy = {
|
||||
"schema_version": 2,
|
||||
"source": "docs/accuracy-program/05-metric-framework.md section 4",
|
||||
"observed_codes": [
|
||||
{"error_code": code, "kind": kind} for code, kind in taxonomy_entries
|
||||
],
|
||||
"observed_failure_count": len(evaluation["failures"]),
|
||||
"claim_boundary": evaluation["claim_boundary"],
|
||||
}
|
||||
object_task_names = {
|
||||
"object_detection",
|
||||
"footprint_segmentation",
|
||||
"vector_comparison",
|
||||
"change_detection",
|
||||
"geospatial_data_validation",
|
||||
}
|
||||
object_metrics = {
|
||||
"schema_version": 2,
|
||||
"status": "fixture_contract_only",
|
||||
"results": [
|
||||
item for item in metric_results if item["task"] in object_task_names
|
||||
],
|
||||
}
|
||||
tile_metrics = {
|
||||
"schema_version": 2,
|
||||
"status": "fixture_contract_only",
|
||||
"results": [
|
||||
item
|
||||
for item in metric_results
|
||||
if item["task"] in {"raster_classification", "terrain_interpretation"}
|
||||
],
|
||||
}
|
||||
aoi_metrics = {
|
||||
"schema_version": 2,
|
||||
"status": "not_evaluable",
|
||||
"reason": (
|
||||
"Synthetic single-case fixtures do not provide independent product AOI clusters. "
|
||||
"AOI micro/macro and cluster-bootstrap evidence requires the protected product corpus."
|
||||
),
|
||||
"required_future_outputs": [
|
||||
"per-AOI primary metrics",
|
||||
"micro and macro aggregation",
|
||||
"paired candidate-minus-incumbent deltas",
|
||||
"cluster-bootstrap confidence intervals",
|
||||
],
|
||||
}
|
||||
stratified_metrics = evaluation["subgroups"]
|
||||
calibration_items = []
|
||||
for item in metric_results:
|
||||
calibration = item["metrics"].get("calibration")
|
||||
coverage_risk = item["metrics"].get("coverage_risk")
|
||||
if calibration is not None or coverage_risk is not None:
|
||||
calibration_items.append(
|
||||
{
|
||||
"sample_id": item["sample_id"],
|
||||
"task": item["task"],
|
||||
"calibration": calibration,
|
||||
"coverage_risk": coverage_risk,
|
||||
}
|
||||
)
|
||||
calibration_metrics = {
|
||||
"schema_version": 2,
|
||||
"status": "fixture_diagnostic_only",
|
||||
"selection_allowed": False,
|
||||
"items": calibration_items,
|
||||
"note": (
|
||||
"Fixed diagnostic bins and risk thresholds test metric arithmetic; they do not "
|
||||
"select or change any operating point."
|
||||
),
|
||||
}
|
||||
latency_reliability = {
|
||||
"schema_version": 2,
|
||||
"status": "not_evaluable",
|
||||
"model_inference_executed": False,
|
||||
"reason": (
|
||||
"The reference harness performs deterministic evaluator arithmetic only. "
|
||||
"GPU latency, VRAM, throughput and failure-rate gates require the real active model."
|
||||
),
|
||||
}
|
||||
human_review_summary = {
|
||||
"schema_version": 2,
|
||||
"status": product_gates["human_review_complete"]["status"],
|
||||
"reviewed": product_gates["human_review_complete"].get("observed"),
|
||||
"required": product_gates["human_review_complete"].get("required"),
|
||||
"source": "readiness-snapshot.json bound to Phase-1/3 evidence",
|
||||
"ai_review_is_human_signoff": False,
|
||||
}
|
||||
candidate_vs_incumbent = {
|
||||
"schema_version": 2,
|
||||
"status": "not_evaluable",
|
||||
"reason": (
|
||||
"Phase 4 has no valid real incumbent product baseline and no pre-registered "
|
||||
"candidate; synthetic fixture values cannot define non-inferiority."
|
||||
),
|
||||
"future_gate_contract": {
|
||||
"unit": "paired independent AOI",
|
||||
"global_and_critical_subgroups_required": True,
|
||||
"missing_or_insufficient_support": "not_evaluable",
|
||||
"aggregate_improvement_may_mask_subgroup_regression": False,
|
||||
"numeric_margin": "to_be_frozen_before_protected_access",
|
||||
},
|
||||
}
|
||||
input_manifest_file_sha256 = hashlib.sha256(json_bytes(input_manifest)).hexdigest()
|
||||
benchmark_manifest = {
|
||||
"schema_version": 2,
|
||||
"benchmark_id": BENCHMARK_ID,
|
||||
"workflow_version": WORKFLOW_VERSION,
|
||||
"evaluator_version": EVALUATOR_VERSION,
|
||||
"split_generator_version": GENERATOR_VERSION,
|
||||
"repository_commit": input_manifest["repository_commit"],
|
||||
"input_manifest": {
|
||||
"path": "input-manifest.json",
|
||||
"sha256": input_manifest_file_sha256,
|
||||
},
|
||||
"inputs": {
|
||||
"split_source": repository_file(
|
||||
repo_root, "fixtures/accuracy/p4/split-source-manifest.json"
|
||||
),
|
||||
"protected_cases": repository_file(
|
||||
repo_root, "fixtures/accuracy/p4/protected-baseline-cases.json"
|
||||
),
|
||||
"golden_qa_manifest": repository_file(
|
||||
repo_root, "fixtures/golden/golden_qa_benchmarks.json"
|
||||
),
|
||||
"phase3_full_scan": repository_file(
|
||||
repo_root, "artifacts/evidence/accuracy/P3/full-scan-manifest.json"
|
||||
),
|
||||
"phase3_leakage": repository_file(
|
||||
repo_root, "artifacts/evidence/accuracy/P3/leakage-report.json"
|
||||
),
|
||||
},
|
||||
"code": input_manifest["code"],
|
||||
"runtime": input_manifest["runtime"],
|
||||
"split_manifests": {
|
||||
"development_sha256": development["manifest_sha256"],
|
||||
"protected_sha256": protected["manifest_sha256"],
|
||||
"leakage_status": leakage["status"],
|
||||
},
|
||||
"inference_and_selection": {
|
||||
"synthetic_reference_harness": True,
|
||||
"production_model_inference_executed": False,
|
||||
"test_used_for_selection": False,
|
||||
"background_test_used_for_selection": False,
|
||||
"challenge_labels_available": False,
|
||||
"raw_predictions_retained": True,
|
||||
"threshold_source": "pre_registered_configuration_only",
|
||||
},
|
||||
"evaluation_results_canonical_json_sha256": evaluation[
|
||||
"results_canonical_json_sha256"
|
||||
],
|
||||
"reference_baseline_sha256": golden["content_sha256"],
|
||||
"claim_boundary": evaluation["claim_boundary"],
|
||||
}
|
||||
benchmark_manifest["manifest_sha256"] = canonical_hash(benchmark_manifest)
|
||||
gate_report["benchmark_manifest_sha256"] = benchmark_manifest["manifest_sha256"]
|
||||
workflow_summary = {
|
||||
"schema_version": 2,
|
||||
"status": gate_report["status"],
|
||||
"phase_decision": gate_report["phase_decision"],
|
||||
"local_harness_status": gate_report["local_harness_status"],
|
||||
"product_benchmark_status": gate_report["product_benchmark_status"],
|
||||
"benchmark_manifest_sha256": benchmark_manifest["manifest_sha256"],
|
||||
"split_counts": leakage["split_counts"],
|
||||
"task_family_count": evaluation["task_count"],
|
||||
"implemented_capability_count": len(evaluation["task_inventory"]),
|
||||
"case_count": evaluation["case_count"],
|
||||
"failure_example_count": len(evaluation["failures"]),
|
||||
"evaluation_results_canonical_json_sha256": evaluation[
|
||||
"results_canonical_json_sha256"
|
||||
],
|
||||
"reference_baseline_sha256": golden["content_sha256"],
|
||||
"promotion_allowed": False,
|
||||
"phase4_done": gate_report["status"] == "pass",
|
||||
"phase5_ready": gate_report["status"] == "pass",
|
||||
}
|
||||
artifacts: dict[str, Any] = {
|
||||
"acceptance-gates.json": gate_report,
|
||||
"aoi-metrics.json": aoi_metrics,
|
||||
"baseline-raw-predictions.json": raw_predictions,
|
||||
"benchmark-manifest.json": benchmark_manifest,
|
||||
"calibration-metrics.json": calibration_metrics,
|
||||
"candidate-vs-incumbent.json": candidate_vs_incumbent,
|
||||
"development-split-manifest.json": development,
|
||||
"error-taxonomy.json": error_taxonomy,
|
||||
"evaluation-contract.json": evaluation_contract,
|
||||
"failure-gallery.json": failure_gallery,
|
||||
"generation-status.json": split_result["generation_status"],
|
||||
"human-review-summary.json": human_review_summary,
|
||||
"input-manifest.json": input_manifest,
|
||||
"latency-and-reliability.json": latency_reliability,
|
||||
"leakage-gate-report.json": leakage,
|
||||
"metric-report.json": metric_report,
|
||||
"object-metrics.json": object_metrics,
|
||||
"protected-split-manifest.json": protected,
|
||||
"reference-implementation-baseline.json": golden,
|
||||
"release-gate-report.json": gate_report,
|
||||
"split-and-leakage-audit.json": leakage,
|
||||
"stratified-metrics.json": stratified_metrics,
|
||||
"tile-metrics.json": tile_metrics,
|
||||
"workflow-summary.json": workflow_summary,
|
||||
}
|
||||
evidence = build_evidence_manifest(artifacts, benchmark_manifest)
|
||||
for name, payload in sorted(artifacts.items()):
|
||||
write_json_immutable(output_dir / name, payload)
|
||||
write_json_immutable(output_dir / "evidence-manifest.json", evidence)
|
||||
return workflow_summary
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--repo-root", type=Path, default=ROOT)
|
||||
parser.add_argument(
|
||||
"--output-dir",
|
||||
type=Path,
|
||||
default=ROOT / "artifacts/evidence/accuracy/P4/reference-harness-v2",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--product-baseline-manifest",
|
||||
type=Path,
|
||||
help=(
|
||||
"Optional governed product incumbent manifest. It can pass only when real "
|
||||
"active-model inference and all referenced artifacts validate."
|
||||
),
|
||||
)
|
||||
parser.add_argument(
|
||||
"--allow-product-blocked",
|
||||
action="store_true",
|
||||
help=(
|
||||
"Return zero when the local harness passes while product evidence remains "
|
||||
"fail/not_evaluable. This never changes a gate or phase decision."
|
||||
),
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
try:
|
||||
summary = run_workflow(
|
||||
args.repo_root.resolve(),
|
||||
args.output_dir.resolve(),
|
||||
args.product_baseline_manifest.resolve()
|
||||
if args.product_baseline_manifest
|
||||
else None,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 - workflow evidence must fail closed
|
||||
print(
|
||||
json.dumps(
|
||||
{"status": "fail", "error": f"{type(exc).__name__}: {exc}"},
|
||||
indent=2,
|
||||
)
|
||||
)
|
||||
return 2
|
||||
print(json.dumps(summary, indent=2, sort_keys=True))
|
||||
if summary["status"] == "pass":
|
||||
return 0
|
||||
return (
|
||||
0
|
||||
if args.allow_product_blocked and summary["local_harness_status"] == "pass"
|
||||
else 2
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -35,11 +35,12 @@ def test_regional_dataset_balances_contexts_and_keeps_validation_closed() -> Non
|
||||
assert evidence["protected_samples_in_training"] == []
|
||||
|
||||
|
||||
def test_regional_dataset_rejects_protected_tiles() -> None:
|
||||
@pytest.mark.parametrize("protected_split", ["calibration", "test", "background-test", "challenge"])
|
||||
def test_regional_dataset_rejects_protected_tiles(protected_split: str) -> None:
|
||||
manifest = {"samples": [
|
||||
{"sample_slug": "f-cal", "region": "flanders", "split": "calibration", "context": "ribbon"},
|
||||
{"sample_slug": "f-cal", "region": "flanders", "split": protected_split, "context": "ribbon"},
|
||||
]}
|
||||
summary = {"tiles": [{"sample_slug": "f-cal", "split": "calibration", "image_path": "/cal.png"}]}
|
||||
summary = {"tiles": [{"sample_slug": "f-cal", "split": protected_split, "image_path": "/cal.png"}]}
|
||||
with pytest.raises(ValueError, match="protected"):
|
||||
module.build(
|
||||
summary=summary, manifest=manifest, region="flanders",
|
||||
|
||||
@@ -114,8 +114,9 @@ def test_classify_proposals_consumes_reference_once() -> None:
|
||||
assert [item[0] for item in miner.classify_proposals(proposals, reference, 0.25)] == ["positive", "negative"]
|
||||
|
||||
|
||||
def test_eligible_tiles_rejects_protected_manifest_split() -> None:
|
||||
manifest = {"samples": [{"sample_slug": "x", "region": "flanders", "split": "test"}]}
|
||||
@pytest.mark.parametrize("protected_split", ["calibration", "test", "background-test", "challenge"])
|
||||
def test_eligible_tiles_rejects_protected_manifest_split(protected_split: str) -> None:
|
||||
manifest = {"samples": [{"sample_slug": "x", "region": "flanders", "split": protected_split}]}
|
||||
summary = {"tiles": [{"sample_slug": "x", "split": "train", "image_path": "x.png"}]}
|
||||
with pytest.raises(ValueError, match="protected"):
|
||||
miner.eligible_tiles(summary, manifest, "flanders")
|
||||
|
||||
Reference in New Issue
Block a user