Files
geointel/docs/accuracy-program/status.json
T
Jens f41392a415
GeoIntel release gates / Compile, test, contracts and builds (push) Canceled after 0s
GeoIntel release gates / Python and npm vulnerability policy (push) Canceled after 0s
GeoIntel release gates / GIS image, SBOM and container scan (push) Canceled after 0s
docs(accuracy): publish phase 4 benchmark evidence
2026-08-02 05:10:05 +02:00

408 lines
16 KiB
JSON

{
"schema_version": 1,
"program": "GeoIntel Accuracy Improvement Program",
"phase": "P4",
"generated_at": "2026-08-02T04:40:51+02:00",
"scope": {
"product": "Belgium and the Belgian North Sea",
"active_building_model_claim": "Mol/Kempen only, operator review required",
"national_building_validation": false
},
"baseline": {
"branch": "codex/geointel-accuracy-program",
"repository_commit": "0c019bb22f816db1e4b7a68379bcad08924d9a21",
"database_migration_head": "202607260001",
"phase2_migration_revision": "202608010001",
"phase2_migration_live_disposable_verified": true,
"phase2_migration_production_deployed": false,
"phase1_mutation_scope": "audit tooling, tests, documentation and retained evidence only"
},
"phase1": {
"status": "complete",
"meaning": "The forensic inventory, reproducible baseline, lineage assessment, risk register, metric contract, implementation roadmap and retained evidence exist.",
"does_not_mean": [
"release ready",
"nationally validated",
"human-reviewed corpus",
"strictly independent protected test",
"calibrated confidence",
"fully trained"
]
},
"release": {
"status": "blocked",
"promotion_allowed": false,
"scope_widening_allowed": false,
"training_allowed_now": false,
"training_unlock_gate": "P2-08 after P2-00 through P2-07 have passed",
"critical_risk_count": 8,
"high_risk_count": 17,
"medium_risk_count": 4
},
"phase2": {
"status": "in_progress",
"meaning": "The source/provenance foundation is implemented and verified in a disposable PostGIS environment; the full P2 roadmap and its training/release gates are not complete.",
"roadmap": "docs/accuracy-program/06-implementation-roadmap.md",
"first_work_package": "P2-00",
"required_order": [
"P2-00",
"P2-01",
"P2-02",
"P2-03",
"P2-04",
"P2-05",
"P2-06",
"P2-07",
"P2-08",
"P2-09",
"P2-10",
"P2-11",
"P2-12"
],
"protected_test_rule": "Open exactly once for a pre-registered immutable candidate after all pre-test gates pass; never feed its results back into that candidate family."
},
"phase2_source_provenance": {
"status": "implemented_and_verified_in_disposable_environment",
"evidence_root": "artifacts/evidence/accuracy/P2",
"implemented": [
"server-owned source registry and immutable source snapshots",
"versioned vector, raster, label and PyTorch model contracts",
"checksum-bound dataset and dataset-version provenance",
"lineage graph, quarantine propagation and consumption gates",
"live disposable PostGIS upgrade, guard and downgrade verification"
],
"does_not_mean": [
"all legacy records are provenance complete",
"physical quarantine or protected-test storage isolation is proven",
"every UI result has end-to-end provenance rendering evidence",
"the full P2 roadmap is complete",
"training, promotion or national validation is allowed"
]
},
"phase3": {
"status": "done",
"meaning": "All 295 safe local files in the configured GeoIntel roots were scanned read-only; three known external boundaries were explicitly recorded as unreachable.",
"scanner": "scripts/run_accuracy_phase3_full_data_scan.py",
"scanner_version": "3.0.3",
"scan_id": "p3-46ee3f8d3a2dc52b",
"evidence_root": "artifacts/evidence/accuracy/P3",
"content_hash": "1a219362c6cb2ac00489625f1a9b36e2fd4809ab58ee55ab1f67c3cc34773f3e",
"reconciliation": {
"examined": 295,
"skipped": 0,
"unreachable": 3,
"inventory_total": 298,
"reconciles": true
},
"anomaly_count": 172,
"quarantine_item_count": 163,
"does_not_mean": [
"all anomalies are repaired",
"AOI split independence is proven",
"GRB ground truth is locally available",
"training, promotion or national validation is allowed"
]
},
"phase4": {
"status": "in_progress",
"meaning": "The content-addressed local evaluation harness passes, but the governed product benchmark fails on confirmed Phase-3 leakage and remains not evaluable for missing product evidence.",
"workflow": "scripts/run_accuracy_phase4_benchmark.py",
"workflow_version": "2.0.1",
"evaluator_version": "2.1.0",
"split_generator_version": "1.3.0",
"local_harness_status": "pass",
"product_benchmark_status": "fail",
"phase4_done": false,
"evidence_run_id": "p4-2.0.1-9677d0ef37db82bcf39b",
"evidence_root": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b",
"repository_commit": "70fb4b94e5cb7c248beec5a936ce186f38cc183c",
"benchmark_manifest_sha256": "0ea5ab07f46c509a7a24943b31d5e9bfd6368920609227bc47613e18e52c4642",
"benchmark_file_sha256": "868dd7eb2dfa8344e417879ed3f5dd7c7d9e5673add7e3e2294c5cbde4b75d57",
"evidence_manifest_file_sha256": "fdc15a95ee2a0754dfa169f4b41036e084b8d8909afc37fd8ea68ee6b9210f98",
"release_gate_report_file_sha256": "e5f1c9c43c8da22acfa5486e641c010a278431f90634f91335b039ab7bdb50a4",
"product_gate_evidence_sha256": "c9f5e3de3ab6d64911f8807b26caa661ccf911a5fd6b995107edda2004836e0c",
"case_count": 9,
"task_family_count": 7,
"implemented_capability_count": 15,
"failure_example_count": 14,
"split_counts": {
"background-test": 2,
"calibration": 2,
"challenge": 4,
"test": 7,
"train": 3,
"val": 3
},
"blockers": [
"configured active model bytes are not locally accessible",
"no governed seven-task product baseline manifest with current CUDA receipt exists",
"Phase-3 leakage status is attention and therefore a failing product gate",
"no checksum-bound representative human review ledger exists",
"no governed zero-under-2-km product split audit exists",
"no protected vault evidence with hash-chained access log exists",
"no complete task-zone authority portfolio exists",
"no representative support over all thirteen subgroup dimensions exists"
],
"does_not_mean": [
"Phase 4 is complete",
"production accuracy is measured",
"Phase 5 is ready",
"training or promotion is allowed"
]
},
"phase5": {
"status": "not_ready",
"meaning": "Phase 5 remains blocked until every governed Phase-4 product gate passes in the same checksum-bound workflow.",
"blocked_by": "phase4"
},
"runtime": {
"cuda_available": true,
"device": "NVIDIA GeForce RTX 4080 SUPER",
"configured_device": "cuda:0",
"python": "3.11.2",
"torch": "2.11.0+cu128",
"cuda_runtime": "12.8",
"ultralytics": "8.4.99",
"active_model": {
"model_id": "yolo-configured",
"model_version": "",
"path": "/app/models/geointel-building-yolov8s-smallbld-minpx3-img640-ft30.pt",
"sha256": "a9088b8491dfae36694b53e9e9406cb4e3511d334a5712fa34f75078a47759c1",
"size_bytes": 22516074,
"validated_area_names": [
"Mol",
"Kempen"
],
"nationally_validated": false
},
"gpu_smoke": {
"status": "passed",
"read_only": true,
"tile_crs": "EPSG:31370",
"tile_sha256": "134a9e86850c92c577c73bc6ee57a9df7d4c1c513ae6450263e800b6dd47b6ee",
"manifest_sha256": "6ab8a96bf2a1405e224932afb90255311a09bbeaed4e9a2fdcdf8b1bc2230abd",
"raw_detection_count": 17,
"inference_seconds": 0.8836944859940559,
"claim_boundary": "Runtime execution only; no accuracy, calibration, generalization or release claim."
}
},
"database": {
"mode": "read_only_audit",
"storage_references_checked": 5816,
"storage_references_missing": 0,
"table_counts": {
"projects": 1097,
"areas": 387,
"datasets": 3377,
"dataset_versions": 1671,
"vector_features": 6689447,
"analysis_runs": 1146,
"detections": 299233,
"segmentations": 0,
"detection_reviews": 0,
"quality_checks": 697,
"metrics": 4182,
"jobs": 4460,
"exports": 768,
"aoi_operations": 4,
"aoi_operation_partitions": 61
},
"lineage_gaps": {
"datasets_missing_observed_at": 2377,
"datasets_missing_source_version": 1761,
"detection_runs_with_empty_model_version": 1146,
"detections_with_empty_model_version": 299233,
"detection_runs_missing_model_hash": 3,
"detection_runs_missing_tile_manifest_hash": 1146
},
"geometry": {
"areas_invalid_or_wrong_srid": 0,
"vector_features_invalid_or_wrong_srid": 0,
"detections_outside_epsg4326_domain": 4,
"outside_domain_interpretation": "Four Geel detections contain Lambert-domain coordinates while persisted under SRID 4326."
}
},
"ml_data": {
"model_asset_count": 26,
"training_checkpoint_count": 229,
"training_json_report_count": 424,
"operator_manifest_count": 36,
"v56": {
"sample_count": 180,
"reviewed_sample_count": 0,
"review_complete": false,
"input_feature_count": 31452,
"accepted_feature_count": 30662,
"below_resolvable_pixel_size": 326,
"created_after_imagery_period": 464,
"pure_empty_background_count": 3,
"pure_empty_background_by_region": {
"flanders": 2,
"wallonia": 1,
"brussels": 0
},
"minimum_cross_split_aoi_distance_m": 95.72033647650719,
"cross_split_pairs_below_2000_m": 24,
"exact_cross_split_raster_hash_duplicates": 0,
"perceptual_pairs_hamming_at_or_below_4": 0,
"split_independence_proven": false
},
"candidate_evidence": {
"v58_v62_kind": "calibration-only tile-level bbox metrics",
"threshold_0_15_aggregate_f1": 0.512905360688286,
"threshold_0_15_flanders_recall": 0.0,
"protected_test_evidence": false,
"background_test_release_evidence": false,
"promotion_evidence": false
},
"protected_test_isolation": false,
"human_label_acceptance": false
},
"verification": {
"backend_full_suite": {
"status": "failed",
"passed": 1282,
"failed": 17,
"duration_seconds": 116.62,
"classification": "The canonical backend test import boundary now collects. Sixteen remaining failures are historical source-text assertions for changed UI/deployment/README behavior; one full Windows run also hit an intermittent WSL-backed bash.exe host failure in a shell-wrapper syntax test. This still prevents a whole-suite-green claim."
},
"backend_ci_entrypoint": {
"status": "collected_with_failures",
"collected": 1299,
"result": "1282 passed, 17 failed",
"note": "The former scripts.render_operator_polygon_label_qa collection failure is resolved by the canonical backend-test import boundary."
},
"phase1_tooling_tests": {
"status": "passed",
"passed": 4
},
"new_phase1_code_ruff": {
"status": "passed"
},
"repository_ruff": {
"status": "failed",
"finding_count": 95,
"note": "P2-changed Python paths pass their scoped Ruff check; repository-wide remediation remains an explicit P2-01 gate."
},
"frontend_unit": {
"status": "passed",
"test_files": 16,
"tests": 51,
"command": "npm run test:unit"
},
"frontend_typecheck": {
"status": "passed"
},
"frontend_build": {
"status": "passed"
},
"frontend_lint": {
"status": "missing",
"error": "npm run lint: Missing script"
},
"openapi_contract": {
"status": "passed",
"implemented_routes": 147,
"explicit_non_envelope_endpoints": 10
},
"alembic": {
"status": "passed_offline_and_disposable_postgis",
"heads": [
"202608010001"
],
"offline_upgrade_rendered": true,
"offline_downgrade_rendered": true,
"live_migration_tested_locally": true,
"production_migration_deployed": false
},
"phase2_source_provenance": {
"status": "passed_in_disposable_environment",
"source_registry_definitions": 40,
"migration_revision": "202608010001",
"migration_guards": "artifacts/evidence/accuracy/P2/postgres-migration-guards.json",
"static_inventory": "artifacts/evidence/accuracy/P2/source-contract-inventory.json",
"claim_boundary": "This is not a production migration deployment, corpus-release, accuracy or promotion result."
},
"phase4_evaluation": {
"status": "local_pass_product_fail",
"targeted_tests": {
"passed": 60,
"duration_seconds": 22.28
},
"relevant_regression_tests": {
"passed": 103,
"duration_seconds": 27.61,
"shell_provider": "C:/Program Files/Git/bin/bash.exe",
"note": "The Windows Store WSL bash stub was unavailable; the same shell syntax test passed with the installed Git Bash provider."
},
"ruff": "pass",
"format_check": "pass",
"compileall": "pass",
"python_typecheck": "not_configured",
"alembic_head": "202608010001",
"migration_and_provenance_tests": {
"passed": 29,
"duration_seconds": 5.73
},
"workflow_reproduction": {
"allow_product_blocked_exit": 0,
"default_exit": 2,
"byte_identical": true,
"status_bookkeeping_stable": true
},
"claim_boundary": "Local harness and contract gates pass. Product accuracy is not measured; Phase-3 leakage fails and missing governed product evidence remains not evaluable."
},
"golden_qa": {
"semantic_results_stable": true,
"byte_identical": false,
"reason": "UUID4-backed run identity"
}
},
"reproduced_contract_violations": [
"P1-COV-001",
"P1-CRS-001",
"P1-CRS-002",
"P1-AUTH-001",
"P1-AI-001",
"P1-COV-002",
"P1-API-001"
],
"critical_blockers": [
"ACC-R01",
"ACC-R02",
"ACC-R03",
"ACC-R04",
"ACC-R06",
"ACC-R09",
"ACC-R16",
"ACC-R17"
],
"documents": [
"docs/accuracy-program/00-execution-contract.md",
"docs/accuracy-program/01-system-inventory.md",
"docs/accuracy-program/02-data-lineage.md",
"docs/accuracy-program/03-baseline-and-gaps.md",
"docs/accuracy-program/04-risk-register.md",
"docs/accuracy-program/05-metric-framework.md",
"docs/accuracy-program/06-implementation-roadmap.md",
"docs/accuracy-program/07-source-authority-matrix.md",
"docs/accuracy-program/08-data-contracts.md",
"docs/accuracy-program/09-full-data-scan.md",
"docs/accuracy-program/10-evaluation-protocol.md",
"docs/accuracy-program/11-baseline-benchmark.md",
"docs/accuracy-program/12-release-gates.md"
],
"evidence_root": "artifacts/evidence/accuracy/P1",
"evidence_manifest": "artifacts/evidence/accuracy/P1/evidence-manifest.json",
"phase2_evidence_root": "artifacts/evidence/accuracy/P2",
"phase2_evidence_manifest": "artifacts/evidence/accuracy/P2/evidence-manifest.json",
"phase3_evidence_root": "artifacts/evidence/accuracy/P3",
"phase3_manifest": "artifacts/evidence/accuracy/P3/full-scan-manifest.json",
"phase4_evidence_root": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b",
"phase4_evidence_manifest": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b/evidence-manifest.json",
"phase4_benchmark_manifest": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b/benchmark-manifest.json",
"phase4_release_gate_report": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b/release-gate-report.json",
"phase4_metric_report": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b/metric-report.json",
"phase4_failure_gallery": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b/failure-gallery.json"
}