408 lines
16 KiB
JSON
408 lines
16 KiB
JSON
{
|
|
"schema_version": 1,
|
|
"program": "GeoIntel Accuracy Improvement Program",
|
|
"phase": "P4",
|
|
"generated_at": "2026-08-02T04:40:51+02:00",
|
|
"scope": {
|
|
"product": "Belgium and the Belgian North Sea",
|
|
"active_building_model_claim": "Mol/Kempen only, operator review required",
|
|
"national_building_validation": false
|
|
},
|
|
"baseline": {
|
|
"branch": "codex/geointel-accuracy-program",
|
|
"repository_commit": "0c019bb22f816db1e4b7a68379bcad08924d9a21",
|
|
"database_migration_head": "202607260001",
|
|
"phase2_migration_revision": "202608010001",
|
|
"phase2_migration_live_disposable_verified": true,
|
|
"phase2_migration_production_deployed": false,
|
|
"phase1_mutation_scope": "audit tooling, tests, documentation and retained evidence only"
|
|
},
|
|
"phase1": {
|
|
"status": "complete",
|
|
"meaning": "The forensic inventory, reproducible baseline, lineage assessment, risk register, metric contract, implementation roadmap and retained evidence exist.",
|
|
"does_not_mean": [
|
|
"release ready",
|
|
"nationally validated",
|
|
"human-reviewed corpus",
|
|
"strictly independent protected test",
|
|
"calibrated confidence",
|
|
"fully trained"
|
|
]
|
|
},
|
|
"release": {
|
|
"status": "blocked",
|
|
"promotion_allowed": false,
|
|
"scope_widening_allowed": false,
|
|
"training_allowed_now": false,
|
|
"training_unlock_gate": "P2-08 after P2-00 through P2-07 have passed",
|
|
"critical_risk_count": 8,
|
|
"high_risk_count": 17,
|
|
"medium_risk_count": 4
|
|
},
|
|
"phase2": {
|
|
"status": "in_progress",
|
|
"meaning": "The source/provenance foundation is implemented and verified in a disposable PostGIS environment; the full P2 roadmap and its training/release gates are not complete.",
|
|
"roadmap": "docs/accuracy-program/06-implementation-roadmap.md",
|
|
"first_work_package": "P2-00",
|
|
"required_order": [
|
|
"P2-00",
|
|
"P2-01",
|
|
"P2-02",
|
|
"P2-03",
|
|
"P2-04",
|
|
"P2-05",
|
|
"P2-06",
|
|
"P2-07",
|
|
"P2-08",
|
|
"P2-09",
|
|
"P2-10",
|
|
"P2-11",
|
|
"P2-12"
|
|
],
|
|
"protected_test_rule": "Open exactly once for a pre-registered immutable candidate after all pre-test gates pass; never feed its results back into that candidate family."
|
|
},
|
|
"phase2_source_provenance": {
|
|
"status": "implemented_and_verified_in_disposable_environment",
|
|
"evidence_root": "artifacts/evidence/accuracy/P2",
|
|
"implemented": [
|
|
"server-owned source registry and immutable source snapshots",
|
|
"versioned vector, raster, label and PyTorch model contracts",
|
|
"checksum-bound dataset and dataset-version provenance",
|
|
"lineage graph, quarantine propagation and consumption gates",
|
|
"live disposable PostGIS upgrade, guard and downgrade verification"
|
|
],
|
|
"does_not_mean": [
|
|
"all legacy records are provenance complete",
|
|
"physical quarantine or protected-test storage isolation is proven",
|
|
"every UI result has end-to-end provenance rendering evidence",
|
|
"the full P2 roadmap is complete",
|
|
"training, promotion or national validation is allowed"
|
|
]
|
|
},
|
|
"phase3": {
|
|
"status": "done",
|
|
"meaning": "All 295 safe local files in the configured GeoIntel roots were scanned read-only; three known external boundaries were explicitly recorded as unreachable.",
|
|
"scanner": "scripts/run_accuracy_phase3_full_data_scan.py",
|
|
"scanner_version": "3.0.3",
|
|
"scan_id": "p3-46ee3f8d3a2dc52b",
|
|
"evidence_root": "artifacts/evidence/accuracy/P3",
|
|
"content_hash": "1a219362c6cb2ac00489625f1a9b36e2fd4809ab58ee55ab1f67c3cc34773f3e",
|
|
"reconciliation": {
|
|
"examined": 295,
|
|
"skipped": 0,
|
|
"unreachable": 3,
|
|
"inventory_total": 298,
|
|
"reconciles": true
|
|
},
|
|
"anomaly_count": 172,
|
|
"quarantine_item_count": 163,
|
|
"does_not_mean": [
|
|
"all anomalies are repaired",
|
|
"AOI split independence is proven",
|
|
"GRB ground truth is locally available",
|
|
"training, promotion or national validation is allowed"
|
|
]
|
|
},
|
|
"phase4": {
|
|
"status": "in_progress",
|
|
"meaning": "The content-addressed local evaluation harness passes, but the governed product benchmark fails on confirmed Phase-3 leakage and remains not evaluable for missing product evidence.",
|
|
"workflow": "scripts/run_accuracy_phase4_benchmark.py",
|
|
"workflow_version": "2.0.1",
|
|
"evaluator_version": "2.1.0",
|
|
"split_generator_version": "1.3.0",
|
|
"local_harness_status": "pass",
|
|
"product_benchmark_status": "fail",
|
|
"phase4_done": false,
|
|
"evidence_run_id": "p4-2.0.1-9677d0ef37db82bcf39b",
|
|
"evidence_root": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b",
|
|
"repository_commit": "70fb4b94e5cb7c248beec5a936ce186f38cc183c",
|
|
"benchmark_manifest_sha256": "0ea5ab07f46c509a7a24943b31d5e9bfd6368920609227bc47613e18e52c4642",
|
|
"benchmark_file_sha256": "868dd7eb2dfa8344e417879ed3f5dd7c7d9e5673add7e3e2294c5cbde4b75d57",
|
|
"evidence_manifest_file_sha256": "fdc15a95ee2a0754dfa169f4b41036e084b8d8909afc37fd8ea68ee6b9210f98",
|
|
"release_gate_report_file_sha256": "e5f1c9c43c8da22acfa5486e641c010a278431f90634f91335b039ab7bdb50a4",
|
|
"product_gate_evidence_sha256": "c9f5e3de3ab6d64911f8807b26caa661ccf911a5fd6b995107edda2004836e0c",
|
|
"case_count": 9,
|
|
"task_family_count": 7,
|
|
"implemented_capability_count": 15,
|
|
"failure_example_count": 14,
|
|
"split_counts": {
|
|
"background-test": 2,
|
|
"calibration": 2,
|
|
"challenge": 4,
|
|
"test": 7,
|
|
"train": 3,
|
|
"val": 3
|
|
},
|
|
"blockers": [
|
|
"configured active model bytes are not locally accessible",
|
|
"no governed seven-task product baseline manifest with current CUDA receipt exists",
|
|
"Phase-3 leakage status is attention and therefore a failing product gate",
|
|
"no checksum-bound representative human review ledger exists",
|
|
"no governed zero-under-2-km product split audit exists",
|
|
"no protected vault evidence with hash-chained access log exists",
|
|
"no complete task-zone authority portfolio exists",
|
|
"no representative support over all thirteen subgroup dimensions exists"
|
|
],
|
|
"does_not_mean": [
|
|
"Phase 4 is complete",
|
|
"production accuracy is measured",
|
|
"Phase 5 is ready",
|
|
"training or promotion is allowed"
|
|
]
|
|
},
|
|
"phase5": {
|
|
"status": "not_ready",
|
|
"meaning": "Phase 5 remains blocked until every governed Phase-4 product gate passes in the same checksum-bound workflow.",
|
|
"blocked_by": "phase4"
|
|
},
|
|
"runtime": {
|
|
"cuda_available": true,
|
|
"device": "NVIDIA GeForce RTX 4080 SUPER",
|
|
"configured_device": "cuda:0",
|
|
"python": "3.11.2",
|
|
"torch": "2.11.0+cu128",
|
|
"cuda_runtime": "12.8",
|
|
"ultralytics": "8.4.99",
|
|
"active_model": {
|
|
"model_id": "yolo-configured",
|
|
"model_version": "",
|
|
"path": "/app/models/geointel-building-yolov8s-smallbld-minpx3-img640-ft30.pt",
|
|
"sha256": "a9088b8491dfae36694b53e9e9406cb4e3511d334a5712fa34f75078a47759c1",
|
|
"size_bytes": 22516074,
|
|
"validated_area_names": [
|
|
"Mol",
|
|
"Kempen"
|
|
],
|
|
"nationally_validated": false
|
|
},
|
|
"gpu_smoke": {
|
|
"status": "passed",
|
|
"read_only": true,
|
|
"tile_crs": "EPSG:31370",
|
|
"tile_sha256": "134a9e86850c92c577c73bc6ee57a9df7d4c1c513ae6450263e800b6dd47b6ee",
|
|
"manifest_sha256": "6ab8a96bf2a1405e224932afb90255311a09bbeaed4e9a2fdcdf8b1bc2230abd",
|
|
"raw_detection_count": 17,
|
|
"inference_seconds": 0.8836944859940559,
|
|
"claim_boundary": "Runtime execution only; no accuracy, calibration, generalization or release claim."
|
|
}
|
|
},
|
|
"database": {
|
|
"mode": "read_only_audit",
|
|
"storage_references_checked": 5816,
|
|
"storage_references_missing": 0,
|
|
"table_counts": {
|
|
"projects": 1097,
|
|
"areas": 387,
|
|
"datasets": 3377,
|
|
"dataset_versions": 1671,
|
|
"vector_features": 6689447,
|
|
"analysis_runs": 1146,
|
|
"detections": 299233,
|
|
"segmentations": 0,
|
|
"detection_reviews": 0,
|
|
"quality_checks": 697,
|
|
"metrics": 4182,
|
|
"jobs": 4460,
|
|
"exports": 768,
|
|
"aoi_operations": 4,
|
|
"aoi_operation_partitions": 61
|
|
},
|
|
"lineage_gaps": {
|
|
"datasets_missing_observed_at": 2377,
|
|
"datasets_missing_source_version": 1761,
|
|
"detection_runs_with_empty_model_version": 1146,
|
|
"detections_with_empty_model_version": 299233,
|
|
"detection_runs_missing_model_hash": 3,
|
|
"detection_runs_missing_tile_manifest_hash": 1146
|
|
},
|
|
"geometry": {
|
|
"areas_invalid_or_wrong_srid": 0,
|
|
"vector_features_invalid_or_wrong_srid": 0,
|
|
"detections_outside_epsg4326_domain": 4,
|
|
"outside_domain_interpretation": "Four Geel detections contain Lambert-domain coordinates while persisted under SRID 4326."
|
|
}
|
|
},
|
|
"ml_data": {
|
|
"model_asset_count": 26,
|
|
"training_checkpoint_count": 229,
|
|
"training_json_report_count": 424,
|
|
"operator_manifest_count": 36,
|
|
"v56": {
|
|
"sample_count": 180,
|
|
"reviewed_sample_count": 0,
|
|
"review_complete": false,
|
|
"input_feature_count": 31452,
|
|
"accepted_feature_count": 30662,
|
|
"below_resolvable_pixel_size": 326,
|
|
"created_after_imagery_period": 464,
|
|
"pure_empty_background_count": 3,
|
|
"pure_empty_background_by_region": {
|
|
"flanders": 2,
|
|
"wallonia": 1,
|
|
"brussels": 0
|
|
},
|
|
"minimum_cross_split_aoi_distance_m": 95.72033647650719,
|
|
"cross_split_pairs_below_2000_m": 24,
|
|
"exact_cross_split_raster_hash_duplicates": 0,
|
|
"perceptual_pairs_hamming_at_or_below_4": 0,
|
|
"split_independence_proven": false
|
|
},
|
|
"candidate_evidence": {
|
|
"v58_v62_kind": "calibration-only tile-level bbox metrics",
|
|
"threshold_0_15_aggregate_f1": 0.512905360688286,
|
|
"threshold_0_15_flanders_recall": 0.0,
|
|
"protected_test_evidence": false,
|
|
"background_test_release_evidence": false,
|
|
"promotion_evidence": false
|
|
},
|
|
"protected_test_isolation": false,
|
|
"human_label_acceptance": false
|
|
},
|
|
"verification": {
|
|
"backend_full_suite": {
|
|
"status": "failed",
|
|
"passed": 1282,
|
|
"failed": 17,
|
|
"duration_seconds": 116.62,
|
|
"classification": "The canonical backend test import boundary now collects. Sixteen remaining failures are historical source-text assertions for changed UI/deployment/README behavior; one full Windows run also hit an intermittent WSL-backed bash.exe host failure in a shell-wrapper syntax test. This still prevents a whole-suite-green claim."
|
|
},
|
|
"backend_ci_entrypoint": {
|
|
"status": "collected_with_failures",
|
|
"collected": 1299,
|
|
"result": "1282 passed, 17 failed",
|
|
"note": "The former scripts.render_operator_polygon_label_qa collection failure is resolved by the canonical backend-test import boundary."
|
|
},
|
|
"phase1_tooling_tests": {
|
|
"status": "passed",
|
|
"passed": 4
|
|
},
|
|
"new_phase1_code_ruff": {
|
|
"status": "passed"
|
|
},
|
|
"repository_ruff": {
|
|
"status": "failed",
|
|
"finding_count": 95,
|
|
"note": "P2-changed Python paths pass their scoped Ruff check; repository-wide remediation remains an explicit P2-01 gate."
|
|
},
|
|
"frontend_unit": {
|
|
"status": "passed",
|
|
"test_files": 16,
|
|
"tests": 51,
|
|
"command": "npm run test:unit"
|
|
},
|
|
"frontend_typecheck": {
|
|
"status": "passed"
|
|
},
|
|
"frontend_build": {
|
|
"status": "passed"
|
|
},
|
|
"frontend_lint": {
|
|
"status": "missing",
|
|
"error": "npm run lint: Missing script"
|
|
},
|
|
"openapi_contract": {
|
|
"status": "passed",
|
|
"implemented_routes": 147,
|
|
"explicit_non_envelope_endpoints": 10
|
|
},
|
|
"alembic": {
|
|
"status": "passed_offline_and_disposable_postgis",
|
|
"heads": [
|
|
"202608010001"
|
|
],
|
|
"offline_upgrade_rendered": true,
|
|
"offline_downgrade_rendered": true,
|
|
"live_migration_tested_locally": true,
|
|
"production_migration_deployed": false
|
|
},
|
|
"phase2_source_provenance": {
|
|
"status": "passed_in_disposable_environment",
|
|
"source_registry_definitions": 40,
|
|
"migration_revision": "202608010001",
|
|
"migration_guards": "artifacts/evidence/accuracy/P2/postgres-migration-guards.json",
|
|
"static_inventory": "artifacts/evidence/accuracy/P2/source-contract-inventory.json",
|
|
"claim_boundary": "This is not a production migration deployment, corpus-release, accuracy or promotion result."
|
|
},
|
|
"phase4_evaluation": {
|
|
"status": "local_pass_product_fail",
|
|
"targeted_tests": {
|
|
"passed": 60,
|
|
"duration_seconds": 22.28
|
|
},
|
|
"relevant_regression_tests": {
|
|
"passed": 103,
|
|
"duration_seconds": 27.61,
|
|
"shell_provider": "C:/Program Files/Git/bin/bash.exe",
|
|
"note": "The Windows Store WSL bash stub was unavailable; the same shell syntax test passed with the installed Git Bash provider."
|
|
},
|
|
"ruff": "pass",
|
|
"format_check": "pass",
|
|
"compileall": "pass",
|
|
"python_typecheck": "not_configured",
|
|
"alembic_head": "202608010001",
|
|
"migration_and_provenance_tests": {
|
|
"passed": 29,
|
|
"duration_seconds": 5.73
|
|
},
|
|
"workflow_reproduction": {
|
|
"allow_product_blocked_exit": 0,
|
|
"default_exit": 2,
|
|
"byte_identical": true,
|
|
"status_bookkeeping_stable": true
|
|
},
|
|
"claim_boundary": "Local harness and contract gates pass. Product accuracy is not measured; Phase-3 leakage fails and missing governed product evidence remains not evaluable."
|
|
},
|
|
"golden_qa": {
|
|
"semantic_results_stable": true,
|
|
"byte_identical": false,
|
|
"reason": "UUID4-backed run identity"
|
|
}
|
|
},
|
|
"reproduced_contract_violations": [
|
|
"P1-COV-001",
|
|
"P1-CRS-001",
|
|
"P1-CRS-002",
|
|
"P1-AUTH-001",
|
|
"P1-AI-001",
|
|
"P1-COV-002",
|
|
"P1-API-001"
|
|
],
|
|
"critical_blockers": [
|
|
"ACC-R01",
|
|
"ACC-R02",
|
|
"ACC-R03",
|
|
"ACC-R04",
|
|
"ACC-R06",
|
|
"ACC-R09",
|
|
"ACC-R16",
|
|
"ACC-R17"
|
|
],
|
|
"documents": [
|
|
"docs/accuracy-program/00-execution-contract.md",
|
|
"docs/accuracy-program/01-system-inventory.md",
|
|
"docs/accuracy-program/02-data-lineage.md",
|
|
"docs/accuracy-program/03-baseline-and-gaps.md",
|
|
"docs/accuracy-program/04-risk-register.md",
|
|
"docs/accuracy-program/05-metric-framework.md",
|
|
"docs/accuracy-program/06-implementation-roadmap.md",
|
|
"docs/accuracy-program/07-source-authority-matrix.md",
|
|
"docs/accuracy-program/08-data-contracts.md",
|
|
"docs/accuracy-program/09-full-data-scan.md",
|
|
"docs/accuracy-program/10-evaluation-protocol.md",
|
|
"docs/accuracy-program/11-baseline-benchmark.md",
|
|
"docs/accuracy-program/12-release-gates.md"
|
|
],
|
|
"evidence_root": "artifacts/evidence/accuracy/P1",
|
|
"evidence_manifest": "artifacts/evidence/accuracy/P1/evidence-manifest.json",
|
|
"phase2_evidence_root": "artifacts/evidence/accuracy/P2",
|
|
"phase2_evidence_manifest": "artifacts/evidence/accuracy/P2/evidence-manifest.json",
|
|
"phase3_evidence_root": "artifacts/evidence/accuracy/P3",
|
|
"phase3_manifest": "artifacts/evidence/accuracy/P3/full-scan-manifest.json",
|
|
"phase4_evidence_root": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b",
|
|
"phase4_evidence_manifest": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b/evidence-manifest.json",
|
|
"phase4_benchmark_manifest": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b/benchmark-manifest.json",
|
|
"phase4_release_gate_report": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b/release-gate-report.json",
|
|
"phase4_metric_report": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b/metric-report.json",
|
|
"phase4_failure_gallery": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b/failure-gallery.json"
|
|
}
|