docs(accuracy): publish phase 4 benchmark evidence
GeoIntel release gates / Compile, test, contracts and builds (push) Canceled after 0s
GeoIntel release gates / Python and npm vulnerability policy (push) Canceled after 0s
GeoIntel release gates / GIS image, SBOM and container scan (push) Canceled after 0s

This commit is contained in:
Jens
2026-08-02 05:10:05 +02:00
parent 70fb4b94e5
commit f41392a415
34 changed files with 25672 additions and 8 deletions
+91 -6
View File
@@ -2,7 +2,7 @@
"schema_version": 1,
"program": "GeoIntel Accuracy Improvement Program",
"phase": "P4",
"generated_at": "2026-08-02T01:04:48+02:00",
"generated_at": "2026-08-02T04:40:51+02:00",
"scope": {
"product": "Belgium and the Belgian North Sea",
"active_building_model_claim": "Mol/Kempen only, operator review required",
@@ -104,9 +104,56 @@
]
},
"phase4": {
"status": "ready",
"meaning": "Ready to remediate and review the P3 anomaly and quarantine manifests; release gates remain governed by P2 and the execution contract.",
"evidence_root": "artifacts/evidence/accuracy/P3"
"status": "in_progress",
"meaning": "The content-addressed local evaluation harness passes, but the governed product benchmark fails on confirmed Phase-3 leakage and remains not evaluable for missing product evidence.",
"workflow": "scripts/run_accuracy_phase4_benchmark.py",
"workflow_version": "2.0.1",
"evaluator_version": "2.1.0",
"split_generator_version": "1.3.0",
"local_harness_status": "pass",
"product_benchmark_status": "fail",
"phase4_done": false,
"evidence_run_id": "p4-2.0.1-9677d0ef37db82bcf39b",
"evidence_root": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b",
"repository_commit": "70fb4b94e5cb7c248beec5a936ce186f38cc183c",
"benchmark_manifest_sha256": "0ea5ab07f46c509a7a24943b31d5e9bfd6368920609227bc47613e18e52c4642",
"benchmark_file_sha256": "868dd7eb2dfa8344e417879ed3f5dd7c7d9e5673add7e3e2294c5cbde4b75d57",
"evidence_manifest_file_sha256": "fdc15a95ee2a0754dfa169f4b41036e084b8d8909afc37fd8ea68ee6b9210f98",
"release_gate_report_file_sha256": "e5f1c9c43c8da22acfa5486e641c010a278431f90634f91335b039ab7bdb50a4",
"product_gate_evidence_sha256": "c9f5e3de3ab6d64911f8807b26caa661ccf911a5fd6b995107edda2004836e0c",
"case_count": 9,
"task_family_count": 7,
"implemented_capability_count": 15,
"failure_example_count": 14,
"split_counts": {
"background-test": 2,
"calibration": 2,
"challenge": 4,
"test": 7,
"train": 3,
"val": 3
},
"blockers": [
"configured active model bytes are not locally accessible",
"no governed seven-task product baseline manifest with current CUDA receipt exists",
"Phase-3 leakage status is attention and therefore a failing product gate",
"no checksum-bound representative human review ledger exists",
"no governed zero-under-2-km product split audit exists",
"no protected vault evidence with hash-chained access log exists",
"no complete task-zone authority portfolio exists",
"no representative support over all thirteen subgroup dimensions exists"
],
"does_not_mean": [
"Phase 4 is complete",
"production accuracy is measured",
"Phase 5 is ready",
"training or promotion is allowed"
]
},
"phase5": {
"status": "not_ready",
"meaning": "Phase 5 remains blocked until every governed Phase-4 product gate passes in the same checksum-bound workflow.",
"blocked_by": "phase4"
},
"runtime": {
"cuda_available": true,
@@ -276,6 +323,35 @@
"static_inventory": "artifacts/evidence/accuracy/P2/source-contract-inventory.json",
"claim_boundary": "This is not a production migration deployment, corpus-release, accuracy or promotion result."
},
"phase4_evaluation": {
"status": "local_pass_product_fail",
"targeted_tests": {
"passed": 60,
"duration_seconds": 22.28
},
"relevant_regression_tests": {
"passed": 103,
"duration_seconds": 27.61,
"shell_provider": "C:/Program Files/Git/bin/bash.exe",
"note": "The Windows Store WSL bash stub was unavailable; the same shell syntax test passed with the installed Git Bash provider."
},
"ruff": "pass",
"format_check": "pass",
"compileall": "pass",
"python_typecheck": "not_configured",
"alembic_head": "202608010001",
"migration_and_provenance_tests": {
"passed": 29,
"duration_seconds": 5.73
},
"workflow_reproduction": {
"allow_product_blocked_exit": 0,
"default_exit": 2,
"byte_identical": true,
"status_bookkeeping_stable": true
},
"claim_boundary": "Local harness and contract gates pass. Product accuracy is not measured; Phase-3 leakage fails and missing governed product evidence remains not evaluable."
},
"golden_qa": {
"semantic_results_stable": true,
"byte_identical": false,
@@ -311,12 +387,21 @@
"docs/accuracy-program/06-implementation-roadmap.md",
"docs/accuracy-program/07-source-authority-matrix.md",
"docs/accuracy-program/08-data-contracts.md",
"docs/accuracy-program/09-full-data-scan.md"
"docs/accuracy-program/09-full-data-scan.md",
"docs/accuracy-program/10-evaluation-protocol.md",
"docs/accuracy-program/11-baseline-benchmark.md",
"docs/accuracy-program/12-release-gates.md"
],
"evidence_root": "artifacts/evidence/accuracy/P1",
"evidence_manifest": "artifacts/evidence/accuracy/P1/evidence-manifest.json",
"phase2_evidence_root": "artifacts/evidence/accuracy/P2",
"phase2_evidence_manifest": "artifacts/evidence/accuracy/P2/evidence-manifest.json",
"phase3_evidence_root": "artifacts/evidence/accuracy/P3",
"phase3_manifest": "artifacts/evidence/accuracy/P3/full-scan-manifest.json"
"phase3_manifest": "artifacts/evidence/accuracy/P3/full-scan-manifest.json",
"phase4_evidence_root": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b",
"phase4_evidence_manifest": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b/evidence-manifest.json",
"phase4_benchmark_manifest": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b/benchmark-manifest.json",
"phase4_release_gate_report": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b/release-gate-report.json",
"phase4_metric_report": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b/metric-report.json",
"phase4_failure_gallery": "artifacts/evidence/accuracy/P4/runs/p4-2.0.1-9677d0ef37db82bcf39b/failure-gallery.json"
}