docs(accuracy): refresh governed scan and training evidence

This commit is contained in:
Jens
2026-08-30 06:00:57 +02:00
parent c272220277
commit 0e3c1b20e9
27 changed files with 88787 additions and 318 deletions
+46 -31
View File
@@ -2,11 +2,12 @@
"schema_version": 1,
"program": "GeoIntel Accuracy Improvement Program",
"phase": "P4",
"generated_at": "2026-08-02T04:40:51+02:00",
"generated_at": "2026-08-30T05:58:29+02:00",
"scope": {
"product": "Belgium and the Belgian North Sea",
"active_building_model_claim": "Mol/Kempen only, operator review required",
"national_building_validation": false
"national_building_validation": false,
"result_authority_policy": "Task-appropriate official sources remain decisive for product results; AI output is a reviewable proposal unless a governed task-specific release gate proves otherwise."
},
"baseline": {
"branch": "codex/geointel-accuracy-program",
@@ -34,7 +35,8 @@
"promotion_allowed": false,
"scope_widening_allowed": false,
"training_allowed_now": false,
"training_unlock_gate": "P2-08 after P2-00 through P2-07 have passed",
"training_unlock_gate": "A corrected checksum-frozen corpus, independent splits, representative human acceptance and every applicable pre-registered training gate must pass before a new governed run.",
"activation_gate": "Activation requires the exact candidate key and model SHA-256 bound to a passing governed Phase-4/5 release-gate report and its benchmark-manifest SHA-256; a legacy diagnostic promotion report alone has no activation authority.",
"critical_risk_count": 8,
"high_risk_count": 17,
"medium_risk_count": 4
@@ -80,23 +82,24 @@
]
},
"phase3": {
"status": "done",
"meaning": "All 295 safe local files in the configured GeoIntel roots were scanned read-only; three known external boundaries were explicitly recorded as unreachable.",
"status": "local_complete_production_inventory_pending",
"meaning": "All 1013 safe local files in the configured GeoIntel roots were scanned read-only; three declared external production boundaries were recorded explicitly and still require a controlled server-side inventory.",
"scanner": "scripts/run_accuracy_phase3_full_data_scan.py",
"scanner_version": "3.0.3",
"scan_id": "p3-46ee3f8d3a2dc52b",
"scanner_version": "3.1.0",
"scan_id": "p3-eb67185e61107cc7",
"evidence_root": "artifacts/evidence/accuracy/P3",
"content_hash": "1a219362c6cb2ac00489625f1a9b36e2fd4809ab58ee55ab1f67c3cc34773f3e",
"content_hash": "f4ea193d2bdd96a5b391cf7d9d93685cc58e98742c40e4513bcbc434907e720a",
"reconciliation": {
"examined": 295,
"examined": 1013,
"skipped": 0,
"unreachable": 3,
"inventory_total": 298,
"inventory_total": 1016,
"reconciles": true
},
"anomaly_count": 172,
"quarantine_item_count": 163,
"anomaly_count": 711,
"quarantine_item_count": 682,
"does_not_mean": [
"the total production inventory is complete",
"all anomalies are repaired",
"AOI split independence is proven",
"GRB ground truth is locally available",
@@ -256,21 +259,31 @@
"promotion_evidence": false
},
"protected_test_isolation": false,
"human_label_acceptance": false
"human_label_acceptance": false,
"latest_independent_ai_visual_review": {
"review_id": "20260830-independent-ai-visual-review",
"evidence": "artifacts/evidence/accuracy/model-training/20260830-independent-ai-visual-review.json",
"human_reviewer": false,
"corpus_accepted": false,
"training_gate": "blocked",
"promotion_allowed": false,
"claim_boundary": "Independent AI visual inspection is triage evidence only; it is not representative human acceptance, protected-test evidence or an accuracy measurement."
}
},
"verification": {
"backend_full_suite": {
"status": "failed",
"passed": 1282,
"failed": 17,
"duration_seconds": 116.62,
"classification": "The canonical backend test import boundary now collects. Sixteen remaining failures are historical source-text assertions for changed UI/deployment/README behavior; one full Windows run also hit an intermittent WSL-backed bash.exe host failure in a shell-wrapper syntax test. This still prevents a whole-suite-green claim."
"status": "passed",
"passed": 1702,
"skipped": 1,
"failed": 0,
"duration_seconds": 279.63,
"classification": "The complete canonical backend suite passes with deprecations promoted to errors. The single skip is the Windows-host symlink fixture; the symlink refusal path remains covered by static and Linux-targeted release checks."
},
"backend_ci_entrypoint": {
"status": "collected_with_failures",
"collected": 1299,
"result": "1282 passed, 17 failed",
"note": "The former scripts.render_operator_polygon_label_qa collection failure is resolved by the canonical backend-test import boundary."
"status": "passed",
"collected": 1703,
"result": "1702 passed, 1 skipped",
"note": "The canonical backend test import boundary, deployment contracts, guest analysis flow and governed raster handoff all collect and pass."
},
"phase1_tooling_tests": {
"status": "passed",
@@ -280,14 +293,14 @@
"status": "passed"
},
"repository_ruff": {
"status": "failed",
"finding_count": 95,
"note": "P2-changed Python paths pass their scoped Ruff check; repository-wide remediation remains an explicit P2-01 gate."
"status": "passed",
"finding_count": 0,
"note": "Repository-wide Ruff validation passes across backend, scripts and tests without policy weakening."
},
"frontend_unit": {
"status": "passed",
"test_files": 16,
"tests": 51,
"test_files": 40,
"tests": 170,
"command": "npm run test:unit"
},
"frontend_typecheck": {
@@ -302,13 +315,13 @@
},
"openapi_contract": {
"status": "passed",
"implemented_routes": 147,
"explicit_non_envelope_endpoints": 10
"implemented_routes": 155,
"explicit_non_envelope_endpoints": 12
},
"alembic": {
"status": "passed_offline_and_disposable_postgis",
"heads": [
"202608010001"
"202608230001"
],
"offline_upgrade_rendered": true,
"offline_downgrade_rendered": true,
@@ -390,10 +403,12 @@
"docs/accuracy-program/09-full-data-scan.md",
"docs/accuracy-program/10-evaluation-protocol.md",
"docs/accuracy-program/11-baseline-benchmark.md",
"docs/accuracy-program/12-release-gates.md"
"docs/accuracy-program/12-release-gates.md",
"docs/accuracy-program/13-nested-mirror-retirement.md"
],
"evidence_root": "artifacts/evidence/accuracy/P1",
"evidence_manifest": "artifacts/evidence/accuracy/P1/evidence-manifest.json",
"latest_verification_evidence_manifest": "artifacts/evidence/accuracy/P1/evidence-manifest-20260830-verification.json",
"phase2_evidence_root": "artifacts/evidence/accuracy/P2",
"phase2_evidence_manifest": "artifacts/evidence/accuracy/P2/evidence-manifest.json",
"phase3_evidence_root": "artifacts/evidence/accuracy/P3",