docs(accuracy): refresh governed scan and training evidence
This commit is contained in:
@@ -2,11 +2,12 @@
|
||||
"schema_version": 1,
|
||||
"program": "GeoIntel Accuracy Improvement Program",
|
||||
"phase": "P4",
|
||||
"generated_at": "2026-08-02T04:40:51+02:00",
|
||||
"generated_at": "2026-08-30T05:58:29+02:00",
|
||||
"scope": {
|
||||
"product": "Belgium and the Belgian North Sea",
|
||||
"active_building_model_claim": "Mol/Kempen only, operator review required",
|
||||
"national_building_validation": false
|
||||
"national_building_validation": false,
|
||||
"result_authority_policy": "Task-appropriate official sources remain decisive for product results; AI output is a reviewable proposal unless a governed task-specific release gate proves otherwise."
|
||||
},
|
||||
"baseline": {
|
||||
"branch": "codex/geointel-accuracy-program",
|
||||
@@ -34,7 +35,8 @@
|
||||
"promotion_allowed": false,
|
||||
"scope_widening_allowed": false,
|
||||
"training_allowed_now": false,
|
||||
"training_unlock_gate": "P2-08 after P2-00 through P2-07 have passed",
|
||||
"training_unlock_gate": "A corrected checksum-frozen corpus, independent splits, representative human acceptance and every applicable pre-registered training gate must pass before a new governed run.",
|
||||
"activation_gate": "Activation requires the exact candidate key and model SHA-256 bound to a passing governed Phase-4/5 release-gate report and its benchmark-manifest SHA-256; a legacy diagnostic promotion report alone has no activation authority.",
|
||||
"critical_risk_count": 8,
|
||||
"high_risk_count": 17,
|
||||
"medium_risk_count": 4
|
||||
@@ -80,23 +82,24 @@
|
||||
]
|
||||
},
|
||||
"phase3": {
|
||||
"status": "done",
|
||||
"meaning": "All 295 safe local files in the configured GeoIntel roots were scanned read-only; three known external boundaries were explicitly recorded as unreachable.",
|
||||
"status": "local_complete_production_inventory_pending",
|
||||
"meaning": "All 1013 safe local files in the configured GeoIntel roots were scanned read-only; three declared external production boundaries were recorded explicitly and still require a controlled server-side inventory.",
|
||||
"scanner": "scripts/run_accuracy_phase3_full_data_scan.py",
|
||||
"scanner_version": "3.0.3",
|
||||
"scan_id": "p3-46ee3f8d3a2dc52b",
|
||||
"scanner_version": "3.1.0",
|
||||
"scan_id": "p3-eb67185e61107cc7",
|
||||
"evidence_root": "artifacts/evidence/accuracy/P3",
|
||||
"content_hash": "1a219362c6cb2ac00489625f1a9b36e2fd4809ab58ee55ab1f67c3cc34773f3e",
|
||||
"content_hash": "f4ea193d2bdd96a5b391cf7d9d93685cc58e98742c40e4513bcbc434907e720a",
|
||||
"reconciliation": {
|
||||
"examined": 295,
|
||||
"examined": 1013,
|
||||
"skipped": 0,
|
||||
"unreachable": 3,
|
||||
"inventory_total": 298,
|
||||
"inventory_total": 1016,
|
||||
"reconciles": true
|
||||
},
|
||||
"anomaly_count": 172,
|
||||
"quarantine_item_count": 163,
|
||||
"anomaly_count": 711,
|
||||
"quarantine_item_count": 682,
|
||||
"does_not_mean": [
|
||||
"the total production inventory is complete",
|
||||
"all anomalies are repaired",
|
||||
"AOI split independence is proven",
|
||||
"GRB ground truth is locally available",
|
||||
@@ -256,21 +259,31 @@
|
||||
"promotion_evidence": false
|
||||
},
|
||||
"protected_test_isolation": false,
|
||||
"human_label_acceptance": false
|
||||
"human_label_acceptance": false,
|
||||
"latest_independent_ai_visual_review": {
|
||||
"review_id": "20260830-independent-ai-visual-review",
|
||||
"evidence": "artifacts/evidence/accuracy/model-training/20260830-independent-ai-visual-review.json",
|
||||
"human_reviewer": false,
|
||||
"corpus_accepted": false,
|
||||
"training_gate": "blocked",
|
||||
"promotion_allowed": false,
|
||||
"claim_boundary": "Independent AI visual inspection is triage evidence only; it is not representative human acceptance, protected-test evidence or an accuracy measurement."
|
||||
}
|
||||
},
|
||||
"verification": {
|
||||
"backend_full_suite": {
|
||||
"status": "failed",
|
||||
"passed": 1282,
|
||||
"failed": 17,
|
||||
"duration_seconds": 116.62,
|
||||
"classification": "The canonical backend test import boundary now collects. Sixteen remaining failures are historical source-text assertions for changed UI/deployment/README behavior; one full Windows run also hit an intermittent WSL-backed bash.exe host failure in a shell-wrapper syntax test. This still prevents a whole-suite-green claim."
|
||||
"status": "passed",
|
||||
"passed": 1702,
|
||||
"skipped": 1,
|
||||
"failed": 0,
|
||||
"duration_seconds": 279.63,
|
||||
"classification": "The complete canonical backend suite passes with deprecations promoted to errors. The single skip is the Windows-host symlink fixture; the symlink refusal path remains covered by static and Linux-targeted release checks."
|
||||
},
|
||||
"backend_ci_entrypoint": {
|
||||
"status": "collected_with_failures",
|
||||
"collected": 1299,
|
||||
"result": "1282 passed, 17 failed",
|
||||
"note": "The former scripts.render_operator_polygon_label_qa collection failure is resolved by the canonical backend-test import boundary."
|
||||
"status": "passed",
|
||||
"collected": 1703,
|
||||
"result": "1702 passed, 1 skipped",
|
||||
"note": "The canonical backend test import boundary, deployment contracts, guest analysis flow and governed raster handoff all collect and pass."
|
||||
},
|
||||
"phase1_tooling_tests": {
|
||||
"status": "passed",
|
||||
@@ -280,14 +293,14 @@
|
||||
"status": "passed"
|
||||
},
|
||||
"repository_ruff": {
|
||||
"status": "failed",
|
||||
"finding_count": 95,
|
||||
"note": "P2-changed Python paths pass their scoped Ruff check; repository-wide remediation remains an explicit P2-01 gate."
|
||||
"status": "passed",
|
||||
"finding_count": 0,
|
||||
"note": "Repository-wide Ruff validation passes across backend, scripts and tests without policy weakening."
|
||||
},
|
||||
"frontend_unit": {
|
||||
"status": "passed",
|
||||
"test_files": 16,
|
||||
"tests": 51,
|
||||
"test_files": 40,
|
||||
"tests": 170,
|
||||
"command": "npm run test:unit"
|
||||
},
|
||||
"frontend_typecheck": {
|
||||
@@ -302,13 +315,13 @@
|
||||
},
|
||||
"openapi_contract": {
|
||||
"status": "passed",
|
||||
"implemented_routes": 147,
|
||||
"explicit_non_envelope_endpoints": 10
|
||||
"implemented_routes": 155,
|
||||
"explicit_non_envelope_endpoints": 12
|
||||
},
|
||||
"alembic": {
|
||||
"status": "passed_offline_and_disposable_postgis",
|
||||
"heads": [
|
||||
"202608010001"
|
||||
"202608230001"
|
||||
],
|
||||
"offline_upgrade_rendered": true,
|
||||
"offline_downgrade_rendered": true,
|
||||
@@ -390,10 +403,12 @@
|
||||
"docs/accuracy-program/09-full-data-scan.md",
|
||||
"docs/accuracy-program/10-evaluation-protocol.md",
|
||||
"docs/accuracy-program/11-baseline-benchmark.md",
|
||||
"docs/accuracy-program/12-release-gates.md"
|
||||
"docs/accuracy-program/12-release-gates.md",
|
||||
"docs/accuracy-program/13-nested-mirror-retirement.md"
|
||||
],
|
||||
"evidence_root": "artifacts/evidence/accuracy/P1",
|
||||
"evidence_manifest": "artifacts/evidence/accuracy/P1/evidence-manifest.json",
|
||||
"latest_verification_evidence_manifest": "artifacts/evidence/accuracy/P1/evidence-manifest-20260830-verification.json",
|
||||
"phase2_evidence_root": "artifacts/evidence/accuracy/P2",
|
||||
"phase2_evidence_manifest": "artifacts/evidence/accuracy/P2/evidence-manifest.json",
|
||||
"phase3_evidence_root": "artifacts/evidence/accuracy/P3",
|
||||
|
||||
Reference in New Issue
Block a user