997 lines
38 KiB
Python
997 lines
38 KiB
Python
#!/usr/bin/env python3
|
|
"""Run the complete local Phase 4 evaluation workflow from frozen inputs."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import hashlib
|
|
import json
|
|
import platform
|
|
import subprocess
|
|
import sys
|
|
from importlib import metadata as importlib_metadata
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
ROOT = Path(__file__).resolve().parents[1]
|
|
BACKEND_ROOT = ROOT / "backend"
|
|
for entry in (str(ROOT), str(BACKEND_ROOT), str(ROOT / "scripts")):
|
|
if entry not in sys.path:
|
|
sys.path.insert(0, entry)
|
|
|
|
from accuracy_phase4_evaluator import EVALUATOR_VERSION, canonical_hash, evaluate_cases # noqa: E402
|
|
from generate_accuracy_phase4_splits import ( # noqa: E402
|
|
GENERATOR_VERSION,
|
|
LeakageError,
|
|
assert_training_inputs_safe,
|
|
build_manifests,
|
|
)
|
|
from run_golden_qa_benchmark import run_benchmark # noqa: E402
|
|
|
|
WORKFLOW_VERSION = "2.0.0"
|
|
BENCHMARK_ID = "geointel-p4-reference-harness-v2"
|
|
GATE_STATES = {"pass", "fail", "not_evaluable"}
|
|
|
|
|
|
class EvidenceConflictError(RuntimeError):
|
|
"""Raised when an immutable evidence path already contains different bytes."""
|
|
|
|
|
|
def sha256(path: Path) -> str:
|
|
digest = hashlib.sha256()
|
|
with path.open("rb") as stream:
|
|
for chunk in iter(lambda: stream.read(1024 * 1024), b""):
|
|
digest.update(chunk)
|
|
return digest.hexdigest()
|
|
|
|
|
|
def json_bytes(payload: Any) -> bytes:
|
|
return (
|
|
json.dumps(payload, ensure_ascii=False, indent=2, sort_keys=True) + "\n"
|
|
).encode("utf-8")
|
|
|
|
|
|
def write_json_immutable(path: Path, payload: Any) -> None:
|
|
content = json_bytes(payload)
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
if path.exists():
|
|
if path.read_bytes() != content:
|
|
raise EvidenceConflictError(
|
|
f"Refusing to overwrite immutable evidence with different content: {path}"
|
|
)
|
|
return
|
|
temporary = path.with_name(f".{path.name}.tmp")
|
|
temporary.write_bytes(content)
|
|
temporary.replace(path)
|
|
|
|
|
|
def repository_commit(repo_root: Path) -> str | None:
|
|
try:
|
|
return subprocess.run(
|
|
["git", "rev-parse", "HEAD"],
|
|
cwd=repo_root,
|
|
check=True,
|
|
capture_output=True,
|
|
text=True,
|
|
).stdout.strip()
|
|
except (OSError, subprocess.CalledProcessError):
|
|
return None
|
|
|
|
|
|
def dependency_version(distribution: str) -> str | None:
|
|
try:
|
|
return importlib_metadata.version(distribution)
|
|
except importlib_metadata.PackageNotFoundError:
|
|
return None
|
|
|
|
|
|
def repository_file(repo_root: Path, relative_path: str) -> dict[str, Any]:
|
|
path = repo_root / relative_path
|
|
return {
|
|
"path": relative_path,
|
|
"sha256": sha256(path),
|
|
"size_bytes": path.stat().st_size,
|
|
}
|
|
|
|
|
|
def canonical_golden_baseline() -> dict[str, Any]:
|
|
result = run_benchmark()
|
|
scenarios = []
|
|
for item in result["scenarios"]:
|
|
normalized = dict(item)
|
|
normalized.pop("quality_check_id", None)
|
|
scenarios.append(normalized)
|
|
return {
|
|
"status": result["status"],
|
|
"version": result["version"],
|
|
"scenario_count": result["scenario_count"],
|
|
"scenarios": scenarios,
|
|
"persistence": result["persistence"],
|
|
"implementation": "backend/app/services/qa_service.py via scripts/run_golden_qa_benchmark.py",
|
|
"claim_boundary": "Reference implementation regression evidence; not production model accuracy.",
|
|
"content_sha256": canonical_hash(scenarios),
|
|
}
|
|
|
|
|
|
def product_baseline_manifest_gate(
|
|
repo_root: Path,
|
|
manifest_path: Path,
|
|
active_model: dict[str, Any],
|
|
) -> dict[str, Any]:
|
|
relative = manifest_path
|
|
try:
|
|
relative = manifest_path.resolve().relative_to(repo_root.resolve())
|
|
except (OSError, ValueError):
|
|
return {
|
|
"status": "fail",
|
|
"reason": "Product baseline manifest must reside inside the governed repository evidence root.",
|
|
"path": str(manifest_path),
|
|
}
|
|
if not manifest_path.is_file():
|
|
return {
|
|
"status": "not_evaluable",
|
|
"reason": "No executed, hash-bound product incumbent baseline manifest is available.",
|
|
"expected_path": relative.as_posix(),
|
|
}
|
|
try:
|
|
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
|
|
except (OSError, json.JSONDecodeError) as exc:
|
|
return {
|
|
"status": "fail",
|
|
"reason": f"Unreadable product baseline manifest: {exc}",
|
|
}
|
|
required_keys = {
|
|
"schema_version",
|
|
"status",
|
|
"synthetic",
|
|
"active_model_sha256",
|
|
"evaluator_sha256",
|
|
"configuration_sha256",
|
|
"protected_split_manifest",
|
|
"authoritative_reference_manifest",
|
|
"raw_predictions",
|
|
"metric_report",
|
|
"inference",
|
|
}
|
|
missing = sorted(required_keys - set(manifest))
|
|
violations: list[str] = []
|
|
if missing:
|
|
violations.append(f"missing_fields:{','.join(missing)}")
|
|
if manifest.get("status") != "pass":
|
|
violations.append("manifest_status_not_pass")
|
|
if manifest.get("synthetic") is not False:
|
|
violations.append("synthetic_or_unspecified")
|
|
if manifest.get("active_model_sha256") != active_model.get("sha256"):
|
|
violations.append("active_model_hash_mismatch")
|
|
evaluator_path = repo_root / "scripts/accuracy_phase4_evaluator.py"
|
|
if manifest.get("evaluator_sha256") != sha256(evaluator_path):
|
|
violations.append("evaluator_hash_mismatch")
|
|
inference = manifest.get("inference") or {}
|
|
if inference.get("executed") is not True:
|
|
violations.append("inference_not_executed")
|
|
if inference.get("test_used_for_selection") is not False:
|
|
violations.append("protected_test_selection_policy_invalid")
|
|
if not str(inference.get("device") or "").lower().startswith("cuda"):
|
|
violations.append("governed_cuda_execution_not_proven")
|
|
checked_artifacts: list[dict[str, Any]] = []
|
|
for key in (
|
|
"protected_split_manifest",
|
|
"authoritative_reference_manifest",
|
|
"raw_predictions",
|
|
"metric_report",
|
|
):
|
|
item = manifest.get(key) or {}
|
|
item_path = repo_root / str(item.get("path") or "")
|
|
try:
|
|
item_path.resolve().relative_to(repo_root.resolve())
|
|
except (OSError, ValueError):
|
|
violations.append(f"{key}_outside_repository")
|
|
continue
|
|
if not item_path.is_file():
|
|
violations.append(f"{key}_missing")
|
|
continue
|
|
observed_hash = sha256(item_path)
|
|
checked_artifacts.append(
|
|
{
|
|
"role": key,
|
|
"path": item_path.relative_to(repo_root).as_posix(),
|
|
"sha256": observed_hash,
|
|
}
|
|
)
|
|
if observed_hash != item.get("sha256"):
|
|
violations.append(f"{key}_hash_mismatch")
|
|
return {
|
|
"status": "fail" if violations else "pass",
|
|
"path": relative.as_posix(),
|
|
"manifest_sha256": sha256(manifest_path),
|
|
"violations": sorted(violations),
|
|
"checked_artifacts": checked_artifacts,
|
|
"evidence": (
|
|
"A non-synthetic active-model inference, protected split, authority reference, "
|
|
"raw predictions and metric report are all checksum-bound."
|
|
if not violations
|
|
else None
|
|
),
|
|
}
|
|
|
|
|
|
def readiness_snapshot(repo_root: Path) -> dict[str, Any]:
|
|
status_path = repo_root / "docs/accuracy-program/status.json"
|
|
p3_path = repo_root / "artifacts/evidence/accuracy/P3/full-scan-manifest.json"
|
|
leakage_path = repo_root / "artifacts/evidence/accuracy/P3/leakage-report.json"
|
|
status = json.loads(status_path.read_text(encoding="utf-8"))
|
|
p3 = json.loads(p3_path.read_text(encoding="utf-8"))
|
|
leakage = json.loads(leakage_path.read_text(encoding="utf-8"))
|
|
ml_data = status.get("ml_data") or {}
|
|
return {
|
|
"schema_version": 1,
|
|
"source_paths": {
|
|
"phase3_full_scan": repository_file(
|
|
repo_root, "artifacts/evidence/accuracy/P3/full-scan-manifest.json"
|
|
),
|
|
"phase3_leakage": repository_file(
|
|
repo_root, "artifacts/evidence/accuracy/P3/leakage-report.json"
|
|
),
|
|
},
|
|
"active_model": (status.get("runtime") or {}).get("active_model"),
|
|
"v56_review_and_split": (ml_data.get("v56") or {}),
|
|
"protected_test_isolation": ml_data.get("protected_test_isolation"),
|
|
"phase3_scan": {
|
|
"scan_id": p3.get("scan_id"),
|
|
"content_hash": p3.get("content_hash"),
|
|
"grb_consistency": p3.get("grb_consistency"),
|
|
},
|
|
"phase3_leakage_status": leakage.get("status"),
|
|
"authority_requirements": [
|
|
{"task": "building_validation", "zone": "flanders", "primary": "grb"},
|
|
{"task": "building_validation", "zone": "wallonia", "primary": "picc"},
|
|
{"task": "building_validation", "zone": "brussels", "primary": "urbis"},
|
|
{"task": "terrain_height", "zone": "flanders", "primary": "dhmv"},
|
|
{"task": "terrain_height", "zone": "wallonia", "primary": "spw_terrain"},
|
|
{
|
|
"task": "north_sea_bathymetry",
|
|
"zone": "belgian_north_sea",
|
|
"primary": "mdk",
|
|
},
|
|
{
|
|
"task": "imagery_corroboration",
|
|
"zone": "belgium",
|
|
"primary": "official_orthophoto",
|
|
"contextual": "sentinel-2",
|
|
},
|
|
],
|
|
}
|
|
|
|
|
|
def product_gate_evidence(
|
|
repo_root: Path,
|
|
snapshot: dict[str, Any],
|
|
product_baseline_manifest: Path,
|
|
) -> dict[str, Any]:
|
|
v56 = snapshot["v56_review_and_split"]
|
|
active_model = snapshot["active_model"] or {}
|
|
configured_path = Path(str(active_model.get("path") or ""))
|
|
p3_grb = snapshot["phase3_scan"].get("grb_consistency") or {}
|
|
baseline_gate = product_baseline_manifest_gate(
|
|
repo_root,
|
|
product_baseline_manifest,
|
|
active_model,
|
|
)
|
|
return {
|
|
"active_model_available_and_hash_verified": {
|
|
"status": "pass"
|
|
if configured_path.is_file()
|
|
and sha256(configured_path) == active_model.get("sha256")
|
|
else "not_evaluable",
|
|
"configured_path": str(configured_path),
|
|
"configured_sha256": active_model.get("sha256"),
|
|
"reason": None
|
|
if configured_path.is_file()
|
|
else "Configured active model is not locally accessible.",
|
|
},
|
|
"authoritative_reference_portfolio_available": {
|
|
"status": "not_evaluable",
|
|
"requirements": snapshot["authority_requirements"],
|
|
"observed_grb": p3_grb,
|
|
"reason": "A task- and zone-complete governed GRB/PICC/UrbIS/DHMV/SPW/MDK reference portfolio is not locally accessible.",
|
|
},
|
|
"human_review_complete": {
|
|
"status": "pass" if bool(v56.get("review_complete")) else "fail",
|
|
"observed": v56.get("reviewed_sample_count"),
|
|
"required": v56.get("sample_count"),
|
|
},
|
|
"split_independence": {
|
|
"status": "pass" if bool(v56.get("split_independence_proven")) else "fail",
|
|
"cross_split_pairs_below_2000_m": v56.get("cross_split_pairs_below_2000_m"),
|
|
},
|
|
"phase3_leakage_resolved": {
|
|
"status": "pass"
|
|
if snapshot.get("phase3_leakage_status") == "pass"
|
|
else "not_evaluable",
|
|
"observed": snapshot.get("phase3_leakage_status"),
|
|
},
|
|
"protected_storage_isolation": {
|
|
"status": "pass"
|
|
if snapshot.get("protected_test_isolation") is True
|
|
else "not_evaluable",
|
|
"reason": "A physically isolated vault, scoped credentials and immutable access log are not proven.",
|
|
},
|
|
"executed_product_incumbent_baseline": baseline_gate,
|
|
"representative_product_subgroup_support": {
|
|
"status": "not_evaluable",
|
|
"reason": "No real protected raw-prediction portfolio is available for AOI/region/context subgroup support.",
|
|
},
|
|
}
|
|
|
|
|
|
def build_release_gate_report(
|
|
split_result: dict[str, Any],
|
|
evaluation: dict[str, Any],
|
|
portfolio: dict[str, Any],
|
|
golden: dict[str, Any],
|
|
firewall_checks: dict[str, bool],
|
|
product_gates: dict[str, Any],
|
|
) -> dict[str, Any]:
|
|
declared_families = {item["task"] for item in evaluation["task_inventory"]}
|
|
observed_families = {item["task"] for item in evaluation["results"]}
|
|
required_split_roles = {"train", "val", "calibration", "test", "background-test"}
|
|
observed_split_roles = set(split_result["leakage"]["split_counts"])
|
|
required_raw_fields = {
|
|
"references",
|
|
"predictions_pre_filter",
|
|
"predictions_post_filter",
|
|
"config",
|
|
"input_lineage",
|
|
"portfolio_lineage",
|
|
}
|
|
raw_violations = [
|
|
item["sample_id"]
|
|
for item in evaluation["results"]
|
|
if not required_raw_fields <= set(item.get("raw") or {})
|
|
]
|
|
protected_policy = portfolio.get("protected_policy") or {}
|
|
selection_contract_valid = (
|
|
protected_policy.get("operating_point_selection_allowed") is False
|
|
and protected_policy.get("diagnostic_curves_select_operating_point") is False
|
|
and protected_policy.get("test_feedback_allowed") is False
|
|
and protected_policy.get("threshold_selection_source")
|
|
== "pre_registered_configuration_only"
|
|
and all(
|
|
isinstance((item.get("raw") or {}).get("config"), dict)
|
|
for item in evaluation["results"]
|
|
)
|
|
)
|
|
empty_case = next(
|
|
(
|
|
item
|
|
for item in evaluation["results"]
|
|
if item["sample_id"] == "background-test-pure-empty"
|
|
),
|
|
None,
|
|
)
|
|
empty_metrics = (empty_case or {}).get("metrics") or {}
|
|
null_semantics_valid = (
|
|
empty_case is not None
|
|
and empty_metrics.get("reference_count") == 0
|
|
and empty_metrics.get("prediction_count") == 0
|
|
and empty_metrics.get("precision") is None
|
|
and empty_metrics.get("recall") is None
|
|
and empty_metrics.get("f1") is None
|
|
)
|
|
subgroup_report = evaluation.get("subgroups") or {}
|
|
subgroup_contract_valid = (
|
|
subgroup_report.get("overall_status")
|
|
in {"not_evaluable", "evaluable_no_release_target"}
|
|
and isinstance(subgroup_report.get("dimensions"), dict)
|
|
and bool(subgroup_report.get("dimensions"))
|
|
and all(
|
|
isinstance(dimension.get("strata"), dict)
|
|
and isinstance(dimension.get("worst_stratum_by_task"), dict)
|
|
for dimension in subgroup_report["dimensions"].values()
|
|
)
|
|
)
|
|
capability_inventory = evaluation.get("task_inventory") or []
|
|
capability_contract_valid = bool(capability_inventory) and all(
|
|
item.get("capability_id")
|
|
and item.get("implementation_paths")
|
|
and item.get("suitable_metrics")
|
|
and item.get("evaluation_status")
|
|
in {
|
|
"synthetic_contract_case_only",
|
|
"covered_by_family_not_separately_benchmarked",
|
|
"not_separately_benchmarked",
|
|
"no_independent_accuracy_score_underlying_tool_results_are_authoritative",
|
|
"synthetic_metric_contract_only_no_generic_learned_classifier_claim",
|
|
}
|
|
for item in capability_inventory
|
|
)
|
|
local_gates = {
|
|
"all_declared_evaluator_families_exercised": {
|
|
"status": "pass" if declared_families == observed_families else "fail",
|
|
"declared": sorted(declared_families),
|
|
"observed": sorted(observed_families),
|
|
},
|
|
"implemented_capability_inventory": {
|
|
"status": "pass" if capability_contract_valid else "fail",
|
|
"capability_count": len(capability_inventory),
|
|
},
|
|
"normative_split_roles_and_leakage": {
|
|
"status": (
|
|
"pass"
|
|
if required_split_roles <= observed_split_roles
|
|
and split_result["leakage"]["status"] == "pass"
|
|
else "fail"
|
|
),
|
|
"required_roles": sorted(required_split_roles),
|
|
"observed_roles": sorted(observed_split_roles),
|
|
"leakage_status": split_result["leakage"]["status"],
|
|
},
|
|
"manifest_training_firewall_contract": {
|
|
"status": "pass"
|
|
if firewall_checks and all(firewall_checks.values())
|
|
else "fail",
|
|
"checks": firewall_checks,
|
|
},
|
|
"protected_operating_point_contract": {
|
|
"status": "pass" if selection_contract_valid else "fail",
|
|
"evidence": (
|
|
"Protected cases carry pre-registered configurations. Fixed AP/risk-coverage "
|
|
"diagnostics cannot select an operating point or feed back into training."
|
|
),
|
|
},
|
|
"complete_raw_predictions_retained": {
|
|
"status": "pass" if not raw_violations else "fail",
|
|
"violating_samples": raw_violations,
|
|
},
|
|
"reference_implementation_baseline": {
|
|
"status": "pass" if golden.get("status") == "passed" else "fail",
|
|
},
|
|
"stratified_metric_contract": {
|
|
"status": "pass" if subgroup_contract_valid else "fail",
|
|
"observed_overall_status": subgroup_report.get("overall_status"),
|
|
},
|
|
"undefined_metric_truth_table": {
|
|
"status": "pass" if null_semantics_valid else "fail",
|
|
"sample_id": "background-test-pure-empty",
|
|
},
|
|
}
|
|
invalid_gate_states = {
|
|
f"{family}.{name}": item.get("status")
|
|
for family, gates in (("local", local_gates), ("product", product_gates))
|
|
for name, item in gates.items()
|
|
if item.get("status") not in GATE_STATES
|
|
}
|
|
local_green = not invalid_gate_states and all(
|
|
item["status"] == "pass" for item in local_gates.values()
|
|
)
|
|
product_green = not invalid_gate_states and all(
|
|
item["status"] == "pass" for item in product_gates.values()
|
|
)
|
|
if not local_green:
|
|
overall_status = "fail"
|
|
elif not product_green:
|
|
overall_status = (
|
|
"not_evaluable"
|
|
if any(item["status"] == "not_evaluable" for item in product_gates.values())
|
|
else "fail"
|
|
)
|
|
else:
|
|
overall_status = "pass"
|
|
return {
|
|
"schema_version": 2,
|
|
"gate_policy": "geointel-p4-evaluation-harness-v2",
|
|
"status": overall_status,
|
|
"phase_decision": "ready_for_phase5" if overall_status == "pass" else "blocked",
|
|
"local_harness_status": "pass" if local_green else "fail",
|
|
"product_benchmark_status": "pass" if product_green else "not_evaluable",
|
|
"promotion_allowed": False,
|
|
"numeric_model_release_targets": "not_frozen_without_reviewed_representative_incumbent_baseline",
|
|
"local_gates": local_gates,
|
|
"product_gates": product_gates,
|
|
"invalid_gate_states": invalid_gate_states,
|
|
"critical_subgroup_policy": (
|
|
"Any required subgroup with insufficient support, missing metrics, a failed "
|
|
"non-inferiority comparison or regression blocks promotion; averages cannot override it."
|
|
),
|
|
"decision": (
|
|
"The local contract harness passes, but Phase 4 remains in progress and Phase 5 "
|
|
"is not ready until a real protected product incumbent baseline is evaluable."
|
|
if local_green and not product_green
|
|
else "All Phase 4 completion gates pass."
|
|
if local_green and product_green
|
|
else "The local Phase 4 harness has failing contract gates."
|
|
),
|
|
}
|
|
|
|
|
|
def build_evidence_manifest(
|
|
artifacts: dict[str, Any],
|
|
benchmark_manifest: dict[str, Any],
|
|
) -> dict[str, Any]:
|
|
retained = []
|
|
for name, payload in sorted(artifacts.items()):
|
|
content = json_bytes(payload)
|
|
retained.append(
|
|
{
|
|
"path": name,
|
|
"sha256": hashlib.sha256(content).hexdigest(),
|
|
"size_bytes": len(content),
|
|
}
|
|
)
|
|
return {
|
|
"schema_version": 2,
|
|
"phase": "P4",
|
|
"benchmark_id": benchmark_manifest["benchmark_id"],
|
|
"artifacts": retained,
|
|
"artifact_count": len(retained),
|
|
"claim_boundary": (
|
|
"Immutable local reference-harness evidence only; product accuracy and release "
|
|
"remain blocked while product gates are not evaluable or fail."
|
|
),
|
|
}
|
|
|
|
|
|
def firewall_contract_checks(
|
|
split_result: dict[str, Any],
|
|
protected_cases_path: Path,
|
|
) -> dict[str, bool]:
|
|
development = split_result["development"]["samples"]
|
|
protected = split_result["protected"]
|
|
train = [item for item in development if item["split"] == "train"]
|
|
validation = next(item for item in development if item["split"] == "val")
|
|
protected_item = protected["samples"][0]
|
|
checks: dict[str, bool] = {}
|
|
try:
|
|
assert_training_inputs_safe([], train, protected)
|
|
except LeakageError:
|
|
checks["clean_train_allowed"] = False
|
|
else:
|
|
checks["clean_train_allowed"] = True
|
|
for name, paths, records in (
|
|
("non_train_role_blocked", [], [validation]),
|
|
("protected_path_blocked", [protected_cases_path], []),
|
|
(
|
|
"renamed_protected_lineage_blocked",
|
|
[],
|
|
[{**train[0], "source_family": protected_item["source_family"]}],
|
|
),
|
|
):
|
|
try:
|
|
assert_training_inputs_safe(paths, records, protected)
|
|
except LeakageError:
|
|
checks[name] = True
|
|
else:
|
|
checks[name] = False
|
|
return checks
|
|
|
|
|
|
def runtime_identity() -> dict[str, Any]:
|
|
return {
|
|
"python": platform.python_version(),
|
|
"python_implementation": platform.python_implementation(),
|
|
"platform": platform.platform(),
|
|
"dependencies": {
|
|
"numpy": dependency_version("numpy"),
|
|
"pyproj": dependency_version("pyproj"),
|
|
"shapely": dependency_version("shapely"),
|
|
},
|
|
"execution_device": "CPU deterministic evaluator arithmetic; no production model inference",
|
|
"cuda_used_for_reference_harness": False,
|
|
}
|
|
|
|
|
|
def metric_results_without_raw(evaluation: dict[str, Any]) -> list[dict[str, Any]]:
|
|
return [
|
|
{key: value for key, value in item.items() if key not in {"raw", "failures"}}
|
|
for item in evaluation["results"]
|
|
]
|
|
|
|
|
|
def build_input_manifest(
|
|
repo_root: Path,
|
|
snapshot: dict[str, Any],
|
|
) -> dict[str, Any]:
|
|
code_paths = [
|
|
"scripts/accuracy_phase4_evaluator.py",
|
|
"scripts/generate_accuracy_phase4_splits.py",
|
|
"scripts/run_accuracy_phase4_benchmark.py",
|
|
"scripts/run_golden_qa_benchmark.py",
|
|
"backend/app/services/qa_service.py",
|
|
]
|
|
input_paths = [
|
|
"fixtures/accuracy/p4/split-source-manifest.json",
|
|
"fixtures/accuracy/p4/protected-baseline-cases.json",
|
|
"fixtures/golden/golden_qa_benchmarks.json",
|
|
"docs/accuracy-program/05-metric-framework.md",
|
|
"docs/accuracy-program/07-source-authority-matrix.md",
|
|
"artifacts/evidence/accuracy/P3/full-scan-manifest.json",
|
|
"artifacts/evidence/accuracy/P3/leakage-report.json",
|
|
]
|
|
return {
|
|
"schema_version": 2,
|
|
"repository_commit": repository_commit(repo_root),
|
|
"code": [repository_file(repo_root, path) for path in code_paths],
|
|
"inputs": [repository_file(repo_root, path) for path in input_paths],
|
|
"readiness_snapshot": snapshot,
|
|
"readiness_snapshot_sha256": canonical_hash(snapshot),
|
|
"runtime": runtime_identity(),
|
|
"model_execution": {
|
|
"status": "not_evaluable",
|
|
"reason": "The configured active model and governed protected product inputs are not locally accessible.",
|
|
"configured_active_model": snapshot.get("active_model"),
|
|
},
|
|
}
|
|
|
|
|
|
def run_workflow(
|
|
repo_root: Path,
|
|
output_dir: Path,
|
|
product_baseline_manifest: Path | None = None,
|
|
) -> dict[str, Any]:
|
|
source_path = repo_root / "fixtures/accuracy/p4/split-source-manifest.json"
|
|
cases_path = repo_root / "fixtures/accuracy/p4/protected-baseline-cases.json"
|
|
source = json.loads(source_path.read_text(encoding="utf-8"))
|
|
development, protected, leakage = build_manifests(source)
|
|
if leakage["status"] != "pass":
|
|
raise LeakageError(
|
|
f"Leakage gate failed with {leakage['finding_count']} findings"
|
|
)
|
|
split_result = {
|
|
"development": development,
|
|
"protected": protected,
|
|
"leakage": leakage,
|
|
"generation_status": {
|
|
"schema_version": 1,
|
|
"generator_version": GENERATOR_VERSION,
|
|
"status": "pass",
|
|
"source_manifest_sha256": leakage["source_manifest_sha256"],
|
|
"development_manifest_sha256": development["manifest_sha256"],
|
|
"protected_manifest_sha256": protected["manifest_sha256"],
|
|
},
|
|
}
|
|
protected_evaluation_ids = {
|
|
item["sample_id"]
|
|
for item in protected["samples"]
|
|
if item["split"] in {"test", "background-test"}
|
|
}
|
|
portfolio = json.loads(cases_path.read_text(encoding="utf-8"))
|
|
evaluation = evaluate_cases(cases_path, protected_evaluation_ids)
|
|
firewall_checks = firewall_contract_checks(split_result, cases_path)
|
|
golden = canonical_golden_baseline()
|
|
snapshot = readiness_snapshot(repo_root)
|
|
baseline_path = (
|
|
product_baseline_manifest
|
|
if product_baseline_manifest is not None
|
|
else repo_root / "artifacts/evidence/accuracy/P4/product-baseline-manifest.json"
|
|
)
|
|
product_gates = product_gate_evidence(repo_root, snapshot, baseline_path)
|
|
gate_report = build_release_gate_report(
|
|
split_result,
|
|
evaluation,
|
|
portfolio,
|
|
golden,
|
|
firewall_checks,
|
|
product_gates,
|
|
)
|
|
input_manifest = build_input_manifest(repo_root, snapshot)
|
|
evaluation_contract = {
|
|
"schema_version": 2,
|
|
"benchmark_id": BENCHMARK_ID,
|
|
"workflow_version": WORKFLOW_VERSION,
|
|
"evaluator_version": EVALUATOR_VERSION,
|
|
"split_generator_version": GENERATOR_VERSION,
|
|
"evaluator_families": sorted(evaluation["evaluated_task_families"]),
|
|
"implemented_capabilities": evaluation["task_inventory"],
|
|
"metric_contract": repository_file(
|
|
repo_root, "docs/accuracy-program/05-metric-framework.md"
|
|
),
|
|
"gate_states": sorted(GATE_STATES),
|
|
"protected_policy": evaluation["protected_policy"],
|
|
"undefined_value_policy": (
|
|
"Undefined denominators are null with numerator, denominator and support; "
|
|
"they are never coerced to a perfect score."
|
|
),
|
|
"claim_boundary": evaluation["claim_boundary"],
|
|
}
|
|
raw_items = [item["raw"] for item in evaluation["results"]]
|
|
raw_predictions = {
|
|
"schema_version": 2,
|
|
"evaluator_version": EVALUATOR_VERSION,
|
|
"portfolio_file_sha256": evaluation["portfolio_file_sha256"],
|
|
"items": raw_items,
|
|
"items_canonical_json_sha256": canonical_hash(raw_items),
|
|
"hash_specification": evaluation["hash_specification"],
|
|
}
|
|
metric_results = metric_results_without_raw(evaluation)
|
|
metric_report = {
|
|
key: value
|
|
for key, value in evaluation.items()
|
|
if key not in {"results", "failures"}
|
|
}
|
|
metric_report["results"] = metric_results
|
|
metric_report["metric_results_canonical_json_sha256"] = canonical_hash(
|
|
metric_results
|
|
)
|
|
metric_report["full_results_canonical_json_sha256"] = evaluation[
|
|
"results_canonical_json_sha256"
|
|
]
|
|
failure_gallery = {
|
|
"schema_version": 2,
|
|
"taxonomy": "docs/accuracy-program/05-metric-framework.md section 4",
|
|
"failure_count": len(evaluation["failures"]),
|
|
"items": evaluation["failures"],
|
|
"items_canonical_json_sha256": canonical_hash(evaluation["failures"]),
|
|
"rendering_status": (
|
|
"machine_readable_examples_retained; a visual production gallery requires "
|
|
"controlled access to protected imagery"
|
|
),
|
|
}
|
|
taxonomy_entries = sorted(
|
|
{(item["error_code"], item["kind"]) for item in evaluation["failures"]}
|
|
)
|
|
error_taxonomy = {
|
|
"schema_version": 2,
|
|
"source": "docs/accuracy-program/05-metric-framework.md section 4",
|
|
"observed_codes": [
|
|
{"error_code": code, "kind": kind} for code, kind in taxonomy_entries
|
|
],
|
|
"observed_failure_count": len(evaluation["failures"]),
|
|
"claim_boundary": evaluation["claim_boundary"],
|
|
}
|
|
object_task_names = {
|
|
"object_detection",
|
|
"footprint_segmentation",
|
|
"vector_comparison",
|
|
"change_detection",
|
|
"geospatial_data_validation",
|
|
}
|
|
object_metrics = {
|
|
"schema_version": 2,
|
|
"status": "fixture_contract_only",
|
|
"results": [
|
|
item for item in metric_results if item["task"] in object_task_names
|
|
],
|
|
}
|
|
tile_metrics = {
|
|
"schema_version": 2,
|
|
"status": "fixture_contract_only",
|
|
"results": [
|
|
item
|
|
for item in metric_results
|
|
if item["task"] in {"raster_classification", "terrain_interpretation"}
|
|
],
|
|
}
|
|
aoi_metrics = {
|
|
"schema_version": 2,
|
|
"status": "not_evaluable",
|
|
"reason": (
|
|
"Synthetic single-case fixtures do not provide independent product AOI clusters. "
|
|
"AOI micro/macro and cluster-bootstrap evidence requires the protected product corpus."
|
|
),
|
|
"required_future_outputs": [
|
|
"per-AOI primary metrics",
|
|
"micro and macro aggregation",
|
|
"paired candidate-minus-incumbent deltas",
|
|
"cluster-bootstrap confidence intervals",
|
|
],
|
|
}
|
|
stratified_metrics = evaluation["subgroups"]
|
|
calibration_items = []
|
|
for item in metric_results:
|
|
calibration = item["metrics"].get("calibration")
|
|
coverage_risk = item["metrics"].get("coverage_risk")
|
|
if calibration is not None or coverage_risk is not None:
|
|
calibration_items.append(
|
|
{
|
|
"sample_id": item["sample_id"],
|
|
"task": item["task"],
|
|
"calibration": calibration,
|
|
"coverage_risk": coverage_risk,
|
|
}
|
|
)
|
|
calibration_metrics = {
|
|
"schema_version": 2,
|
|
"status": "fixture_diagnostic_only",
|
|
"selection_allowed": False,
|
|
"items": calibration_items,
|
|
"note": (
|
|
"Fixed diagnostic bins and risk thresholds test metric arithmetic; they do not "
|
|
"select or change any operating point."
|
|
),
|
|
}
|
|
latency_reliability = {
|
|
"schema_version": 2,
|
|
"status": "not_evaluable",
|
|
"model_inference_executed": False,
|
|
"reason": (
|
|
"The reference harness performs deterministic evaluator arithmetic only. "
|
|
"GPU latency, VRAM, throughput and failure-rate gates require the real active model."
|
|
),
|
|
}
|
|
human_review_summary = {
|
|
"schema_version": 2,
|
|
"status": product_gates["human_review_complete"]["status"],
|
|
"reviewed": product_gates["human_review_complete"].get("observed"),
|
|
"required": product_gates["human_review_complete"].get("required"),
|
|
"source": "readiness-snapshot.json bound to Phase-1/3 evidence",
|
|
"ai_review_is_human_signoff": False,
|
|
}
|
|
candidate_vs_incumbent = {
|
|
"schema_version": 2,
|
|
"status": "not_evaluable",
|
|
"reason": (
|
|
"Phase 4 has no valid real incumbent product baseline and no pre-registered "
|
|
"candidate; synthetic fixture values cannot define non-inferiority."
|
|
),
|
|
"future_gate_contract": {
|
|
"unit": "paired independent AOI",
|
|
"global_and_critical_subgroups_required": True,
|
|
"missing_or_insufficient_support": "not_evaluable",
|
|
"aggregate_improvement_may_mask_subgroup_regression": False,
|
|
"numeric_margin": "to_be_frozen_before_protected_access",
|
|
},
|
|
}
|
|
input_manifest_file_sha256 = hashlib.sha256(json_bytes(input_manifest)).hexdigest()
|
|
benchmark_manifest = {
|
|
"schema_version": 2,
|
|
"benchmark_id": BENCHMARK_ID,
|
|
"workflow_version": WORKFLOW_VERSION,
|
|
"evaluator_version": EVALUATOR_VERSION,
|
|
"split_generator_version": GENERATOR_VERSION,
|
|
"repository_commit": input_manifest["repository_commit"],
|
|
"input_manifest": {
|
|
"path": "input-manifest.json",
|
|
"sha256": input_manifest_file_sha256,
|
|
},
|
|
"inputs": {
|
|
"split_source": repository_file(
|
|
repo_root, "fixtures/accuracy/p4/split-source-manifest.json"
|
|
),
|
|
"protected_cases": repository_file(
|
|
repo_root, "fixtures/accuracy/p4/protected-baseline-cases.json"
|
|
),
|
|
"golden_qa_manifest": repository_file(
|
|
repo_root, "fixtures/golden/golden_qa_benchmarks.json"
|
|
),
|
|
"phase3_full_scan": repository_file(
|
|
repo_root, "artifacts/evidence/accuracy/P3/full-scan-manifest.json"
|
|
),
|
|
"phase3_leakage": repository_file(
|
|
repo_root, "artifacts/evidence/accuracy/P3/leakage-report.json"
|
|
),
|
|
},
|
|
"code": input_manifest["code"],
|
|
"runtime": input_manifest["runtime"],
|
|
"split_manifests": {
|
|
"development_sha256": development["manifest_sha256"],
|
|
"protected_sha256": protected["manifest_sha256"],
|
|
"leakage_status": leakage["status"],
|
|
},
|
|
"inference_and_selection": {
|
|
"synthetic_reference_harness": True,
|
|
"production_model_inference_executed": False,
|
|
"test_used_for_selection": False,
|
|
"background_test_used_for_selection": False,
|
|
"challenge_labels_available": False,
|
|
"raw_predictions_retained": True,
|
|
"threshold_source": "pre_registered_configuration_only",
|
|
},
|
|
"evaluation_results_canonical_json_sha256": evaluation[
|
|
"results_canonical_json_sha256"
|
|
],
|
|
"reference_baseline_sha256": golden["content_sha256"],
|
|
"claim_boundary": evaluation["claim_boundary"],
|
|
}
|
|
benchmark_manifest["manifest_sha256"] = canonical_hash(benchmark_manifest)
|
|
gate_report["benchmark_manifest_sha256"] = benchmark_manifest["manifest_sha256"]
|
|
workflow_summary = {
|
|
"schema_version": 2,
|
|
"status": gate_report["status"],
|
|
"phase_decision": gate_report["phase_decision"],
|
|
"local_harness_status": gate_report["local_harness_status"],
|
|
"product_benchmark_status": gate_report["product_benchmark_status"],
|
|
"benchmark_manifest_sha256": benchmark_manifest["manifest_sha256"],
|
|
"split_counts": leakage["split_counts"],
|
|
"task_family_count": evaluation["task_count"],
|
|
"implemented_capability_count": len(evaluation["task_inventory"]),
|
|
"case_count": evaluation["case_count"],
|
|
"failure_example_count": len(evaluation["failures"]),
|
|
"evaluation_results_canonical_json_sha256": evaluation[
|
|
"results_canonical_json_sha256"
|
|
],
|
|
"reference_baseline_sha256": golden["content_sha256"],
|
|
"promotion_allowed": False,
|
|
"phase4_done": gate_report["status"] == "pass",
|
|
"phase5_ready": gate_report["status"] == "pass",
|
|
}
|
|
artifacts: dict[str, Any] = {
|
|
"acceptance-gates.json": gate_report,
|
|
"aoi-metrics.json": aoi_metrics,
|
|
"baseline-raw-predictions.json": raw_predictions,
|
|
"benchmark-manifest.json": benchmark_manifest,
|
|
"calibration-metrics.json": calibration_metrics,
|
|
"candidate-vs-incumbent.json": candidate_vs_incumbent,
|
|
"development-split-manifest.json": development,
|
|
"error-taxonomy.json": error_taxonomy,
|
|
"evaluation-contract.json": evaluation_contract,
|
|
"failure-gallery.json": failure_gallery,
|
|
"generation-status.json": split_result["generation_status"],
|
|
"human-review-summary.json": human_review_summary,
|
|
"input-manifest.json": input_manifest,
|
|
"latency-and-reliability.json": latency_reliability,
|
|
"leakage-gate-report.json": leakage,
|
|
"metric-report.json": metric_report,
|
|
"object-metrics.json": object_metrics,
|
|
"protected-split-manifest.json": protected,
|
|
"reference-implementation-baseline.json": golden,
|
|
"release-gate-report.json": gate_report,
|
|
"split-and-leakage-audit.json": leakage,
|
|
"stratified-metrics.json": stratified_metrics,
|
|
"tile-metrics.json": tile_metrics,
|
|
"workflow-summary.json": workflow_summary,
|
|
}
|
|
evidence = build_evidence_manifest(artifacts, benchmark_manifest)
|
|
for name, payload in sorted(artifacts.items()):
|
|
write_json_immutable(output_dir / name, payload)
|
|
write_json_immutable(output_dir / "evidence-manifest.json", evidence)
|
|
return workflow_summary
|
|
|
|
|
|
def parse_args() -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
parser.add_argument("--repo-root", type=Path, default=ROOT)
|
|
parser.add_argument(
|
|
"--output-dir",
|
|
type=Path,
|
|
default=ROOT / "artifacts/evidence/accuracy/P4/reference-harness-v2",
|
|
)
|
|
parser.add_argument(
|
|
"--product-baseline-manifest",
|
|
type=Path,
|
|
help=(
|
|
"Optional governed product incumbent manifest. It can pass only when real "
|
|
"active-model inference and all referenced artifacts validate."
|
|
),
|
|
)
|
|
parser.add_argument(
|
|
"--allow-product-blocked",
|
|
action="store_true",
|
|
help=(
|
|
"Return zero when the local harness passes while product evidence remains "
|
|
"fail/not_evaluable. This never changes a gate or phase decision."
|
|
),
|
|
)
|
|
return parser.parse_args()
|
|
|
|
|
|
def main() -> int:
|
|
args = parse_args()
|
|
try:
|
|
summary = run_workflow(
|
|
args.repo_root.resolve(),
|
|
args.output_dir.resolve(),
|
|
args.product_baseline_manifest.resolve()
|
|
if args.product_baseline_manifest
|
|
else None,
|
|
)
|
|
except Exception as exc: # noqa: BLE001 - workflow evidence must fail closed
|
|
print(
|
|
json.dumps(
|
|
{"status": "fail", "error": f"{type(exc).__name__}: {exc}"},
|
|
indent=2,
|
|
)
|
|
)
|
|
return 2
|
|
print(json.dumps(summary, indent=2, sort_keys=True))
|
|
if summary["status"] == "pass":
|
|
return 0
|
|
return (
|
|
0
|
|
if args.allow_product_blocked and summary["local_harness_status"] == "pass"
|
|
else 2
|
|
)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|