#!/usr/bin/env python3 """Run the complete local Phase 4 evaluation workflow from frozen inputs.""" from __future__ import annotations import argparse import hashlib import json import platform import subprocess import sys from importlib import metadata as importlib_metadata from pathlib import Path from typing import Any ROOT = Path(__file__).resolve().parents[1] BACKEND_ROOT = ROOT / "backend" for entry in (str(ROOT), str(BACKEND_ROOT), str(ROOT / "scripts")): if entry not in sys.path: sys.path.insert(0, entry) from accuracy_phase4_evaluator import EVALUATOR_VERSION, canonical_hash, evaluate_cases # noqa: E402 from generate_accuracy_phase4_splits import ( # noqa: E402 GENERATOR_VERSION, LeakageError, assert_training_inputs_safe, build_manifests, ) from run_golden_qa_benchmark import run_benchmark # noqa: E402 WORKFLOW_VERSION = "2.0.0" BENCHMARK_ID = "geointel-p4-reference-harness-v2" GATE_STATES = {"pass", "fail", "not_evaluable"} class EvidenceConflictError(RuntimeError): """Raised when an immutable evidence path already contains different bytes.""" def sha256(path: Path) -> str: digest = hashlib.sha256() with path.open("rb") as stream: for chunk in iter(lambda: stream.read(1024 * 1024), b""): digest.update(chunk) return digest.hexdigest() def json_bytes(payload: Any) -> bytes: return ( json.dumps(payload, ensure_ascii=False, indent=2, sort_keys=True) + "\n" ).encode("utf-8") def write_json_immutable(path: Path, payload: Any) -> None: content = json_bytes(payload) path.parent.mkdir(parents=True, exist_ok=True) if path.exists(): if path.read_bytes() != content: raise EvidenceConflictError( f"Refusing to overwrite immutable evidence with different content: {path}" ) return temporary = path.with_name(f".{path.name}.tmp") temporary.write_bytes(content) temporary.replace(path) def repository_commit(repo_root: Path) -> str | None: try: return subprocess.run( ["git", "rev-parse", "HEAD"], cwd=repo_root, check=True, capture_output=True, text=True, ).stdout.strip() except (OSError, subprocess.CalledProcessError): return None def dependency_version(distribution: str) -> str | None: try: return importlib_metadata.version(distribution) except importlib_metadata.PackageNotFoundError: return None def repository_file(repo_root: Path, relative_path: str) -> dict[str, Any]: path = repo_root / relative_path return { "path": relative_path, "sha256": sha256(path), "size_bytes": path.stat().st_size, } def canonical_golden_baseline() -> dict[str, Any]: result = run_benchmark() scenarios = [] for item in result["scenarios"]: normalized = dict(item) normalized.pop("quality_check_id", None) scenarios.append(normalized) return { "status": result["status"], "version": result["version"], "scenario_count": result["scenario_count"], "scenarios": scenarios, "persistence": result["persistence"], "implementation": "backend/app/services/qa_service.py via scripts/run_golden_qa_benchmark.py", "claim_boundary": "Reference implementation regression evidence; not production model accuracy.", "content_sha256": canonical_hash(scenarios), } def product_baseline_manifest_gate( repo_root: Path, manifest_path: Path, active_model: dict[str, Any], ) -> dict[str, Any]: relative = manifest_path try: relative = manifest_path.resolve().relative_to(repo_root.resolve()) except (OSError, ValueError): return { "status": "fail", "reason": "Product baseline manifest must reside inside the governed repository evidence root.", "path": str(manifest_path), } if not manifest_path.is_file(): return { "status": "not_evaluable", "reason": "No executed, hash-bound product incumbent baseline manifest is available.", "expected_path": relative.as_posix(), } try: manifest = json.loads(manifest_path.read_text(encoding="utf-8")) except (OSError, json.JSONDecodeError) as exc: return { "status": "fail", "reason": f"Unreadable product baseline manifest: {exc}", } required_keys = { "schema_version", "status", "synthetic", "active_model_sha256", "evaluator_sha256", "configuration_sha256", "protected_split_manifest", "authoritative_reference_manifest", "raw_predictions", "metric_report", "inference", } missing = sorted(required_keys - set(manifest)) violations: list[str] = [] if missing: violations.append(f"missing_fields:{','.join(missing)}") if manifest.get("status") != "pass": violations.append("manifest_status_not_pass") if manifest.get("synthetic") is not False: violations.append("synthetic_or_unspecified") if manifest.get("active_model_sha256") != active_model.get("sha256"): violations.append("active_model_hash_mismatch") evaluator_path = repo_root / "scripts/accuracy_phase4_evaluator.py" if manifest.get("evaluator_sha256") != sha256(evaluator_path): violations.append("evaluator_hash_mismatch") inference = manifest.get("inference") or {} if inference.get("executed") is not True: violations.append("inference_not_executed") if inference.get("test_used_for_selection") is not False: violations.append("protected_test_selection_policy_invalid") if not str(inference.get("device") or "").lower().startswith("cuda"): violations.append("governed_cuda_execution_not_proven") checked_artifacts: list[dict[str, Any]] = [] for key in ( "protected_split_manifest", "authoritative_reference_manifest", "raw_predictions", "metric_report", ): item = manifest.get(key) or {} item_path = repo_root / str(item.get("path") or "") try: item_path.resolve().relative_to(repo_root.resolve()) except (OSError, ValueError): violations.append(f"{key}_outside_repository") continue if not item_path.is_file(): violations.append(f"{key}_missing") continue observed_hash = sha256(item_path) checked_artifacts.append( { "role": key, "path": item_path.relative_to(repo_root).as_posix(), "sha256": observed_hash, } ) if observed_hash != item.get("sha256"): violations.append(f"{key}_hash_mismatch") return { "status": "fail" if violations else "pass", "path": relative.as_posix(), "manifest_sha256": sha256(manifest_path), "violations": sorted(violations), "checked_artifacts": checked_artifacts, "evidence": ( "A non-synthetic active-model inference, protected split, authority reference, " "raw predictions and metric report are all checksum-bound." if not violations else None ), } def readiness_snapshot(repo_root: Path) -> dict[str, Any]: status_path = repo_root / "docs/accuracy-program/status.json" p3_path = repo_root / "artifacts/evidence/accuracy/P3/full-scan-manifest.json" leakage_path = repo_root / "artifacts/evidence/accuracy/P3/leakage-report.json" status = json.loads(status_path.read_text(encoding="utf-8")) p3 = json.loads(p3_path.read_text(encoding="utf-8")) leakage = json.loads(leakage_path.read_text(encoding="utf-8")) ml_data = status.get("ml_data") or {} return { "schema_version": 1, "source_paths": { "phase3_full_scan": repository_file( repo_root, "artifacts/evidence/accuracy/P3/full-scan-manifest.json" ), "phase3_leakage": repository_file( repo_root, "artifacts/evidence/accuracy/P3/leakage-report.json" ), }, "active_model": (status.get("runtime") or {}).get("active_model"), "v56_review_and_split": (ml_data.get("v56") or {}), "protected_test_isolation": ml_data.get("protected_test_isolation"), "phase3_scan": { "scan_id": p3.get("scan_id"), "content_hash": p3.get("content_hash"), "grb_consistency": p3.get("grb_consistency"), }, "phase3_leakage_status": leakage.get("status"), "authority_requirements": [ {"task": "building_validation", "zone": "flanders", "primary": "grb"}, {"task": "building_validation", "zone": "wallonia", "primary": "picc"}, {"task": "building_validation", "zone": "brussels", "primary": "urbis"}, {"task": "terrain_height", "zone": "flanders", "primary": "dhmv"}, {"task": "terrain_height", "zone": "wallonia", "primary": "spw_terrain"}, { "task": "north_sea_bathymetry", "zone": "belgian_north_sea", "primary": "mdk", }, { "task": "imagery_corroboration", "zone": "belgium", "primary": "official_orthophoto", "contextual": "sentinel-2", }, ], } def product_gate_evidence( repo_root: Path, snapshot: dict[str, Any], product_baseline_manifest: Path, ) -> dict[str, Any]: v56 = snapshot["v56_review_and_split"] active_model = snapshot["active_model"] or {} configured_path = Path(str(active_model.get("path") or "")) p3_grb = snapshot["phase3_scan"].get("grb_consistency") or {} baseline_gate = product_baseline_manifest_gate( repo_root, product_baseline_manifest, active_model, ) return { "active_model_available_and_hash_verified": { "status": "pass" if configured_path.is_file() and sha256(configured_path) == active_model.get("sha256") else "not_evaluable", "configured_path": str(configured_path), "configured_sha256": active_model.get("sha256"), "reason": None if configured_path.is_file() else "Configured active model is not locally accessible.", }, "authoritative_reference_portfolio_available": { "status": "not_evaluable", "requirements": snapshot["authority_requirements"], "observed_grb": p3_grb, "reason": "A task- and zone-complete governed GRB/PICC/UrbIS/DHMV/SPW/MDK reference portfolio is not locally accessible.", }, "human_review_complete": { "status": "pass" if bool(v56.get("review_complete")) else "fail", "observed": v56.get("reviewed_sample_count"), "required": v56.get("sample_count"), }, "split_independence": { "status": "pass" if bool(v56.get("split_independence_proven")) else "fail", "cross_split_pairs_below_2000_m": v56.get("cross_split_pairs_below_2000_m"), }, "phase3_leakage_resolved": { "status": "pass" if snapshot.get("phase3_leakage_status") == "pass" else "not_evaluable", "observed": snapshot.get("phase3_leakage_status"), }, "protected_storage_isolation": { "status": "pass" if snapshot.get("protected_test_isolation") is True else "not_evaluable", "reason": "A physically isolated vault, scoped credentials and immutable access log are not proven.", }, "executed_product_incumbent_baseline": baseline_gate, "representative_product_subgroup_support": { "status": "not_evaluable", "reason": "No real protected raw-prediction portfolio is available for AOI/region/context subgroup support.", }, } def build_release_gate_report( split_result: dict[str, Any], evaluation: dict[str, Any], portfolio: dict[str, Any], golden: dict[str, Any], firewall_checks: dict[str, bool], product_gates: dict[str, Any], ) -> dict[str, Any]: declared_families = {item["task"] for item in evaluation["task_inventory"]} observed_families = {item["task"] for item in evaluation["results"]} required_split_roles = {"train", "val", "calibration", "test", "background-test"} observed_split_roles = set(split_result["leakage"]["split_counts"]) required_raw_fields = { "references", "predictions_pre_filter", "predictions_post_filter", "config", "input_lineage", "portfolio_lineage", } raw_violations = [ item["sample_id"] for item in evaluation["results"] if not required_raw_fields <= set(item.get("raw") or {}) ] protected_policy = portfolio.get("protected_policy") or {} selection_contract_valid = ( protected_policy.get("operating_point_selection_allowed") is False and protected_policy.get("diagnostic_curves_select_operating_point") is False and protected_policy.get("test_feedback_allowed") is False and protected_policy.get("threshold_selection_source") == "pre_registered_configuration_only" and all( isinstance((item.get("raw") or {}).get("config"), dict) for item in evaluation["results"] ) ) empty_case = next( ( item for item in evaluation["results"] if item["sample_id"] == "background-test-pure-empty" ), None, ) empty_metrics = (empty_case or {}).get("metrics") or {} null_semantics_valid = ( empty_case is not None and empty_metrics.get("reference_count") == 0 and empty_metrics.get("prediction_count") == 0 and empty_metrics.get("precision") is None and empty_metrics.get("recall") is None and empty_metrics.get("f1") is None ) subgroup_report = evaluation.get("subgroups") or {} subgroup_contract_valid = ( subgroup_report.get("overall_status") in {"not_evaluable", "evaluable_no_release_target"} and isinstance(subgroup_report.get("dimensions"), dict) and bool(subgroup_report.get("dimensions")) and all( isinstance(dimension.get("strata"), dict) and isinstance(dimension.get("worst_stratum_by_task"), dict) for dimension in subgroup_report["dimensions"].values() ) ) capability_inventory = evaluation.get("task_inventory") or [] capability_contract_valid = bool(capability_inventory) and all( item.get("capability_id") and item.get("implementation_paths") and item.get("suitable_metrics") and item.get("evaluation_status") in { "synthetic_contract_case_only", "covered_by_family_not_separately_benchmarked", "not_separately_benchmarked", "no_independent_accuracy_score_underlying_tool_results_are_authoritative", "synthetic_metric_contract_only_no_generic_learned_classifier_claim", } for item in capability_inventory ) local_gates = { "all_declared_evaluator_families_exercised": { "status": "pass" if declared_families == observed_families else "fail", "declared": sorted(declared_families), "observed": sorted(observed_families), }, "implemented_capability_inventory": { "status": "pass" if capability_contract_valid else "fail", "capability_count": len(capability_inventory), }, "normative_split_roles_and_leakage": { "status": ( "pass" if required_split_roles <= observed_split_roles and split_result["leakage"]["status"] == "pass" else "fail" ), "required_roles": sorted(required_split_roles), "observed_roles": sorted(observed_split_roles), "leakage_status": split_result["leakage"]["status"], }, "manifest_training_firewall_contract": { "status": "pass" if firewall_checks and all(firewall_checks.values()) else "fail", "checks": firewall_checks, }, "protected_operating_point_contract": { "status": "pass" if selection_contract_valid else "fail", "evidence": ( "Protected cases carry pre-registered configurations. Fixed AP/risk-coverage " "diagnostics cannot select an operating point or feed back into training." ), }, "complete_raw_predictions_retained": { "status": "pass" if not raw_violations else "fail", "violating_samples": raw_violations, }, "reference_implementation_baseline": { "status": "pass" if golden.get("status") == "passed" else "fail", }, "stratified_metric_contract": { "status": "pass" if subgroup_contract_valid else "fail", "observed_overall_status": subgroup_report.get("overall_status"), }, "undefined_metric_truth_table": { "status": "pass" if null_semantics_valid else "fail", "sample_id": "background-test-pure-empty", }, } invalid_gate_states = { f"{family}.{name}": item.get("status") for family, gates in (("local", local_gates), ("product", product_gates)) for name, item in gates.items() if item.get("status") not in GATE_STATES } local_green = not invalid_gate_states and all( item["status"] == "pass" for item in local_gates.values() ) product_green = not invalid_gate_states and all( item["status"] == "pass" for item in product_gates.values() ) if not local_green: overall_status = "fail" elif not product_green: overall_status = ( "not_evaluable" if any(item["status"] == "not_evaluable" for item in product_gates.values()) else "fail" ) else: overall_status = "pass" return { "schema_version": 2, "gate_policy": "geointel-p4-evaluation-harness-v2", "status": overall_status, "phase_decision": "ready_for_phase5" if overall_status == "pass" else "blocked", "local_harness_status": "pass" if local_green else "fail", "product_benchmark_status": "pass" if product_green else "not_evaluable", "promotion_allowed": False, "numeric_model_release_targets": "not_frozen_without_reviewed_representative_incumbent_baseline", "local_gates": local_gates, "product_gates": product_gates, "invalid_gate_states": invalid_gate_states, "critical_subgroup_policy": ( "Any required subgroup with insufficient support, missing metrics, a failed " "non-inferiority comparison or regression blocks promotion; averages cannot override it." ), "decision": ( "The local contract harness passes, but Phase 4 remains in progress and Phase 5 " "is not ready until a real protected product incumbent baseline is evaluable." if local_green and not product_green else "All Phase 4 completion gates pass." if local_green and product_green else "The local Phase 4 harness has failing contract gates." ), } def build_evidence_manifest( artifacts: dict[str, Any], benchmark_manifest: dict[str, Any], ) -> dict[str, Any]: retained = [] for name, payload in sorted(artifacts.items()): content = json_bytes(payload) retained.append( { "path": name, "sha256": hashlib.sha256(content).hexdigest(), "size_bytes": len(content), } ) return { "schema_version": 2, "phase": "P4", "benchmark_id": benchmark_manifest["benchmark_id"], "artifacts": retained, "artifact_count": len(retained), "claim_boundary": ( "Immutable local reference-harness evidence only; product accuracy and release " "remain blocked while product gates are not evaluable or fail." ), } def firewall_contract_checks( split_result: dict[str, Any], protected_cases_path: Path, ) -> dict[str, bool]: development = split_result["development"]["samples"] protected = split_result["protected"] train = [item for item in development if item["split"] == "train"] validation = next(item for item in development if item["split"] == "val") protected_item = protected["samples"][0] checks: dict[str, bool] = {} try: assert_training_inputs_safe([], train, protected) except LeakageError: checks["clean_train_allowed"] = False else: checks["clean_train_allowed"] = True for name, paths, records in ( ("non_train_role_blocked", [], [validation]), ("protected_path_blocked", [protected_cases_path], []), ( "renamed_protected_lineage_blocked", [], [{**train[0], "source_family": protected_item["source_family"]}], ), ): try: assert_training_inputs_safe(paths, records, protected) except LeakageError: checks[name] = True else: checks[name] = False return checks def runtime_identity() -> dict[str, Any]: return { "python": platform.python_version(), "python_implementation": platform.python_implementation(), "platform": platform.platform(), "dependencies": { "numpy": dependency_version("numpy"), "pyproj": dependency_version("pyproj"), "shapely": dependency_version("shapely"), }, "execution_device": "CPU deterministic evaluator arithmetic; no production model inference", "cuda_used_for_reference_harness": False, } def metric_results_without_raw(evaluation: dict[str, Any]) -> list[dict[str, Any]]: return [ {key: value for key, value in item.items() if key not in {"raw", "failures"}} for item in evaluation["results"] ] def build_input_manifest( repo_root: Path, snapshot: dict[str, Any], ) -> dict[str, Any]: code_paths = [ "scripts/accuracy_phase4_evaluator.py", "scripts/generate_accuracy_phase4_splits.py", "scripts/run_accuracy_phase4_benchmark.py", "scripts/run_golden_qa_benchmark.py", "backend/app/services/qa_service.py", ] input_paths = [ "fixtures/accuracy/p4/split-source-manifest.json", "fixtures/accuracy/p4/protected-baseline-cases.json", "fixtures/golden/golden_qa_benchmarks.json", "docs/accuracy-program/05-metric-framework.md", "docs/accuracy-program/07-source-authority-matrix.md", "artifacts/evidence/accuracy/P3/full-scan-manifest.json", "artifacts/evidence/accuracy/P3/leakage-report.json", ] return { "schema_version": 2, "repository_commit": repository_commit(repo_root), "code": [repository_file(repo_root, path) for path in code_paths], "inputs": [repository_file(repo_root, path) for path in input_paths], "readiness_snapshot": snapshot, "readiness_snapshot_sha256": canonical_hash(snapshot), "runtime": runtime_identity(), "model_execution": { "status": "not_evaluable", "reason": "The configured active model and governed protected product inputs are not locally accessible.", "configured_active_model": snapshot.get("active_model"), }, } def run_workflow( repo_root: Path, output_dir: Path, product_baseline_manifest: Path | None = None, ) -> dict[str, Any]: source_path = repo_root / "fixtures/accuracy/p4/split-source-manifest.json" cases_path = repo_root / "fixtures/accuracy/p4/protected-baseline-cases.json" source = json.loads(source_path.read_text(encoding="utf-8")) development, protected, leakage = build_manifests(source) if leakage["status"] != "pass": raise LeakageError( f"Leakage gate failed with {leakage['finding_count']} findings" ) split_result = { "development": development, "protected": protected, "leakage": leakage, "generation_status": { "schema_version": 1, "generator_version": GENERATOR_VERSION, "status": "pass", "source_manifest_sha256": leakage["source_manifest_sha256"], "development_manifest_sha256": development["manifest_sha256"], "protected_manifest_sha256": protected["manifest_sha256"], }, } protected_evaluation_ids = { item["sample_id"] for item in protected["samples"] if item["split"] in {"test", "background-test"} } portfolio = json.loads(cases_path.read_text(encoding="utf-8")) evaluation = evaluate_cases(cases_path, protected_evaluation_ids) firewall_checks = firewall_contract_checks(split_result, cases_path) golden = canonical_golden_baseline() snapshot = readiness_snapshot(repo_root) baseline_path = ( product_baseline_manifest if product_baseline_manifest is not None else repo_root / "artifacts/evidence/accuracy/P4/product-baseline-manifest.json" ) product_gates = product_gate_evidence(repo_root, snapshot, baseline_path) gate_report = build_release_gate_report( split_result, evaluation, portfolio, golden, firewall_checks, product_gates, ) input_manifest = build_input_manifest(repo_root, snapshot) evaluation_contract = { "schema_version": 2, "benchmark_id": BENCHMARK_ID, "workflow_version": WORKFLOW_VERSION, "evaluator_version": EVALUATOR_VERSION, "split_generator_version": GENERATOR_VERSION, "evaluator_families": sorted(evaluation["evaluated_task_families"]), "implemented_capabilities": evaluation["task_inventory"], "metric_contract": repository_file( repo_root, "docs/accuracy-program/05-metric-framework.md" ), "gate_states": sorted(GATE_STATES), "protected_policy": evaluation["protected_policy"], "undefined_value_policy": ( "Undefined denominators are null with numerator, denominator and support; " "they are never coerced to a perfect score." ), "claim_boundary": evaluation["claim_boundary"], } raw_items = [item["raw"] for item in evaluation["results"]] raw_predictions = { "schema_version": 2, "evaluator_version": EVALUATOR_VERSION, "portfolio_file_sha256": evaluation["portfolio_file_sha256"], "items": raw_items, "items_canonical_json_sha256": canonical_hash(raw_items), "hash_specification": evaluation["hash_specification"], } metric_results = metric_results_without_raw(evaluation) metric_report = { key: value for key, value in evaluation.items() if key not in {"results", "failures"} } metric_report["results"] = metric_results metric_report["metric_results_canonical_json_sha256"] = canonical_hash( metric_results ) metric_report["full_results_canonical_json_sha256"] = evaluation[ "results_canonical_json_sha256" ] failure_gallery = { "schema_version": 2, "taxonomy": "docs/accuracy-program/05-metric-framework.md section 4", "failure_count": len(evaluation["failures"]), "items": evaluation["failures"], "items_canonical_json_sha256": canonical_hash(evaluation["failures"]), "rendering_status": ( "machine_readable_examples_retained; a visual production gallery requires " "controlled access to protected imagery" ), } taxonomy_entries = sorted( {(item["error_code"], item["kind"]) for item in evaluation["failures"]} ) error_taxonomy = { "schema_version": 2, "source": "docs/accuracy-program/05-metric-framework.md section 4", "observed_codes": [ {"error_code": code, "kind": kind} for code, kind in taxonomy_entries ], "observed_failure_count": len(evaluation["failures"]), "claim_boundary": evaluation["claim_boundary"], } object_task_names = { "object_detection", "footprint_segmentation", "vector_comparison", "change_detection", "geospatial_data_validation", } object_metrics = { "schema_version": 2, "status": "fixture_contract_only", "results": [ item for item in metric_results if item["task"] in object_task_names ], } tile_metrics = { "schema_version": 2, "status": "fixture_contract_only", "results": [ item for item in metric_results if item["task"] in {"raster_classification", "terrain_interpretation"} ], } aoi_metrics = { "schema_version": 2, "status": "not_evaluable", "reason": ( "Synthetic single-case fixtures do not provide independent product AOI clusters. " "AOI micro/macro and cluster-bootstrap evidence requires the protected product corpus." ), "required_future_outputs": [ "per-AOI primary metrics", "micro and macro aggregation", "paired candidate-minus-incumbent deltas", "cluster-bootstrap confidence intervals", ], } stratified_metrics = evaluation["subgroups"] calibration_items = [] for item in metric_results: calibration = item["metrics"].get("calibration") coverage_risk = item["metrics"].get("coverage_risk") if calibration is not None or coverage_risk is not None: calibration_items.append( { "sample_id": item["sample_id"], "task": item["task"], "calibration": calibration, "coverage_risk": coverage_risk, } ) calibration_metrics = { "schema_version": 2, "status": "fixture_diagnostic_only", "selection_allowed": False, "items": calibration_items, "note": ( "Fixed diagnostic bins and risk thresholds test metric arithmetic; they do not " "select or change any operating point." ), } latency_reliability = { "schema_version": 2, "status": "not_evaluable", "model_inference_executed": False, "reason": ( "The reference harness performs deterministic evaluator arithmetic only. " "GPU latency, VRAM, throughput and failure-rate gates require the real active model." ), } human_review_summary = { "schema_version": 2, "status": product_gates["human_review_complete"]["status"], "reviewed": product_gates["human_review_complete"].get("observed"), "required": product_gates["human_review_complete"].get("required"), "source": "readiness-snapshot.json bound to Phase-1/3 evidence", "ai_review_is_human_signoff": False, } candidate_vs_incumbent = { "schema_version": 2, "status": "not_evaluable", "reason": ( "Phase 4 has no valid real incumbent product baseline and no pre-registered " "candidate; synthetic fixture values cannot define non-inferiority." ), "future_gate_contract": { "unit": "paired independent AOI", "global_and_critical_subgroups_required": True, "missing_or_insufficient_support": "not_evaluable", "aggregate_improvement_may_mask_subgroup_regression": False, "numeric_margin": "to_be_frozen_before_protected_access", }, } input_manifest_file_sha256 = hashlib.sha256(json_bytes(input_manifest)).hexdigest() benchmark_manifest = { "schema_version": 2, "benchmark_id": BENCHMARK_ID, "workflow_version": WORKFLOW_VERSION, "evaluator_version": EVALUATOR_VERSION, "split_generator_version": GENERATOR_VERSION, "repository_commit": input_manifest["repository_commit"], "input_manifest": { "path": "input-manifest.json", "sha256": input_manifest_file_sha256, }, "inputs": { "split_source": repository_file( repo_root, "fixtures/accuracy/p4/split-source-manifest.json" ), "protected_cases": repository_file( repo_root, "fixtures/accuracy/p4/protected-baseline-cases.json" ), "golden_qa_manifest": repository_file( repo_root, "fixtures/golden/golden_qa_benchmarks.json" ), "phase3_full_scan": repository_file( repo_root, "artifacts/evidence/accuracy/P3/full-scan-manifest.json" ), "phase3_leakage": repository_file( repo_root, "artifacts/evidence/accuracy/P3/leakage-report.json" ), }, "code": input_manifest["code"], "runtime": input_manifest["runtime"], "split_manifests": { "development_sha256": development["manifest_sha256"], "protected_sha256": protected["manifest_sha256"], "leakage_status": leakage["status"], }, "inference_and_selection": { "synthetic_reference_harness": True, "production_model_inference_executed": False, "test_used_for_selection": False, "background_test_used_for_selection": False, "challenge_labels_available": False, "raw_predictions_retained": True, "threshold_source": "pre_registered_configuration_only", }, "evaluation_results_canonical_json_sha256": evaluation[ "results_canonical_json_sha256" ], "reference_baseline_sha256": golden["content_sha256"], "claim_boundary": evaluation["claim_boundary"], } benchmark_manifest["manifest_sha256"] = canonical_hash(benchmark_manifest) gate_report["benchmark_manifest_sha256"] = benchmark_manifest["manifest_sha256"] workflow_summary = { "schema_version": 2, "status": gate_report["status"], "phase_decision": gate_report["phase_decision"], "local_harness_status": gate_report["local_harness_status"], "product_benchmark_status": gate_report["product_benchmark_status"], "benchmark_manifest_sha256": benchmark_manifest["manifest_sha256"], "split_counts": leakage["split_counts"], "task_family_count": evaluation["task_count"], "implemented_capability_count": len(evaluation["task_inventory"]), "case_count": evaluation["case_count"], "failure_example_count": len(evaluation["failures"]), "evaluation_results_canonical_json_sha256": evaluation[ "results_canonical_json_sha256" ], "reference_baseline_sha256": golden["content_sha256"], "promotion_allowed": False, "phase4_done": gate_report["status"] == "pass", "phase5_ready": gate_report["status"] == "pass", } artifacts: dict[str, Any] = { "acceptance-gates.json": gate_report, "aoi-metrics.json": aoi_metrics, "baseline-raw-predictions.json": raw_predictions, "benchmark-manifest.json": benchmark_manifest, "calibration-metrics.json": calibration_metrics, "candidate-vs-incumbent.json": candidate_vs_incumbent, "development-split-manifest.json": development, "error-taxonomy.json": error_taxonomy, "evaluation-contract.json": evaluation_contract, "failure-gallery.json": failure_gallery, "generation-status.json": split_result["generation_status"], "human-review-summary.json": human_review_summary, "input-manifest.json": input_manifest, "latency-and-reliability.json": latency_reliability, "leakage-gate-report.json": leakage, "metric-report.json": metric_report, "object-metrics.json": object_metrics, "protected-split-manifest.json": protected, "reference-implementation-baseline.json": golden, "release-gate-report.json": gate_report, "split-and-leakage-audit.json": leakage, "stratified-metrics.json": stratified_metrics, "tile-metrics.json": tile_metrics, "workflow-summary.json": workflow_summary, } evidence = build_evidence_manifest(artifacts, benchmark_manifest) for name, payload in sorted(artifacts.items()): write_json_immutable(output_dir / name, payload) write_json_immutable(output_dir / "evidence-manifest.json", evidence) return workflow_summary def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--repo-root", type=Path, default=ROOT) parser.add_argument( "--output-dir", type=Path, default=ROOT / "artifacts/evidence/accuracy/P4/reference-harness-v2", ) parser.add_argument( "--product-baseline-manifest", type=Path, help=( "Optional governed product incumbent manifest. It can pass only when real " "active-model inference and all referenced artifacts validate." ), ) parser.add_argument( "--allow-product-blocked", action="store_true", help=( "Return zero when the local harness passes while product evidence remains " "fail/not_evaluable. This never changes a gate or phase decision." ), ) return parser.parse_args() def main() -> int: args = parse_args() try: summary = run_workflow( args.repo_root.resolve(), args.output_dir.resolve(), args.product_baseline_manifest.resolve() if args.product_baseline_manifest else None, ) except Exception as exc: # noqa: BLE001 - workflow evidence must fail closed print( json.dumps( {"status": "fail", "error": f"{type(exc).__name__}: {exc}"}, indent=2, ) ) return 2 print(json.dumps(summary, indent=2, sort_keys=True)) if summary["status"] == "pass": return 0 return ( 0 if args.allow_product_blocked and summary["local_harness_status"] == "pass" else 2 ) if __name__ == "__main__": raise SystemExit(main())