From 116b8e291ea77d7943c7b52c20212e70c9a6081e Mon Sep 17 00:00:00 2001 From: Jens Date: Sun, 9 Aug 2026 22:10:50 +0200 Subject: [PATCH] block checkpoint evaluation on ancestral exposure --- ...09-v71-full-lineage-independence-gate.json | 76 ++++++++++++ .../20260809-v71-model-lineage-receipt.json | 115 ++++++++++++++++++ docs/CODEX_EXECUTION_LOG.md | 34 ++++++ docs/TODO.md | 5 +- ...026-08-09-ai-assisted-checkpoint-review.md | 26 ++++ scripts/README.md | 15 ++- scripts/evaluate_yolo_checkpoint_matrix.py | 105 ++++++++++------ tests/test_evaluate_yolo_checkpoint_matrix.py | 75 +++++++++--- 8 files changed, 390 insertions(+), 61 deletions(-) create mode 100644 artifacts/evidence/accuracy/model-training/20260809-v71-full-lineage-independence-gate.json create mode 100644 artifacts/evidence/accuracy/model-training/20260809-v71-model-lineage-receipt.json diff --git a/artifacts/evidence/accuracy/model-training/20260809-v71-full-lineage-independence-gate.json b/artifacts/evidence/accuracy/model-training/20260809-v71-full-lineage-independence-gate.json new file mode 100644 index 00000000..17e67a7d --- /dev/null +++ b/artifacts/evidence/accuracy/model-training/20260809-v71-full-lineage-independence-gate.json @@ -0,0 +1,76 @@ +{ + "claim_boundary": "No checkpoint ranking or release claim is permitted.", + "dataset_yaml": "/app/storage/operator-data/yolo-building-aoi1024-bgaware512r8-nonoverlap-eval-r2/dataset.yaml", + "dataset_yaml_sha256": "c36f6d76aac4bc2f5e880e2f0133d176fb0dc277600e121570a944eee6db2c48", + "generated_at": "2026-08-09T20:08:28.285906+00:00", + "gpu_inference_attempted": false, + "model_lineage_independence_evidence": { + "evaluation_samples": [ + "postel_bos", + "turnhout", + "westerlo" + ], + "evaluation_summary_path": "/app/storage/operator-data/yolo-building-aoi1024-bgaware512r8-nonoverlap-eval-r2/yolo_tile_dataset_summary.json", + "evaluation_summary_sha256": "c17a8c47eaecc2dfb38778ff5a612f9e6c4d67c9e90e31d3a6a66d8857adec4c", + "independent_for_all_supplied_lineage_corpora": false, + "interpretation": "At least one evaluation AOI was exposed in an ancestral train, validation, calibration or other split; the matrix is blocked before model loading.", + "lineage_corpora": [ + { + "exposure_roles": { + "postel_bos": [ + "train" + ], + "turnhout": [ + "val" + ], + "westerlo": [ + "val" + ] + }, + "lineage_sample_count": 19, + "lineage_summary_path": "/app/storage/operator-data/yolo-building-aoi1024-expanded-minpx4vis035/yolo_tile_dataset_summary.json", + "lineage_summary_sha256": "1887e2ba5c719d33675a9ed85db99a274fd66d98d5f6ad44558d673620917201", + "overlapping_evaluation_samples": [ + "postel_bos", + "turnhout", + "westerlo" + ] + }, + { + "exposure_roles": { + "postel_bos": [ + "train" + ] + }, + "lineage_sample_count": 22, + "lineage_summary_path": "/app/storage/operator-data/yolo-building-aoi1024-smallbld-minpx3vis035/yolo_tile_dataset_summary.json", + "lineage_summary_sha256": "49b2a07d2105d08356431757b83eafc1498eaf1fb76965b1efe05b776824942a", + "overlapping_evaluation_samples": [ + "postel_bos" + ] + }, + { + "exposure_roles": { + "postel_bos": [ + "train" + ] + }, + "lineage_sample_count": 28, + "lineage_summary_path": "/app/storage/operator-data/yolo-building-aoi1024-reviewedexp6-minpx3vis035/yolo_tile_dataset_summary.json", + "lineage_summary_sha256": "7c917e31216d1df2174c0f9c736f88a81f3835aa991971f8fb8665e17ddf5c9c", + "overlapping_evaluation_samples": [ + "postel_bos" + ] + } + ], + "overlapping_evaluation_samples": [ + "postel_bos", + "turnhout", + "westerlo" + ], + "status": "overlap" + }, + "model_loading_attempted": false, + "schema_version": 3, + "status": "blocked_model_lineage_sample_exposure" +} diff --git a/artifacts/evidence/accuracy/model-training/20260809-v71-model-lineage-receipt.json b/artifacts/evidence/accuracy/model-training/20260809-v71-model-lineage-receipt.json new file mode 100644 index 00000000..48a20bb3 --- /dev/null +++ b/artifacts/evidence/accuracy/model-training/20260809-v71-model-lineage-receipt.json @@ -0,0 +1,115 @@ +{ + "claim_boundary": "File/hash ancestry only; does not establish human review, protected-test validity or release fitness.", + "generated_at": "2026-08-09T20:10:17.734420+00:00", + "generic_base": { + "path": "/app/models/yolov8s.pt", + "sha256": "1f47a78bf100391c2a140b7ac73a1caae18c32779be7d310658112f7ac9aa78a", + "size_bytes": 22588772 + }, + "schema_version": 1, + "stages": [ + { + "args": { + "path": "/app/storage/training/operator-yolo/geointel-building-yolov8s-aoi1024expandedminpx4vis035e50/args.yaml", + "sha256": "ddb12aa459404a905c2d908210033f18346491823c1b059eec332c1a7f9a0263", + "size_bytes": 1839 + }, + "best_matches_model_copy": true, + "dataset_summary": { + "path": "/app/storage/operator-data/yolo-building-aoi1024-expanded-minpx4vis035/yolo_tile_dataset_summary.json", + "sha256": "1887e2ba5c719d33675a9ed85db99a274fd66d98d5f6ad44558d673620917201", + "size_bytes": 135720 + }, + "dataset_yaml": { + "path": "/app/storage/operator-data/yolo-building-aoi1024-expanded-minpx4vis035/dataset.yaml", + "sha256": "357730f4ee110efdb241d26c0479eeeea5200885cebeb8eae011e257beb058ba", + "size_bytes": 134 + }, + "model_copy": { + "path": "/app/models/geointel-building-yolov8s-aoi1024expandedminpx4vis035e50.pt", + "sha256": "a8a79cf5b0bdc19a0245acc322cf77232c335e222bd5f3c00a17d5f29402c196", + "size_bytes": 22499498 + }, + "name": "expanded", + "parent_model": { + "path": "/app/models/yolov8s.pt", + "sha256": "1f47a78bf100391c2a140b7ac73a1caae18c32779be7d310658112f7ac9aa78a", + "size_bytes": 22588772 + }, + "training_best": { + "path": "/app/storage/training/operator-yolo/geointel-building-yolov8s-aoi1024expandedminpx4vis035e50/weights/best.pt", + "sha256": "a8a79cf5b0bdc19a0245acc322cf77232c335e222bd5f3c00a17d5f29402c196", + "size_bytes": 22499498 + } + }, + { + "args": { + "path": "/app/storage/training/operator-yolo/geointel-building-yolov8s-smallbld-minpx3-img640-ft30/args.yaml", + "sha256": "2b482e6bbef26f433d4406e1acb5cbbf4ce63a63644b180a2d51b93f8c8f0dcb", + "size_bytes": 1882 + }, + "best_matches_model_copy": true, + "dataset_summary": { + "path": "/app/storage/operator-data/yolo-building-aoi1024-smallbld-minpx3vis035/yolo_tile_dataset_summary.json", + "sha256": "49b2a07d2105d08356431757b83eafc1498eaf1fb76965b1efe05b776824942a", + "size_bytes": 158980 + }, + "dataset_yaml": { + "path": "/app/storage/operator-data/yolo-building-aoi1024-smallbld-minpx3vis035/dataset.yaml", + "sha256": "3a2ea97c35a18072a1ab6738cd673c0ecec5344b19461c91d72a15e138d46e8d", + "size_bytes": 134 + }, + "model_copy": { + "path": "/app/models/geointel-building-yolov8s-smallbld-minpx3-img640-ft30.pt", + "sha256": "a9088b8491dfae36694b53e9e9406cb4e3511d334a5712fa34f75078a47759c1", + "size_bytes": 22516074 + }, + "name": "active", + "parent_model": { + "path": "/app/models/geointel-building-yolov8s-aoi1024expandedminpx4vis035e50.pt", + "sha256": "a8a79cf5b0bdc19a0245acc322cf77232c335e222bd5f3c00a17d5f29402c196", + "size_bytes": 22499498 + }, + "training_best": { + "path": "/app/storage/training/operator-yolo/geointel-building-yolov8s-smallbld-minpx3-img640-ft30/weights/best.pt", + "sha256": "a9088b8491dfae36694b53e9e9406cb4e3511d334a5712fa34f75078a47759c1", + "size_bytes": 22516074 + } + }, + { + "args": { + "path": "/app/storage/training/operator-yolo/geointel-building-yolov8s-reviewedexp6-minpx3-img640-ft20/args.yaml", + "sha256": "87e07dbed3ad12cc474e6b895147839431805b7044496277242e1538fcd91a0b", + "size_bytes": 1891 + }, + "best_matches_model_copy": true, + "dataset_summary": { + "path": "/app/storage/operator-data/yolo-building-aoi1024-reviewedexp6-minpx3vis035/yolo_tile_dataset_summary.json", + "sha256": "7c917e31216d1df2174c0f9c736f88a81f3835aa991971f8fb8665e17ddf5c9c", + "size_bytes": 204029 + }, + "dataset_yaml": { + "path": "/app/storage/operator-data/yolo-building-aoi1024-reviewedexp6-minpx3vis035/dataset.yaml", + "sha256": "62e12f4433508cb0640f1d90ec0ffd389cd21ebe655fac4e7d9402f04da1168e", + "size_bytes": 138 + }, + "model_copy": { + "path": "/app/models/geointel-building-yolov8s-reviewedexp6-minpx3-img640-ft20.pt", + "sha256": "038f1f97a6afd534f29e1f392a730a58207b928ca01e31ab8d8fed6106705820", + "size_bytes": 22514794 + }, + "name": "challenger", + "parent_model": { + "path": "/app/models/geointel-building-yolov8s-smallbld-minpx3-img640-ft30.pt", + "sha256": "a9088b8491dfae36694b53e9e9406cb4e3511d334a5712fa34f75078a47759c1", + "size_bytes": 22516074 + }, + "training_best": { + "path": "/app/storage/training/operator-yolo/geointel-building-yolov8s-reviewedexp6-minpx3-img640-ft20/weights/best.pt", + "sha256": "038f1f97a6afd534f29e1f392a730a58207b928ca01e31ab8d8fed6106705820", + "size_bytes": 22514794 + } + } + ], + "status": "complete_retained_lineage_to_generic_base" +} diff --git a/docs/CODEX_EXECUTION_LOG.md b/docs/CODEX_EXECUTION_LOG.md index a0ba60bb..e16b8599 100644 --- a/docs/CODEX_EXECUTION_LOG.md +++ b/docs/CODEX_EXECUTION_LOG.md @@ -12737,3 +12737,37 @@ Open: - Independent real-background AOIs are still required. Literal 100% model correctness is not demonstrated; the system now refuses to misrepresent the known overlap as proof. + +## 2026-08-09 - Full ancestral-corpus exposure correction + +### Corrected + +- Reconstructed the retained checkpoint ancestry from exact Ultralytics + `args.yaml` files and model hashes. The active model descends from the + expanded-min4 corpus; the challenger descends from the active model. +- Found that the expanded ancestor used Postel-bos for training and + Turnhout/Westerlo for validation. All three v69 evaluation AOIs were therefore + exposed somewhere in the model family. +- Replaced last-corpus/train-only checking with fail-closed inspection of every + split in every supplied ancestral corpus. Legacy CLI spellings remain safe + aliases to the full-lineage behavior. + +### Verified + +- The Tower gate blocked all three exposed AOIs with zero model-loading and + zero GPU-inference attempts. +- Expanded ancestor summary SHA-256: + `1887e2ba5c719d33675a9ed85db99a274fd66d98d5f6ad44558d673620917201`. +- Blocked-manifest SHA-256: + `f6effd94befd6df8d699bc66fbbea8a860b056cff9996be5f867f858807af65e`. +- Model-lineage receipt SHA-256: + `84d8d7bd95acfb07b188b07ececfc5417fca80c087d7c04442dd2a4da5f43936`; + every retained training `best.pt` matches its named model copy byte-for-byte. +- The production checkpoint and container image remain unchanged. + +### Accuracy boundary + +- The old Turnhout/Westerlo/Postel numbers are regression diagnostics only. + No independent positive or background accuracy evidence currently exists for + this model family; new AOIs must be provisioned without ancestry or spatial + exposure before another ranking is defensible. diff --git a/docs/TODO.md b/docs/TODO.md index 1f007557..179d0fd4 100644 --- a/docs/TODO.md +++ b/docs/TODO.md @@ -1179,8 +1179,9 @@ This file now starts with the current implementation status. Older preparation/b as release evidence. Postel occurs in both compared train splits and is only a training-seen regression check, not independent validation. - [x] Add a fail-closed checkpoint gate that checks every evaluation AOI - against the exact train split of every supplied model corpus before PyTorch - import, model loading or GPU inference. + against every split of every supplied ancestral model corpus before PyTorch + import, model loading or GPU inference. Turnhout/Westerlo were exposed as + validation in the base building model; Postel was exposed as training. - [ ] Convert the AI-assisted ledger into no stronger claim than experimental triage; a real human must independently review and sign the frozen artifacts before the governed training wrapper may unlock. diff --git a/docs/reviews/2026-08-09-ai-assisted-checkpoint-review.md b/docs/reviews/2026-08-09-ai-assisted-checkpoint-review.md index 2350533c..3d290e85 100644 --- a/docs/reviews/2026-08-09-ai-assisted-checkpoint-review.md +++ b/docs/reviews/2026-08-09-ai-assisted-checkpoint-review.md @@ -237,3 +237,29 @@ mode, fails before PyTorch import, model loading or GPU inference when any evaluation AOI overlaps any supplied train split. The reproduced gate blocked on `postel_bos` for both checkpoints. Its immutable machine-readable record is `artifacts/evidence/accuracy/model-training/20260809-v70-evaluation-independence-gate.json`. + +## Full model-lineage correction + +The preceding correction still considered only the final fine-tune corpus of +each checkpoint. Exact retained Ultralytics arguments establish a longer +ancestry: the active checkpoint was initialized from +`geointel-building-yolov8s-aoi1024expandedminpx4vis035e50`, which was initialized +from the generic `yolov8s.pt`; the challenger was then initialized from the +active checkpoint. The copied model assets and retained `best.pt` files match +byte-for-byte at each building-model stage. + +The ancestral expanded corpus exposes all three evaluation AOIs: `postel_bos` +as train, and `turnhout` plus `westerlo` as validation. Therefore none of the +v69 AOIs is independent of the complete model family. Turnhout/Westerlo can +still be used as familiar regression diagnostics, but their metrics are not a +fresh candidate-ranking result and must not support accuracy, uncertainty, +generalisation or release claims. + +The gate now requires every ancestral corpus summary and checks every recorded +split, including validation and calibration. A reproduced Tower run blocked on +all three AOIs before PyTorch import, model loading or GPU inference. Evidence: +`artifacts/evidence/accuracy/model-training/20260809-v71-full-lineage-independence-gate.json`. +The byte-matching parent/output chain is retained separately in +`artifacts/evidence/accuracy/model-training/20260809-v71-model-lineage-receipt.json`. +Fresh geographically separated AOIs with complete spatial and lineage checks +are required for the next meaningful evaluation. diff --git a/scripts/README.md b/scripts/README.md index 24147923..ee8e90e3 100644 --- a/scripts/README.md +++ b/scripts/README.md @@ -5,13 +5,16 @@ weights on one declared, non-protected `val` split using CUDA. It records exact model and dataset hashes, standard Ultralytics detection metrics and a separate pure-background detection count. The output claim is validation ranking only; the script neither reads protected test data nor promotes a model. -Governed comparisons must pass every candidate's exact tile-summary with -`--training-summary` and enable `--require-training-sample-independence`. If -any validation AOI occurs in any supplied train split, the script writes a +Governed comparisons must pass the exact tile summary for every corpus in the +complete ancestry of every candidate with `--lineage-summary` and enable +`--require-lineage-sample-independence`. The check covers train, validation, +calibration and every other recorded split, because an AOI used for ancestral +checkpoint or threshold selection is also exposed. Any exposure writes a blocked manifest and exits before importing PyTorch, loading a model or using -the GPU. A background detection count from a training-seen AOI is only a -regression check and must never be presented as independent background -evidence. +the GPU. The older `--training-summary` and +`--require-training-sample-independence` spellings remain aliases, but now use +the same safer full-split semantics. A result from any lineage-exposed AOI is +only a regression check and must never be presented as independent evidence. `render_operator_yolo_label_qa_contact_sheets.py` paginates complete visual reviews with `--tiles-per-sheet` (default `64`). This keeps large corpora diff --git a/scripts/evaluate_yolo_checkpoint_matrix.py b/scripts/evaluate_yolo_checkpoint_matrix.py index a69baa97..6afa1d7c 100644 --- a/scripts/evaluate_yolo_checkpoint_matrix.py +++ b/scripts/evaluate_yolo_checkpoint_matrix.py @@ -111,8 +111,8 @@ def _sample_slugs(payload: dict[str, Any], *, split: str | None) -> set[str]: return samples -def training_sample_independence_evidence( - dataset_yaml: Path, training_summaries: list[Path] +def model_lineage_independence_evidence( + dataset_yaml: Path, lineage_summaries: list[Path] ) -> dict[str, Any]: evaluation_summary = dataset_yaml.parent / "yolo_tile_dataset_summary.json" if not evaluation_summary.is_file(): @@ -120,7 +120,7 @@ def training_sample_independence_evidence( "status": "unavailable", "reason": "evaluation dataset summary is unavailable", "evaluation_summary_path": str(evaluation_summary), - "independent_for_all_supplied_training_corpora": False, + "independent_for_all_supplied_lineage_corpora": False, } try: @@ -134,18 +134,23 @@ def training_sample_independence_evidence( "reason": str(exc), "evaluation_summary_path": str(evaluation_summary), "evaluation_summary_sha256": sha256_file(evaluation_summary), - "independent_for_all_supplied_training_corpora": False, + "independent_for_all_supplied_lineage_corpora": False, } rows: list[dict[str, Any]] = [] union_overlap: set[str] = set() - for raw_summary in training_summaries: + for raw_summary in lineage_summaries: summary = raw_summary.expanduser().resolve(strict=True) try: payload = json.loads(summary.read_text(encoding="utf-8")) - training_samples = _sample_slugs(payload, split="train") - if not training_samples: - raise ValueError("training summary contains no training samples") + all_samples = _sample_slugs(payload, split=None) + if not all_samples: + raise ValueError("lineage summary contains no samples") + roles_by_sample: dict[str, set[str]] = {} + for tile in payload["tiles"]: + sample_slug = str(tile["sample_slug"]).strip() + split = str(tile.get("split") or "unknown").strip() + roles_by_sample.setdefault(sample_slug, set()).add(split) except (json.JSONDecodeError, OSError, ValueError) as exc: return { "status": "invalid", @@ -153,32 +158,35 @@ def training_sample_independence_evidence( "evaluation_summary_path": str(evaluation_summary), "evaluation_summary_sha256": sha256_file(evaluation_summary), "evaluation_samples": sorted(evaluation_samples), - "independent_for_all_supplied_training_corpora": False, + "independent_for_all_supplied_lineage_corpora": False, } - overlap = evaluation_samples & training_samples + overlap = evaluation_samples & all_samples union_overlap.update(overlap) rows.append( { - "training_summary_path": str(summary), - "training_summary_sha256": sha256_file(summary), - "training_sample_count": len(training_samples), + "lineage_summary_path": str(summary), + "lineage_summary_sha256": sha256_file(summary), + "lineage_sample_count": len(all_samples), "overlapping_evaluation_samples": sorted(overlap), + "exposure_roles": { + sample: sorted(roles_by_sample[sample]) for sample in sorted(overlap) + }, } ) if not rows: return { "status": "unavailable", - "reason": "no training summaries were supplied", + "reason": "no complete model-lineage summaries were supplied", "evaluation_summary_path": str(evaluation_summary), "evaluation_summary_sha256": sha256_file(evaluation_summary), "evaluation_samples": sorted(evaluation_samples), - "training_corpora": [], + "lineage_corpora": [], "overlapping_evaluation_samples": [], - "independent_for_all_supplied_training_corpora": False, + "independent_for_all_supplied_lineage_corpora": False, "interpretation": ( - "Training/evaluation independence cannot be established " - "without exact training summaries." + "Model-lineage/evaluation independence cannot be established " + "without every ancestral corpus summary." ), } @@ -188,29 +196,37 @@ def training_sample_independence_evidence( "evaluation_summary_path": str(evaluation_summary), "evaluation_summary_sha256": sha256_file(evaluation_summary), "evaluation_samples": sorted(evaluation_samples), - "training_corpora": rows, + "lineage_corpora": rows, "overlapping_evaluation_samples": sorted(union_overlap), - "independent_for_all_supplied_training_corpora": independent, + "independent_for_all_supplied_lineage_corpora": independent, "interpretation": ( - "No evaluation AOI occurs in the train split of any supplied corpus." + "No evaluation AOI occurs in any split of any supplied ancestral corpus." if independent - else "At least one evaluation AOI occurs in a supplied train split; " - "the matrix is blocked before model loading." + else "At least one evaluation AOI was exposed in an ancestral train, " + "validation, calibration or other split; the matrix is blocked before " + "model loading." ), } +def training_sample_independence_evidence( + dataset_yaml: Path, training_summaries: list[Path] +) -> dict[str, Any]: + """Backward-compatible alias; evidence now checks every lineage split.""" + return model_lineage_independence_evidence(dataset_yaml, training_summaries) + + def write_blocked_manifest( output: Path, dataset_yaml: Path, evidence: dict[str, Any] ) -> None: payload = { - "schema_version": 2, + "schema_version": 3, "generated_at": datetime.now(UTC).isoformat(), - "status": "blocked_training_sample_overlap", + "status": "blocked_model_lineage_sample_exposure", "claim_boundary": "No checkpoint ranking or release claim is permitted.", "dataset_yaml": str(dataset_yaml), "dataset_yaml_sha256": sha256_file(dataset_yaml), - "training_sample_independence_evidence": evidence, + "model_lineage_independence_evidence": evidence, "model_loading_attempted": False, "gpu_inference_attempted": False, } @@ -261,8 +277,21 @@ def main() -> int: parser.add_argument("--output", type=Path, required=True) parser.add_argument("--background-prefix", action="append", required=True) parser.add_argument("--background-confidence", type=float, default=0.15) - parser.add_argument("--training-summary", type=Path, action="append", default=[]) - parser.add_argument("--require-training-sample-independence", action="store_true") + parser.add_argument( + "--lineage-summary", + "--training-summary", + dest="lineage_summary", + type=Path, + action="append", + default=[], + help="tile summary for every corpus in the complete model ancestry", + ) + parser.add_argument( + "--require-lineage-sample-independence", + "--require-training-sample-independence", + dest="require_lineage_sample_independence", + action="store_true", + ) parser.add_argument("--device", default="cuda:0") parser.add_argument("--imgsz", type=int, default=640) parser.add_argument("--batch", type=int, default=8) @@ -275,17 +304,19 @@ def main() -> int: dataset_yaml = args.dataset_yaml.expanduser().resolve(strict=True) images = validation_images(dataset_yaml) overlap_evidence = dataset_overlap_evidence(dataset_yaml) - independence_evidence = training_sample_independence_evidence( - dataset_yaml, args.training_summary + independence_evidence = model_lineage_independence_evidence( + dataset_yaml, args.lineage_summary ) - if args.require_training_sample_independence and not args.training_summary: + if args.require_lineage_sample_independence and not args.lineage_summary: parser.error( - "--require-training-sample-independence requires at least one " - "--training-summary" + "--require-lineage-sample-independence requires every ancestral " + "--lineage-summary" ) if ( - args.require_training_sample_independence - and not independence_evidence["independent_for_all_supplied_training_corpora"] + args.require_lineage_sample_independence + and not independence_evidence[ + "independent_for_all_supplied_lineage_corpora" + ] ): write_blocked_manifest(args.output, dataset_yaml, independence_evidence) return 3 @@ -374,14 +405,14 @@ def main() -> int: reverse=True, ) payload = { - "schema_version": 2, + "schema_version": 3, "generated_at": datetime.now(UTC).isoformat(), "status": "ok" if successful else "failed", "claim_boundary": "Non-protected validation ranking only; no test, challenge or promotion claim.", "dataset_yaml": str(dataset_yaml), "dataset_yaml_sha256": sha256_file(dataset_yaml), "dataset_overlap_evidence": overlap_evidence, - "training_sample_independence_evidence": independence_evidence, + "model_lineage_independence_evidence": independence_evidence, "validation_image_count": len(images), "pure_background_prefixes": list(prefixes), "pure_background_confidence": args.background_confidence, diff --git a/tests/test_evaluate_yolo_checkpoint_matrix.py b/tests/test_evaluate_yolo_checkpoint_matrix.py index 136c3b51..89b13eea 100644 --- a/tests/test_evaluate_yolo_checkpoint_matrix.py +++ b/tests/test_evaluate_yolo_checkpoint_matrix.py @@ -1,9 +1,11 @@ import json +import sys from pathlib import Path from scripts.evaluate_yolo_checkpoint_matrix import ( dataset_overlap_evidence, - training_sample_independence_evidence, + main, + model_lineage_independence_evidence, write_blocked_manifest, ) @@ -67,7 +69,7 @@ def _write_summary(path: Path, rows: list[tuple[str, str]]) -> None: ) -def test_training_sample_independence_detects_overlap_for_each_corpus( +def test_model_lineage_independence_detects_exposure_in_every_split( tmp_path: Path, ) -> None: evaluation = tmp_path / "evaluation" @@ -80,24 +82,25 @@ def test_training_sample_independence_detects_overlap_for_each_corpus( ) active = tmp_path / "active.json" challenger = tmp_path / "challenger.json" - _write_summary(active, [("postel_bos", "train"), ("mol", "train")]) + _write_summary(active, [("postel_bos", "train"), ("turnhout", "val")]) _write_summary( challenger, [("postel_bos", "train"), ("dessel", "train")] ) - evidence = training_sample_independence_evidence( + evidence = model_lineage_independence_evidence( dataset_yaml, [active, challenger] ) assert evidence["status"] == "overlap" - assert evidence["overlapping_evaluation_samples"] == ["postel_bos"] - assert evidence["independent_for_all_supplied_training_corpora"] is False + assert evidence["overlapping_evaluation_samples"] == ["postel_bos", "turnhout"] + assert evidence["independent_for_all_supplied_lineage_corpora"] is False assert [ - row["overlapping_evaluation_samples"] for row in evidence["training_corpora"] - ] == [["postel_bos"], ["postel_bos"]] + row["overlapping_evaluation_samples"] for row in evidence["lineage_corpora"] + ] == [["postel_bos", "turnhout"], ["postel_bos"]] + assert evidence["lineage_corpora"][0]["exposure_roles"]["turnhout"] == ["val"] -def test_training_sample_independence_accepts_disjoint_samples(tmp_path: Path) -> None: +def test_model_lineage_independence_accepts_disjoint_samples(tmp_path: Path) -> None: evaluation = tmp_path / "evaluation" evaluation.mkdir() dataset_yaml = evaluation / "dataset.yaml" @@ -108,14 +111,14 @@ def test_training_sample_independence_accepts_disjoint_samples(tmp_path: Path) - training = tmp_path / "training.json" _write_summary(training, [("mol", "train")]) - evidence = training_sample_independence_evidence(dataset_yaml, [training]) + evidence = model_lineage_independence_evidence(dataset_yaml, [training]) assert evidence["status"] == "independent" assert evidence["overlapping_evaluation_samples"] == [] - assert evidence["independent_for_all_supplied_training_corpora"] is True + assert evidence["independent_for_all_supplied_lineage_corpora"] is True -def test_training_sample_independence_is_unavailable_without_training_corpus( +def test_model_lineage_independence_is_unavailable_without_ancestral_corpus( tmp_path: Path, ) -> None: evaluation = tmp_path / "evaluation" @@ -126,11 +129,11 @@ def test_training_sample_independence_is_unavailable_without_training_corpus( evaluation / "yolo_tile_dataset_summary.json", [("turnhout", "val")] ) - evidence = training_sample_independence_evidence(dataset_yaml, []) + evidence = model_lineage_independence_evidence(dataset_yaml, []) assert evidence["status"] == "unavailable" assert evidence["overlapping_evaluation_samples"] == [] - assert evidence["independent_for_all_supplied_training_corpora"] is False + assert evidence["independent_for_all_supplied_lineage_corpora"] is False def test_blocked_manifest_records_that_models_and_gpu_were_not_used( @@ -145,11 +148,51 @@ def test_blocked_manifest_records_that_models_and_gpu_were_not_used( dataset_yaml, { "status": "overlap", - "independent_for_all_supplied_training_corpora": False, + "independent_for_all_supplied_lineage_corpora": False, }, ) payload = json.loads(output.read_text(encoding="utf-8")) - assert payload["status"] == "blocked_training_sample_overlap" + assert payload["status"] == "blocked_model_lineage_sample_exposure" assert payload["model_loading_attempted"] is False assert payload["gpu_inference_attempted"] is False + + +def test_cli_blocks_lineage_validation_exposure_before_model_resolution( + tmp_path: Path, monkeypatch +) -> None: + evaluation = tmp_path / "evaluation" + images = evaluation / "images" / "val" + images.mkdir(parents=True) + (images / "turnhout_r0_c0.png").write_bytes(b"not opened before gate") + dataset_yaml = evaluation / "dataset.yaml" + dataset_yaml.write_text("path: .\nval: images/val\n", encoding="utf-8") + _write_summary( + evaluation / "yolo_tile_dataset_summary.json", [("turnhout", "val")] + ) + ancestor = tmp_path / "ancestor.json" + _write_summary(ancestor, [("turnhout", "val"), ("mol", "train")]) + output = tmp_path / "blocked.json" + monkeypatch.setattr( + sys, + "argv", + [ + "evaluate_yolo_checkpoint_matrix.py", + "--dataset-yaml", + str(dataset_yaml), + "--model", + str(tmp_path / "model-is-never-resolved.pt"), + "--output", + str(output), + "--background-prefix", + "background", + "--lineage-summary", + str(ancestor), + "--require-lineage-sample-independence", + ], + ) + + assert main() == 3 + payload = json.loads(output.read_text(encoding="utf-8")) + assert payload["status"] == "blocked_model_lineage_sample_exposure" + assert payload["model_loading_attempted"] is False