block checkpoint evaluation on ancestral exposure
GeoIntel release gates / Compile, test, contracts and builds (push) Canceled after 0s
GeoIntel release gates / Python and npm vulnerability policy (push) Canceled after 0s
GeoIntel release gates / GIS image, SBOM and container scan (push) Canceled after 0s

This commit is contained in:
Jens
2026-08-09 22:10:50 +02:00
parent fd45f37a38
commit 116b8e291e
8 changed files with 390 additions and 61 deletions
@@ -0,0 +1,76 @@
{
"claim_boundary": "No checkpoint ranking or release claim is permitted.",
"dataset_yaml": "/app/storage/operator-data/yolo-building-aoi1024-bgaware512r8-nonoverlap-eval-r2/dataset.yaml",
"dataset_yaml_sha256": "c36f6d76aac4bc2f5e880e2f0133d176fb0dc277600e121570a944eee6db2c48",
"generated_at": "2026-08-09T20:08:28.285906+00:00",
"gpu_inference_attempted": false,
"model_lineage_independence_evidence": {
"evaluation_samples": [
"postel_bos",
"turnhout",
"westerlo"
],
"evaluation_summary_path": "/app/storage/operator-data/yolo-building-aoi1024-bgaware512r8-nonoverlap-eval-r2/yolo_tile_dataset_summary.json",
"evaluation_summary_sha256": "c17a8c47eaecc2dfb38778ff5a612f9e6c4d67c9e90e31d3a6a66d8857adec4c",
"independent_for_all_supplied_lineage_corpora": false,
"interpretation": "At least one evaluation AOI was exposed in an ancestral train, validation, calibration or other split; the matrix is blocked before model loading.",
"lineage_corpora": [
{
"exposure_roles": {
"postel_bos": [
"train"
],
"turnhout": [
"val"
],
"westerlo": [
"val"
]
},
"lineage_sample_count": 19,
"lineage_summary_path": "/app/storage/operator-data/yolo-building-aoi1024-expanded-minpx4vis035/yolo_tile_dataset_summary.json",
"lineage_summary_sha256": "1887e2ba5c719d33675a9ed85db99a274fd66d98d5f6ad44558d673620917201",
"overlapping_evaluation_samples": [
"postel_bos",
"turnhout",
"westerlo"
]
},
{
"exposure_roles": {
"postel_bos": [
"train"
]
},
"lineage_sample_count": 22,
"lineage_summary_path": "/app/storage/operator-data/yolo-building-aoi1024-smallbld-minpx3vis035/yolo_tile_dataset_summary.json",
"lineage_summary_sha256": "49b2a07d2105d08356431757b83eafc1498eaf1fb76965b1efe05b776824942a",
"overlapping_evaluation_samples": [
"postel_bos"
]
},
{
"exposure_roles": {
"postel_bos": [
"train"
]
},
"lineage_sample_count": 28,
"lineage_summary_path": "/app/storage/operator-data/yolo-building-aoi1024-reviewedexp6-minpx3vis035/yolo_tile_dataset_summary.json",
"lineage_summary_sha256": "7c917e31216d1df2174c0f9c736f88a81f3835aa991971f8fb8665e17ddf5c9c",
"overlapping_evaluation_samples": [
"postel_bos"
]
}
],
"overlapping_evaluation_samples": [
"postel_bos",
"turnhout",
"westerlo"
],
"status": "overlap"
},
"model_loading_attempted": false,
"schema_version": 3,
"status": "blocked_model_lineage_sample_exposure"
}
@@ -0,0 +1,115 @@
{
"claim_boundary": "File/hash ancestry only; does not establish human review, protected-test validity or release fitness.",
"generated_at": "2026-08-09T20:10:17.734420+00:00",
"generic_base": {
"path": "/app/models/yolov8s.pt",
"sha256": "1f47a78bf100391c2a140b7ac73a1caae18c32779be7d310658112f7ac9aa78a",
"size_bytes": 22588772
},
"schema_version": 1,
"stages": [
{
"args": {
"path": "/app/storage/training/operator-yolo/geointel-building-yolov8s-aoi1024expandedminpx4vis035e50/args.yaml",
"sha256": "ddb12aa459404a905c2d908210033f18346491823c1b059eec332c1a7f9a0263",
"size_bytes": 1839
},
"best_matches_model_copy": true,
"dataset_summary": {
"path": "/app/storage/operator-data/yolo-building-aoi1024-expanded-minpx4vis035/yolo_tile_dataset_summary.json",
"sha256": "1887e2ba5c719d33675a9ed85db99a274fd66d98d5f6ad44558d673620917201",
"size_bytes": 135720
},
"dataset_yaml": {
"path": "/app/storage/operator-data/yolo-building-aoi1024-expanded-minpx4vis035/dataset.yaml",
"sha256": "357730f4ee110efdb241d26c0479eeeea5200885cebeb8eae011e257beb058ba",
"size_bytes": 134
},
"model_copy": {
"path": "/app/models/geointel-building-yolov8s-aoi1024expandedminpx4vis035e50.pt",
"sha256": "a8a79cf5b0bdc19a0245acc322cf77232c335e222bd5f3c00a17d5f29402c196",
"size_bytes": 22499498
},
"name": "expanded",
"parent_model": {
"path": "/app/models/yolov8s.pt",
"sha256": "1f47a78bf100391c2a140b7ac73a1caae18c32779be7d310658112f7ac9aa78a",
"size_bytes": 22588772
},
"training_best": {
"path": "/app/storage/training/operator-yolo/geointel-building-yolov8s-aoi1024expandedminpx4vis035e50/weights/best.pt",
"sha256": "a8a79cf5b0bdc19a0245acc322cf77232c335e222bd5f3c00a17d5f29402c196",
"size_bytes": 22499498
}
},
{
"args": {
"path": "/app/storage/training/operator-yolo/geointel-building-yolov8s-smallbld-minpx3-img640-ft30/args.yaml",
"sha256": "2b482e6bbef26f433d4406e1acb5cbbf4ce63a63644b180a2d51b93f8c8f0dcb",
"size_bytes": 1882
},
"best_matches_model_copy": true,
"dataset_summary": {
"path": "/app/storage/operator-data/yolo-building-aoi1024-smallbld-minpx3vis035/yolo_tile_dataset_summary.json",
"sha256": "49b2a07d2105d08356431757b83eafc1498eaf1fb76965b1efe05b776824942a",
"size_bytes": 158980
},
"dataset_yaml": {
"path": "/app/storage/operator-data/yolo-building-aoi1024-smallbld-minpx3vis035/dataset.yaml",
"sha256": "3a2ea97c35a18072a1ab6738cd673c0ecec5344b19461c91d72a15e138d46e8d",
"size_bytes": 134
},
"model_copy": {
"path": "/app/models/geointel-building-yolov8s-smallbld-minpx3-img640-ft30.pt",
"sha256": "a9088b8491dfae36694b53e9e9406cb4e3511d334a5712fa34f75078a47759c1",
"size_bytes": 22516074
},
"name": "active",
"parent_model": {
"path": "/app/models/geointel-building-yolov8s-aoi1024expandedminpx4vis035e50.pt",
"sha256": "a8a79cf5b0bdc19a0245acc322cf77232c335e222bd5f3c00a17d5f29402c196",
"size_bytes": 22499498
},
"training_best": {
"path": "/app/storage/training/operator-yolo/geointel-building-yolov8s-smallbld-minpx3-img640-ft30/weights/best.pt",
"sha256": "a9088b8491dfae36694b53e9e9406cb4e3511d334a5712fa34f75078a47759c1",
"size_bytes": 22516074
}
},
{
"args": {
"path": "/app/storage/training/operator-yolo/geointel-building-yolov8s-reviewedexp6-minpx3-img640-ft20/args.yaml",
"sha256": "87e07dbed3ad12cc474e6b895147839431805b7044496277242e1538fcd91a0b",
"size_bytes": 1891
},
"best_matches_model_copy": true,
"dataset_summary": {
"path": "/app/storage/operator-data/yolo-building-aoi1024-reviewedexp6-minpx3vis035/yolo_tile_dataset_summary.json",
"sha256": "7c917e31216d1df2174c0f9c736f88a81f3835aa991971f8fb8665e17ddf5c9c",
"size_bytes": 204029
},
"dataset_yaml": {
"path": "/app/storage/operator-data/yolo-building-aoi1024-reviewedexp6-minpx3vis035/dataset.yaml",
"sha256": "62e12f4433508cb0640f1d90ec0ffd389cd21ebe655fac4e7d9402f04da1168e",
"size_bytes": 138
},
"model_copy": {
"path": "/app/models/geointel-building-yolov8s-reviewedexp6-minpx3-img640-ft20.pt",
"sha256": "038f1f97a6afd534f29e1f392a730a58207b928ca01e31ab8d8fed6106705820",
"size_bytes": 22514794
},
"name": "challenger",
"parent_model": {
"path": "/app/models/geointel-building-yolov8s-smallbld-minpx3-img640-ft30.pt",
"sha256": "a9088b8491dfae36694b53e9e9406cb4e3511d334a5712fa34f75078a47759c1",
"size_bytes": 22516074
},
"training_best": {
"path": "/app/storage/training/operator-yolo/geointel-building-yolov8s-reviewedexp6-minpx3-img640-ft20/weights/best.pt",
"sha256": "038f1f97a6afd534f29e1f392a730a58207b928ca01e31ab8d8fed6106705820",
"size_bytes": 22514794
}
}
],
"status": "complete_retained_lineage_to_generic_base"
}
+34
View File
@@ -12737,3 +12737,37 @@ Open:
- Independent real-background AOIs are still required. Literal 100% model - Independent real-background AOIs are still required. Literal 100% model
correctness is not demonstrated; the system now refuses to misrepresent the correctness is not demonstrated; the system now refuses to misrepresent the
known overlap as proof. known overlap as proof.
## 2026-08-09 - Full ancestral-corpus exposure correction
### Corrected
- Reconstructed the retained checkpoint ancestry from exact Ultralytics
`args.yaml` files and model hashes. The active model descends from the
expanded-min4 corpus; the challenger descends from the active model.
- Found that the expanded ancestor used Postel-bos for training and
Turnhout/Westerlo for validation. All three v69 evaluation AOIs were therefore
exposed somewhere in the model family.
- Replaced last-corpus/train-only checking with fail-closed inspection of every
split in every supplied ancestral corpus. Legacy CLI spellings remain safe
aliases to the full-lineage behavior.
### Verified
- The Tower gate blocked all three exposed AOIs with zero model-loading and
zero GPU-inference attempts.
- Expanded ancestor summary SHA-256:
`1887e2ba5c719d33675a9ed85db99a274fd66d98d5f6ad44558d673620917201`.
- Blocked-manifest SHA-256:
`f6effd94befd6df8d699bc66fbbea8a860b056cff9996be5f867f858807af65e`.
- Model-lineage receipt SHA-256:
`84d8d7bd95acfb07b188b07ececfc5417fca80c087d7c04442dd2a4da5f43936`;
every retained training `best.pt` matches its named model copy byte-for-byte.
- The production checkpoint and container image remain unchanged.
### Accuracy boundary
- The old Turnhout/Westerlo/Postel numbers are regression diagnostics only.
No independent positive or background accuracy evidence currently exists for
this model family; new AOIs must be provisioned without ancestry or spatial
exposure before another ranking is defensible.
+3 -2
View File
@@ -1179,8 +1179,9 @@ This file now starts with the current implementation status. Older preparation/b
as release evidence. Postel occurs in both compared train splits and is only as release evidence. Postel occurs in both compared train splits and is only
a training-seen regression check, not independent validation. a training-seen regression check, not independent validation.
- [x] Add a fail-closed checkpoint gate that checks every evaluation AOI - [x] Add a fail-closed checkpoint gate that checks every evaluation AOI
against the exact train split of every supplied model corpus before PyTorch against every split of every supplied ancestral model corpus before PyTorch
import, model loading or GPU inference. import, model loading or GPU inference. Turnhout/Westerlo were exposed as
validation in the base building model; Postel was exposed as training.
- [ ] Convert the AI-assisted ledger into no stronger claim than experimental - [ ] Convert the AI-assisted ledger into no stronger claim than experimental
triage; a real human must independently review and sign the frozen artifacts triage; a real human must independently review and sign the frozen artifacts
before the governed training wrapper may unlock. before the governed training wrapper may unlock.
@@ -237,3 +237,29 @@ mode, fails before PyTorch import, model loading or GPU inference when any
evaluation AOI overlaps any supplied train split. The reproduced gate blocked evaluation AOI overlaps any supplied train split. The reproduced gate blocked
on `postel_bos` for both checkpoints. Its immutable machine-readable record is on `postel_bos` for both checkpoints. Its immutable machine-readable record is
`artifacts/evidence/accuracy/model-training/20260809-v70-evaluation-independence-gate.json`. `artifacts/evidence/accuracy/model-training/20260809-v70-evaluation-independence-gate.json`.
## Full model-lineage correction
The preceding correction still considered only the final fine-tune corpus of
each checkpoint. Exact retained Ultralytics arguments establish a longer
ancestry: the active checkpoint was initialized from
`geointel-building-yolov8s-aoi1024expandedminpx4vis035e50`, which was initialized
from the generic `yolov8s.pt`; the challenger was then initialized from the
active checkpoint. The copied model assets and retained `best.pt` files match
byte-for-byte at each building-model stage.
The ancestral expanded corpus exposes all three evaluation AOIs: `postel_bos`
as train, and `turnhout` plus `westerlo` as validation. Therefore none of the
v69 AOIs is independent of the complete model family. Turnhout/Westerlo can
still be used as familiar regression diagnostics, but their metrics are not a
fresh candidate-ranking result and must not support accuracy, uncertainty,
generalisation or release claims.
The gate now requires every ancestral corpus summary and checks every recorded
split, including validation and calibration. A reproduced Tower run blocked on
all three AOIs before PyTorch import, model loading or GPU inference. Evidence:
`artifacts/evidence/accuracy/model-training/20260809-v71-full-lineage-independence-gate.json`.
The byte-matching parent/output chain is retained separately in
`artifacts/evidence/accuracy/model-training/20260809-v71-model-lineage-receipt.json`.
Fresh geographically separated AOIs with complete spatial and lineage checks
are required for the next meaningful evaluation.
+9 -6
View File
@@ -5,13 +5,16 @@ weights on one declared, non-protected `val` split using CUDA. It records exact
model and dataset hashes, standard Ultralytics detection metrics and a separate model and dataset hashes, standard Ultralytics detection metrics and a separate
pure-background detection count. The output claim is validation ranking only; pure-background detection count. The output claim is validation ranking only;
the script neither reads protected test data nor promotes a model. the script neither reads protected test data nor promotes a model.
Governed comparisons must pass every candidate's exact tile-summary with Governed comparisons must pass the exact tile summary for every corpus in the
`--training-summary` and enable `--require-training-sample-independence`. If complete ancestry of every candidate with `--lineage-summary` and enable
any validation AOI occurs in any supplied train split, the script writes a `--require-lineage-sample-independence`. The check covers train, validation,
calibration and every other recorded split, because an AOI used for ancestral
checkpoint or threshold selection is also exposed. Any exposure writes a
blocked manifest and exits before importing PyTorch, loading a model or using blocked manifest and exits before importing PyTorch, loading a model or using
the GPU. A background detection count from a training-seen AOI is only a the GPU. The older `--training-summary` and
regression check and must never be presented as independent background `--require-training-sample-independence` spellings remain aliases, but now use
evidence. the same safer full-split semantics. A result from any lineage-exposed AOI is
only a regression check and must never be presented as independent evidence.
`render_operator_yolo_label_qa_contact_sheets.py` paginates complete visual `render_operator_yolo_label_qa_contact_sheets.py` paginates complete visual
reviews with `--tiles-per-sheet` (default `64`). This keeps large corpora reviews with `--tiles-per-sheet` (default `64`). This keeps large corpora
+68 -37
View File
@@ -111,8 +111,8 @@ def _sample_slugs(payload: dict[str, Any], *, split: str | None) -> set[str]:
return samples return samples
def training_sample_independence_evidence( def model_lineage_independence_evidence(
dataset_yaml: Path, training_summaries: list[Path] dataset_yaml: Path, lineage_summaries: list[Path]
) -> dict[str, Any]: ) -> dict[str, Any]:
evaluation_summary = dataset_yaml.parent / "yolo_tile_dataset_summary.json" evaluation_summary = dataset_yaml.parent / "yolo_tile_dataset_summary.json"
if not evaluation_summary.is_file(): if not evaluation_summary.is_file():
@@ -120,7 +120,7 @@ def training_sample_independence_evidence(
"status": "unavailable", "status": "unavailable",
"reason": "evaluation dataset summary is unavailable", "reason": "evaluation dataset summary is unavailable",
"evaluation_summary_path": str(evaluation_summary), "evaluation_summary_path": str(evaluation_summary),
"independent_for_all_supplied_training_corpora": False, "independent_for_all_supplied_lineage_corpora": False,
} }
try: try:
@@ -134,18 +134,23 @@ def training_sample_independence_evidence(
"reason": str(exc), "reason": str(exc),
"evaluation_summary_path": str(evaluation_summary), "evaluation_summary_path": str(evaluation_summary),
"evaluation_summary_sha256": sha256_file(evaluation_summary), "evaluation_summary_sha256": sha256_file(evaluation_summary),
"independent_for_all_supplied_training_corpora": False, "independent_for_all_supplied_lineage_corpora": False,
} }
rows: list[dict[str, Any]] = [] rows: list[dict[str, Any]] = []
union_overlap: set[str] = set() union_overlap: set[str] = set()
for raw_summary in training_summaries: for raw_summary in lineage_summaries:
summary = raw_summary.expanduser().resolve(strict=True) summary = raw_summary.expanduser().resolve(strict=True)
try: try:
payload = json.loads(summary.read_text(encoding="utf-8")) payload = json.loads(summary.read_text(encoding="utf-8"))
training_samples = _sample_slugs(payload, split="train") all_samples = _sample_slugs(payload, split=None)
if not training_samples: if not all_samples:
raise ValueError("training summary contains no training samples") raise ValueError("lineage summary contains no samples")
roles_by_sample: dict[str, set[str]] = {}
for tile in payload["tiles"]:
sample_slug = str(tile["sample_slug"]).strip()
split = str(tile.get("split") or "unknown").strip()
roles_by_sample.setdefault(sample_slug, set()).add(split)
except (json.JSONDecodeError, OSError, ValueError) as exc: except (json.JSONDecodeError, OSError, ValueError) as exc:
return { return {
"status": "invalid", "status": "invalid",
@@ -153,32 +158,35 @@ def training_sample_independence_evidence(
"evaluation_summary_path": str(evaluation_summary), "evaluation_summary_path": str(evaluation_summary),
"evaluation_summary_sha256": sha256_file(evaluation_summary), "evaluation_summary_sha256": sha256_file(evaluation_summary),
"evaluation_samples": sorted(evaluation_samples), "evaluation_samples": sorted(evaluation_samples),
"independent_for_all_supplied_training_corpora": False, "independent_for_all_supplied_lineage_corpora": False,
} }
overlap = evaluation_samples & training_samples overlap = evaluation_samples & all_samples
union_overlap.update(overlap) union_overlap.update(overlap)
rows.append( rows.append(
{ {
"training_summary_path": str(summary), "lineage_summary_path": str(summary),
"training_summary_sha256": sha256_file(summary), "lineage_summary_sha256": sha256_file(summary),
"training_sample_count": len(training_samples), "lineage_sample_count": len(all_samples),
"overlapping_evaluation_samples": sorted(overlap), "overlapping_evaluation_samples": sorted(overlap),
"exposure_roles": {
sample: sorted(roles_by_sample[sample]) for sample in sorted(overlap)
},
} }
) )
if not rows: if not rows:
return { return {
"status": "unavailable", "status": "unavailable",
"reason": "no training summaries were supplied", "reason": "no complete model-lineage summaries were supplied",
"evaluation_summary_path": str(evaluation_summary), "evaluation_summary_path": str(evaluation_summary),
"evaluation_summary_sha256": sha256_file(evaluation_summary), "evaluation_summary_sha256": sha256_file(evaluation_summary),
"evaluation_samples": sorted(evaluation_samples), "evaluation_samples": sorted(evaluation_samples),
"training_corpora": [], "lineage_corpora": [],
"overlapping_evaluation_samples": [], "overlapping_evaluation_samples": [],
"independent_for_all_supplied_training_corpora": False, "independent_for_all_supplied_lineage_corpora": False,
"interpretation": ( "interpretation": (
"Training/evaluation independence cannot be established " "Model-lineage/evaluation independence cannot be established "
"without exact training summaries." "without every ancestral corpus summary."
), ),
} }
@@ -188,29 +196,37 @@ def training_sample_independence_evidence(
"evaluation_summary_path": str(evaluation_summary), "evaluation_summary_path": str(evaluation_summary),
"evaluation_summary_sha256": sha256_file(evaluation_summary), "evaluation_summary_sha256": sha256_file(evaluation_summary),
"evaluation_samples": sorted(evaluation_samples), "evaluation_samples": sorted(evaluation_samples),
"training_corpora": rows, "lineage_corpora": rows,
"overlapping_evaluation_samples": sorted(union_overlap), "overlapping_evaluation_samples": sorted(union_overlap),
"independent_for_all_supplied_training_corpora": independent, "independent_for_all_supplied_lineage_corpora": independent,
"interpretation": ( "interpretation": (
"No evaluation AOI occurs in the train split of any supplied corpus." "No evaluation AOI occurs in any split of any supplied ancestral corpus."
if independent if independent
else "At least one evaluation AOI occurs in a supplied train split; " else "At least one evaluation AOI was exposed in an ancestral train, "
"the matrix is blocked before model loading." "validation, calibration or other split; the matrix is blocked before "
"model loading."
), ),
} }
def training_sample_independence_evidence(
dataset_yaml: Path, training_summaries: list[Path]
) -> dict[str, Any]:
"""Backward-compatible alias; evidence now checks every lineage split."""
return model_lineage_independence_evidence(dataset_yaml, training_summaries)
def write_blocked_manifest( def write_blocked_manifest(
output: Path, dataset_yaml: Path, evidence: dict[str, Any] output: Path, dataset_yaml: Path, evidence: dict[str, Any]
) -> None: ) -> None:
payload = { payload = {
"schema_version": 2, "schema_version": 3,
"generated_at": datetime.now(UTC).isoformat(), "generated_at": datetime.now(UTC).isoformat(),
"status": "blocked_training_sample_overlap", "status": "blocked_model_lineage_sample_exposure",
"claim_boundary": "No checkpoint ranking or release claim is permitted.", "claim_boundary": "No checkpoint ranking or release claim is permitted.",
"dataset_yaml": str(dataset_yaml), "dataset_yaml": str(dataset_yaml),
"dataset_yaml_sha256": sha256_file(dataset_yaml), "dataset_yaml_sha256": sha256_file(dataset_yaml),
"training_sample_independence_evidence": evidence, "model_lineage_independence_evidence": evidence,
"model_loading_attempted": False, "model_loading_attempted": False,
"gpu_inference_attempted": False, "gpu_inference_attempted": False,
} }
@@ -261,8 +277,21 @@ def main() -> int:
parser.add_argument("--output", type=Path, required=True) parser.add_argument("--output", type=Path, required=True)
parser.add_argument("--background-prefix", action="append", required=True) parser.add_argument("--background-prefix", action="append", required=True)
parser.add_argument("--background-confidence", type=float, default=0.15) parser.add_argument("--background-confidence", type=float, default=0.15)
parser.add_argument("--training-summary", type=Path, action="append", default=[]) parser.add_argument(
parser.add_argument("--require-training-sample-independence", action="store_true") "--lineage-summary",
"--training-summary",
dest="lineage_summary",
type=Path,
action="append",
default=[],
help="tile summary for every corpus in the complete model ancestry",
)
parser.add_argument(
"--require-lineage-sample-independence",
"--require-training-sample-independence",
dest="require_lineage_sample_independence",
action="store_true",
)
parser.add_argument("--device", default="cuda:0") parser.add_argument("--device", default="cuda:0")
parser.add_argument("--imgsz", type=int, default=640) parser.add_argument("--imgsz", type=int, default=640)
parser.add_argument("--batch", type=int, default=8) parser.add_argument("--batch", type=int, default=8)
@@ -275,17 +304,19 @@ def main() -> int:
dataset_yaml = args.dataset_yaml.expanduser().resolve(strict=True) dataset_yaml = args.dataset_yaml.expanduser().resolve(strict=True)
images = validation_images(dataset_yaml) images = validation_images(dataset_yaml)
overlap_evidence = dataset_overlap_evidence(dataset_yaml) overlap_evidence = dataset_overlap_evidence(dataset_yaml)
independence_evidence = training_sample_independence_evidence( independence_evidence = model_lineage_independence_evidence(
dataset_yaml, args.training_summary dataset_yaml, args.lineage_summary
) )
if args.require_training_sample_independence and not args.training_summary: if args.require_lineage_sample_independence and not args.lineage_summary:
parser.error( parser.error(
"--require-training-sample-independence requires at least one " "--require-lineage-sample-independence requires every ancestral "
"--training-summary" "--lineage-summary"
) )
if ( if (
args.require_training_sample_independence args.require_lineage_sample_independence
and not independence_evidence["independent_for_all_supplied_training_corpora"] and not independence_evidence[
"independent_for_all_supplied_lineage_corpora"
]
): ):
write_blocked_manifest(args.output, dataset_yaml, independence_evidence) write_blocked_manifest(args.output, dataset_yaml, independence_evidence)
return 3 return 3
@@ -374,14 +405,14 @@ def main() -> int:
reverse=True, reverse=True,
) )
payload = { payload = {
"schema_version": 2, "schema_version": 3,
"generated_at": datetime.now(UTC).isoformat(), "generated_at": datetime.now(UTC).isoformat(),
"status": "ok" if successful else "failed", "status": "ok" if successful else "failed",
"claim_boundary": "Non-protected validation ranking only; no test, challenge or promotion claim.", "claim_boundary": "Non-protected validation ranking only; no test, challenge or promotion claim.",
"dataset_yaml": str(dataset_yaml), "dataset_yaml": str(dataset_yaml),
"dataset_yaml_sha256": sha256_file(dataset_yaml), "dataset_yaml_sha256": sha256_file(dataset_yaml),
"dataset_overlap_evidence": overlap_evidence, "dataset_overlap_evidence": overlap_evidence,
"training_sample_independence_evidence": independence_evidence, "model_lineage_independence_evidence": independence_evidence,
"validation_image_count": len(images), "validation_image_count": len(images),
"pure_background_prefixes": list(prefixes), "pure_background_prefixes": list(prefixes),
"pure_background_confidence": args.background_confidence, "pure_background_confidence": args.background_confidence,
+59 -16
View File
@@ -1,9 +1,11 @@
import json import json
import sys
from pathlib import Path from pathlib import Path
from scripts.evaluate_yolo_checkpoint_matrix import ( from scripts.evaluate_yolo_checkpoint_matrix import (
dataset_overlap_evidence, dataset_overlap_evidence,
training_sample_independence_evidence, main,
model_lineage_independence_evidence,
write_blocked_manifest, write_blocked_manifest,
) )
@@ -67,7 +69,7 @@ def _write_summary(path: Path, rows: list[tuple[str, str]]) -> None:
) )
def test_training_sample_independence_detects_overlap_for_each_corpus( def test_model_lineage_independence_detects_exposure_in_every_split(
tmp_path: Path, tmp_path: Path,
) -> None: ) -> None:
evaluation = tmp_path / "evaluation" evaluation = tmp_path / "evaluation"
@@ -80,24 +82,25 @@ def test_training_sample_independence_detects_overlap_for_each_corpus(
) )
active = tmp_path / "active.json" active = tmp_path / "active.json"
challenger = tmp_path / "challenger.json" challenger = tmp_path / "challenger.json"
_write_summary(active, [("postel_bos", "train"), ("mol", "train")]) _write_summary(active, [("postel_bos", "train"), ("turnhout", "val")])
_write_summary( _write_summary(
challenger, [("postel_bos", "train"), ("dessel", "train")] challenger, [("postel_bos", "train"), ("dessel", "train")]
) )
evidence = training_sample_independence_evidence( evidence = model_lineage_independence_evidence(
dataset_yaml, [active, challenger] dataset_yaml, [active, challenger]
) )
assert evidence["status"] == "overlap" assert evidence["status"] == "overlap"
assert evidence["overlapping_evaluation_samples"] == ["postel_bos"] assert evidence["overlapping_evaluation_samples"] == ["postel_bos", "turnhout"]
assert evidence["independent_for_all_supplied_training_corpora"] is False assert evidence["independent_for_all_supplied_lineage_corpora"] is False
assert [ assert [
row["overlapping_evaluation_samples"] for row in evidence["training_corpora"] row["overlapping_evaluation_samples"] for row in evidence["lineage_corpora"]
] == [["postel_bos"], ["postel_bos"]] ] == [["postel_bos", "turnhout"], ["postel_bos"]]
assert evidence["lineage_corpora"][0]["exposure_roles"]["turnhout"] == ["val"]
def test_training_sample_independence_accepts_disjoint_samples(tmp_path: Path) -> None: def test_model_lineage_independence_accepts_disjoint_samples(tmp_path: Path) -> None:
evaluation = tmp_path / "evaluation" evaluation = tmp_path / "evaluation"
evaluation.mkdir() evaluation.mkdir()
dataset_yaml = evaluation / "dataset.yaml" dataset_yaml = evaluation / "dataset.yaml"
@@ -108,14 +111,14 @@ def test_training_sample_independence_accepts_disjoint_samples(tmp_path: Path) -
training = tmp_path / "training.json" training = tmp_path / "training.json"
_write_summary(training, [("mol", "train")]) _write_summary(training, [("mol", "train")])
evidence = training_sample_independence_evidence(dataset_yaml, [training]) evidence = model_lineage_independence_evidence(dataset_yaml, [training])
assert evidence["status"] == "independent" assert evidence["status"] == "independent"
assert evidence["overlapping_evaluation_samples"] == [] assert evidence["overlapping_evaluation_samples"] == []
assert evidence["independent_for_all_supplied_training_corpora"] is True assert evidence["independent_for_all_supplied_lineage_corpora"] is True
def test_training_sample_independence_is_unavailable_without_training_corpus( def test_model_lineage_independence_is_unavailable_without_ancestral_corpus(
tmp_path: Path, tmp_path: Path,
) -> None: ) -> None:
evaluation = tmp_path / "evaluation" evaluation = tmp_path / "evaluation"
@@ -126,11 +129,11 @@ def test_training_sample_independence_is_unavailable_without_training_corpus(
evaluation / "yolo_tile_dataset_summary.json", [("turnhout", "val")] evaluation / "yolo_tile_dataset_summary.json", [("turnhout", "val")]
) )
evidence = training_sample_independence_evidence(dataset_yaml, []) evidence = model_lineage_independence_evidence(dataset_yaml, [])
assert evidence["status"] == "unavailable" assert evidence["status"] == "unavailable"
assert evidence["overlapping_evaluation_samples"] == [] assert evidence["overlapping_evaluation_samples"] == []
assert evidence["independent_for_all_supplied_training_corpora"] is False assert evidence["independent_for_all_supplied_lineage_corpora"] is False
def test_blocked_manifest_records_that_models_and_gpu_were_not_used( def test_blocked_manifest_records_that_models_and_gpu_were_not_used(
@@ -145,11 +148,51 @@ def test_blocked_manifest_records_that_models_and_gpu_were_not_used(
dataset_yaml, dataset_yaml,
{ {
"status": "overlap", "status": "overlap",
"independent_for_all_supplied_training_corpora": False, "independent_for_all_supplied_lineage_corpora": False,
}, },
) )
payload = json.loads(output.read_text(encoding="utf-8")) payload = json.loads(output.read_text(encoding="utf-8"))
assert payload["status"] == "blocked_training_sample_overlap" assert payload["status"] == "blocked_model_lineage_sample_exposure"
assert payload["model_loading_attempted"] is False assert payload["model_loading_attempted"] is False
assert payload["gpu_inference_attempted"] is False assert payload["gpu_inference_attempted"] is False
def test_cli_blocks_lineage_validation_exposure_before_model_resolution(
tmp_path: Path, monkeypatch
) -> None:
evaluation = tmp_path / "evaluation"
images = evaluation / "images" / "val"
images.mkdir(parents=True)
(images / "turnhout_r0_c0.png").write_bytes(b"not opened before gate")
dataset_yaml = evaluation / "dataset.yaml"
dataset_yaml.write_text("path: .\nval: images/val\n", encoding="utf-8")
_write_summary(
evaluation / "yolo_tile_dataset_summary.json", [("turnhout", "val")]
)
ancestor = tmp_path / "ancestor.json"
_write_summary(ancestor, [("turnhout", "val"), ("mol", "train")])
output = tmp_path / "blocked.json"
monkeypatch.setattr(
sys,
"argv",
[
"evaluate_yolo_checkpoint_matrix.py",
"--dataset-yaml",
str(dataset_yaml),
"--model",
str(tmp_path / "model-is-never-resolved.pt"),
"--output",
str(output),
"--background-prefix",
"background",
"--lineage-summary",
str(ancestor),
"--require-lineage-sample-independence",
],
)
assert main() == 3
payload = json.loads(output.read_text(encoding="utf-8"))
assert payload["status"] == "blocked_model_lineage_sample_exposure"
assert payload["model_loading_attempted"] is False