block checkpoint evaluation on ancestral exposure
This commit is contained in:
+76
@@ -0,0 +1,76 @@
|
|||||||
|
{
|
||||||
|
"claim_boundary": "No checkpoint ranking or release claim is permitted.",
|
||||||
|
"dataset_yaml": "/app/storage/operator-data/yolo-building-aoi1024-bgaware512r8-nonoverlap-eval-r2/dataset.yaml",
|
||||||
|
"dataset_yaml_sha256": "c36f6d76aac4bc2f5e880e2f0133d176fb0dc277600e121570a944eee6db2c48",
|
||||||
|
"generated_at": "2026-08-09T20:08:28.285906+00:00",
|
||||||
|
"gpu_inference_attempted": false,
|
||||||
|
"model_lineage_independence_evidence": {
|
||||||
|
"evaluation_samples": [
|
||||||
|
"postel_bos",
|
||||||
|
"turnhout",
|
||||||
|
"westerlo"
|
||||||
|
],
|
||||||
|
"evaluation_summary_path": "/app/storage/operator-data/yolo-building-aoi1024-bgaware512r8-nonoverlap-eval-r2/yolo_tile_dataset_summary.json",
|
||||||
|
"evaluation_summary_sha256": "c17a8c47eaecc2dfb38778ff5a612f9e6c4d67c9e90e31d3a6a66d8857adec4c",
|
||||||
|
"independent_for_all_supplied_lineage_corpora": false,
|
||||||
|
"interpretation": "At least one evaluation AOI was exposed in an ancestral train, validation, calibration or other split; the matrix is blocked before model loading.",
|
||||||
|
"lineage_corpora": [
|
||||||
|
{
|
||||||
|
"exposure_roles": {
|
||||||
|
"postel_bos": [
|
||||||
|
"train"
|
||||||
|
],
|
||||||
|
"turnhout": [
|
||||||
|
"val"
|
||||||
|
],
|
||||||
|
"westerlo": [
|
||||||
|
"val"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"lineage_sample_count": 19,
|
||||||
|
"lineage_summary_path": "/app/storage/operator-data/yolo-building-aoi1024-expanded-minpx4vis035/yolo_tile_dataset_summary.json",
|
||||||
|
"lineage_summary_sha256": "1887e2ba5c719d33675a9ed85db99a274fd66d98d5f6ad44558d673620917201",
|
||||||
|
"overlapping_evaluation_samples": [
|
||||||
|
"postel_bos",
|
||||||
|
"turnhout",
|
||||||
|
"westerlo"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"exposure_roles": {
|
||||||
|
"postel_bos": [
|
||||||
|
"train"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"lineage_sample_count": 22,
|
||||||
|
"lineage_summary_path": "/app/storage/operator-data/yolo-building-aoi1024-smallbld-minpx3vis035/yolo_tile_dataset_summary.json",
|
||||||
|
"lineage_summary_sha256": "49b2a07d2105d08356431757b83eafc1498eaf1fb76965b1efe05b776824942a",
|
||||||
|
"overlapping_evaluation_samples": [
|
||||||
|
"postel_bos"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"exposure_roles": {
|
||||||
|
"postel_bos": [
|
||||||
|
"train"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"lineage_sample_count": 28,
|
||||||
|
"lineage_summary_path": "/app/storage/operator-data/yolo-building-aoi1024-reviewedexp6-minpx3vis035/yolo_tile_dataset_summary.json",
|
||||||
|
"lineage_summary_sha256": "7c917e31216d1df2174c0f9c736f88a81f3835aa991971f8fb8665e17ddf5c9c",
|
||||||
|
"overlapping_evaluation_samples": [
|
||||||
|
"postel_bos"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"overlapping_evaluation_samples": [
|
||||||
|
"postel_bos",
|
||||||
|
"turnhout",
|
||||||
|
"westerlo"
|
||||||
|
],
|
||||||
|
"status": "overlap"
|
||||||
|
},
|
||||||
|
"model_loading_attempted": false,
|
||||||
|
"schema_version": 3,
|
||||||
|
"status": "blocked_model_lineage_sample_exposure"
|
||||||
|
}
|
||||||
@@ -0,0 +1,115 @@
|
|||||||
|
{
|
||||||
|
"claim_boundary": "File/hash ancestry only; does not establish human review, protected-test validity or release fitness.",
|
||||||
|
"generated_at": "2026-08-09T20:10:17.734420+00:00",
|
||||||
|
"generic_base": {
|
||||||
|
"path": "/app/models/yolov8s.pt",
|
||||||
|
"sha256": "1f47a78bf100391c2a140b7ac73a1caae18c32779be7d310658112f7ac9aa78a",
|
||||||
|
"size_bytes": 22588772
|
||||||
|
},
|
||||||
|
"schema_version": 1,
|
||||||
|
"stages": [
|
||||||
|
{
|
||||||
|
"args": {
|
||||||
|
"path": "/app/storage/training/operator-yolo/geointel-building-yolov8s-aoi1024expandedminpx4vis035e50/args.yaml",
|
||||||
|
"sha256": "ddb12aa459404a905c2d908210033f18346491823c1b059eec332c1a7f9a0263",
|
||||||
|
"size_bytes": 1839
|
||||||
|
},
|
||||||
|
"best_matches_model_copy": true,
|
||||||
|
"dataset_summary": {
|
||||||
|
"path": "/app/storage/operator-data/yolo-building-aoi1024-expanded-minpx4vis035/yolo_tile_dataset_summary.json",
|
||||||
|
"sha256": "1887e2ba5c719d33675a9ed85db99a274fd66d98d5f6ad44558d673620917201",
|
||||||
|
"size_bytes": 135720
|
||||||
|
},
|
||||||
|
"dataset_yaml": {
|
||||||
|
"path": "/app/storage/operator-data/yolo-building-aoi1024-expanded-minpx4vis035/dataset.yaml",
|
||||||
|
"sha256": "357730f4ee110efdb241d26c0479eeeea5200885cebeb8eae011e257beb058ba",
|
||||||
|
"size_bytes": 134
|
||||||
|
},
|
||||||
|
"model_copy": {
|
||||||
|
"path": "/app/models/geointel-building-yolov8s-aoi1024expandedminpx4vis035e50.pt",
|
||||||
|
"sha256": "a8a79cf5b0bdc19a0245acc322cf77232c335e222bd5f3c00a17d5f29402c196",
|
||||||
|
"size_bytes": 22499498
|
||||||
|
},
|
||||||
|
"name": "expanded",
|
||||||
|
"parent_model": {
|
||||||
|
"path": "/app/models/yolov8s.pt",
|
||||||
|
"sha256": "1f47a78bf100391c2a140b7ac73a1caae18c32779be7d310658112f7ac9aa78a",
|
||||||
|
"size_bytes": 22588772
|
||||||
|
},
|
||||||
|
"training_best": {
|
||||||
|
"path": "/app/storage/training/operator-yolo/geointel-building-yolov8s-aoi1024expandedminpx4vis035e50/weights/best.pt",
|
||||||
|
"sha256": "a8a79cf5b0bdc19a0245acc322cf77232c335e222bd5f3c00a17d5f29402c196",
|
||||||
|
"size_bytes": 22499498
|
||||||
|
}
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"args": {
|
||||||
|
"path": "/app/storage/training/operator-yolo/geointel-building-yolov8s-smallbld-minpx3-img640-ft30/args.yaml",
|
||||||
|
"sha256": "2b482e6bbef26f433d4406e1acb5cbbf4ce63a63644b180a2d51b93f8c8f0dcb",
|
||||||
|
"size_bytes": 1882
|
||||||
|
},
|
||||||
|
"best_matches_model_copy": true,
|
||||||
|
"dataset_summary": {
|
||||||
|
"path": "/app/storage/operator-data/yolo-building-aoi1024-smallbld-minpx3vis035/yolo_tile_dataset_summary.json",
|
||||||
|
"sha256": "49b2a07d2105d08356431757b83eafc1498eaf1fb76965b1efe05b776824942a",
|
||||||
|
"size_bytes": 158980
|
||||||
|
},
|
||||||
|
"dataset_yaml": {
|
||||||
|
"path": "/app/storage/operator-data/yolo-building-aoi1024-smallbld-minpx3vis035/dataset.yaml",
|
||||||
|
"sha256": "3a2ea97c35a18072a1ab6738cd673c0ecec5344b19461c91d72a15e138d46e8d",
|
||||||
|
"size_bytes": 134
|
||||||
|
},
|
||||||
|
"model_copy": {
|
||||||
|
"path": "/app/models/geointel-building-yolov8s-smallbld-minpx3-img640-ft30.pt",
|
||||||
|
"sha256": "a9088b8491dfae36694b53e9e9406cb4e3511d334a5712fa34f75078a47759c1",
|
||||||
|
"size_bytes": 22516074
|
||||||
|
},
|
||||||
|
"name": "active",
|
||||||
|
"parent_model": {
|
||||||
|
"path": "/app/models/geointel-building-yolov8s-aoi1024expandedminpx4vis035e50.pt",
|
||||||
|
"sha256": "a8a79cf5b0bdc19a0245acc322cf77232c335e222bd5f3c00a17d5f29402c196",
|
||||||
|
"size_bytes": 22499498
|
||||||
|
},
|
||||||
|
"training_best": {
|
||||||
|
"path": "/app/storage/training/operator-yolo/geointel-building-yolov8s-smallbld-minpx3-img640-ft30/weights/best.pt",
|
||||||
|
"sha256": "a9088b8491dfae36694b53e9e9406cb4e3511d334a5712fa34f75078a47759c1",
|
||||||
|
"size_bytes": 22516074
|
||||||
|
}
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"args": {
|
||||||
|
"path": "/app/storage/training/operator-yolo/geointel-building-yolov8s-reviewedexp6-minpx3-img640-ft20/args.yaml",
|
||||||
|
"sha256": "87e07dbed3ad12cc474e6b895147839431805b7044496277242e1538fcd91a0b",
|
||||||
|
"size_bytes": 1891
|
||||||
|
},
|
||||||
|
"best_matches_model_copy": true,
|
||||||
|
"dataset_summary": {
|
||||||
|
"path": "/app/storage/operator-data/yolo-building-aoi1024-reviewedexp6-minpx3vis035/yolo_tile_dataset_summary.json",
|
||||||
|
"sha256": "7c917e31216d1df2174c0f9c736f88a81f3835aa991971f8fb8665e17ddf5c9c",
|
||||||
|
"size_bytes": 204029
|
||||||
|
},
|
||||||
|
"dataset_yaml": {
|
||||||
|
"path": "/app/storage/operator-data/yolo-building-aoi1024-reviewedexp6-minpx3vis035/dataset.yaml",
|
||||||
|
"sha256": "62e12f4433508cb0640f1d90ec0ffd389cd21ebe655fac4e7d9402f04da1168e",
|
||||||
|
"size_bytes": 138
|
||||||
|
},
|
||||||
|
"model_copy": {
|
||||||
|
"path": "/app/models/geointel-building-yolov8s-reviewedexp6-minpx3-img640-ft20.pt",
|
||||||
|
"sha256": "038f1f97a6afd534f29e1f392a730a58207b928ca01e31ab8d8fed6106705820",
|
||||||
|
"size_bytes": 22514794
|
||||||
|
},
|
||||||
|
"name": "challenger",
|
||||||
|
"parent_model": {
|
||||||
|
"path": "/app/models/geointel-building-yolov8s-smallbld-minpx3-img640-ft30.pt",
|
||||||
|
"sha256": "a9088b8491dfae36694b53e9e9406cb4e3511d334a5712fa34f75078a47759c1",
|
||||||
|
"size_bytes": 22516074
|
||||||
|
},
|
||||||
|
"training_best": {
|
||||||
|
"path": "/app/storage/training/operator-yolo/geointel-building-yolov8s-reviewedexp6-minpx3-img640-ft20/weights/best.pt",
|
||||||
|
"sha256": "038f1f97a6afd534f29e1f392a730a58207b928ca01e31ab8d8fed6106705820",
|
||||||
|
"size_bytes": 22514794
|
||||||
|
}
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"status": "complete_retained_lineage_to_generic_base"
|
||||||
|
}
|
||||||
@@ -12737,3 +12737,37 @@ Open:
|
|||||||
- Independent real-background AOIs are still required. Literal 100% model
|
- Independent real-background AOIs are still required. Literal 100% model
|
||||||
correctness is not demonstrated; the system now refuses to misrepresent the
|
correctness is not demonstrated; the system now refuses to misrepresent the
|
||||||
known overlap as proof.
|
known overlap as proof.
|
||||||
|
|
||||||
|
## 2026-08-09 - Full ancestral-corpus exposure correction
|
||||||
|
|
||||||
|
### Corrected
|
||||||
|
|
||||||
|
- Reconstructed the retained checkpoint ancestry from exact Ultralytics
|
||||||
|
`args.yaml` files and model hashes. The active model descends from the
|
||||||
|
expanded-min4 corpus; the challenger descends from the active model.
|
||||||
|
- Found that the expanded ancestor used Postel-bos for training and
|
||||||
|
Turnhout/Westerlo for validation. All three v69 evaluation AOIs were therefore
|
||||||
|
exposed somewhere in the model family.
|
||||||
|
- Replaced last-corpus/train-only checking with fail-closed inspection of every
|
||||||
|
split in every supplied ancestral corpus. Legacy CLI spellings remain safe
|
||||||
|
aliases to the full-lineage behavior.
|
||||||
|
|
||||||
|
### Verified
|
||||||
|
|
||||||
|
- The Tower gate blocked all three exposed AOIs with zero model-loading and
|
||||||
|
zero GPU-inference attempts.
|
||||||
|
- Expanded ancestor summary SHA-256:
|
||||||
|
`1887e2ba5c719d33675a9ed85db99a274fd66d98d5f6ad44558d673620917201`.
|
||||||
|
- Blocked-manifest SHA-256:
|
||||||
|
`f6effd94befd6df8d699bc66fbbea8a860b056cff9996be5f867f858807af65e`.
|
||||||
|
- Model-lineage receipt SHA-256:
|
||||||
|
`84d8d7bd95acfb07b188b07ececfc5417fca80c087d7c04442dd2a4da5f43936`;
|
||||||
|
every retained training `best.pt` matches its named model copy byte-for-byte.
|
||||||
|
- The production checkpoint and container image remain unchanged.
|
||||||
|
|
||||||
|
### Accuracy boundary
|
||||||
|
|
||||||
|
- The old Turnhout/Westerlo/Postel numbers are regression diagnostics only.
|
||||||
|
No independent positive or background accuracy evidence currently exists for
|
||||||
|
this model family; new AOIs must be provisioned without ancestry or spatial
|
||||||
|
exposure before another ranking is defensible.
|
||||||
|
|||||||
+3
-2
@@ -1179,8 +1179,9 @@ This file now starts with the current implementation status. Older preparation/b
|
|||||||
as release evidence. Postel occurs in both compared train splits and is only
|
as release evidence. Postel occurs in both compared train splits and is only
|
||||||
a training-seen regression check, not independent validation.
|
a training-seen regression check, not independent validation.
|
||||||
- [x] Add a fail-closed checkpoint gate that checks every evaluation AOI
|
- [x] Add a fail-closed checkpoint gate that checks every evaluation AOI
|
||||||
against the exact train split of every supplied model corpus before PyTorch
|
against every split of every supplied ancestral model corpus before PyTorch
|
||||||
import, model loading or GPU inference.
|
import, model loading or GPU inference. Turnhout/Westerlo were exposed as
|
||||||
|
validation in the base building model; Postel was exposed as training.
|
||||||
- [ ] Convert the AI-assisted ledger into no stronger claim than experimental
|
- [ ] Convert the AI-assisted ledger into no stronger claim than experimental
|
||||||
triage; a real human must independently review and sign the frozen artifacts
|
triage; a real human must independently review and sign the frozen artifacts
|
||||||
before the governed training wrapper may unlock.
|
before the governed training wrapper may unlock.
|
||||||
|
|||||||
@@ -237,3 +237,29 @@ mode, fails before PyTorch import, model loading or GPU inference when any
|
|||||||
evaluation AOI overlaps any supplied train split. The reproduced gate blocked
|
evaluation AOI overlaps any supplied train split. The reproduced gate blocked
|
||||||
on `postel_bos` for both checkpoints. Its immutable machine-readable record is
|
on `postel_bos` for both checkpoints. Its immutable machine-readable record is
|
||||||
`artifacts/evidence/accuracy/model-training/20260809-v70-evaluation-independence-gate.json`.
|
`artifacts/evidence/accuracy/model-training/20260809-v70-evaluation-independence-gate.json`.
|
||||||
|
|
||||||
|
## Full model-lineage correction
|
||||||
|
|
||||||
|
The preceding correction still considered only the final fine-tune corpus of
|
||||||
|
each checkpoint. Exact retained Ultralytics arguments establish a longer
|
||||||
|
ancestry: the active checkpoint was initialized from
|
||||||
|
`geointel-building-yolov8s-aoi1024expandedminpx4vis035e50`, which was initialized
|
||||||
|
from the generic `yolov8s.pt`; the challenger was then initialized from the
|
||||||
|
active checkpoint. The copied model assets and retained `best.pt` files match
|
||||||
|
byte-for-byte at each building-model stage.
|
||||||
|
|
||||||
|
The ancestral expanded corpus exposes all three evaluation AOIs: `postel_bos`
|
||||||
|
as train, and `turnhout` plus `westerlo` as validation. Therefore none of the
|
||||||
|
v69 AOIs is independent of the complete model family. Turnhout/Westerlo can
|
||||||
|
still be used as familiar regression diagnostics, but their metrics are not a
|
||||||
|
fresh candidate-ranking result and must not support accuracy, uncertainty,
|
||||||
|
generalisation or release claims.
|
||||||
|
|
||||||
|
The gate now requires every ancestral corpus summary and checks every recorded
|
||||||
|
split, including validation and calibration. A reproduced Tower run blocked on
|
||||||
|
all three AOIs before PyTorch import, model loading or GPU inference. Evidence:
|
||||||
|
`artifacts/evidence/accuracy/model-training/20260809-v71-full-lineage-independence-gate.json`.
|
||||||
|
The byte-matching parent/output chain is retained separately in
|
||||||
|
`artifacts/evidence/accuracy/model-training/20260809-v71-model-lineage-receipt.json`.
|
||||||
|
Fresh geographically separated AOIs with complete spatial and lineage checks
|
||||||
|
are required for the next meaningful evaluation.
|
||||||
|
|||||||
+9
-6
@@ -5,13 +5,16 @@ weights on one declared, non-protected `val` split using CUDA. It records exact
|
|||||||
model and dataset hashes, standard Ultralytics detection metrics and a separate
|
model and dataset hashes, standard Ultralytics detection metrics and a separate
|
||||||
pure-background detection count. The output claim is validation ranking only;
|
pure-background detection count. The output claim is validation ranking only;
|
||||||
the script neither reads protected test data nor promotes a model.
|
the script neither reads protected test data nor promotes a model.
|
||||||
Governed comparisons must pass every candidate's exact tile-summary with
|
Governed comparisons must pass the exact tile summary for every corpus in the
|
||||||
`--training-summary` and enable `--require-training-sample-independence`. If
|
complete ancestry of every candidate with `--lineage-summary` and enable
|
||||||
any validation AOI occurs in any supplied train split, the script writes a
|
`--require-lineage-sample-independence`. The check covers train, validation,
|
||||||
|
calibration and every other recorded split, because an AOI used for ancestral
|
||||||
|
checkpoint or threshold selection is also exposed. Any exposure writes a
|
||||||
blocked manifest and exits before importing PyTorch, loading a model or using
|
blocked manifest and exits before importing PyTorch, loading a model or using
|
||||||
the GPU. A background detection count from a training-seen AOI is only a
|
the GPU. The older `--training-summary` and
|
||||||
regression check and must never be presented as independent background
|
`--require-training-sample-independence` spellings remain aliases, but now use
|
||||||
evidence.
|
the same safer full-split semantics. A result from any lineage-exposed AOI is
|
||||||
|
only a regression check and must never be presented as independent evidence.
|
||||||
|
|
||||||
`render_operator_yolo_label_qa_contact_sheets.py` paginates complete visual
|
`render_operator_yolo_label_qa_contact_sheets.py` paginates complete visual
|
||||||
reviews with `--tiles-per-sheet` (default `64`). This keeps large corpora
|
reviews with `--tiles-per-sheet` (default `64`). This keeps large corpora
|
||||||
|
|||||||
@@ -111,8 +111,8 @@ def _sample_slugs(payload: dict[str, Any], *, split: str | None) -> set[str]:
|
|||||||
return samples
|
return samples
|
||||||
|
|
||||||
|
|
||||||
def training_sample_independence_evidence(
|
def model_lineage_independence_evidence(
|
||||||
dataset_yaml: Path, training_summaries: list[Path]
|
dataset_yaml: Path, lineage_summaries: list[Path]
|
||||||
) -> dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
evaluation_summary = dataset_yaml.parent / "yolo_tile_dataset_summary.json"
|
evaluation_summary = dataset_yaml.parent / "yolo_tile_dataset_summary.json"
|
||||||
if not evaluation_summary.is_file():
|
if not evaluation_summary.is_file():
|
||||||
@@ -120,7 +120,7 @@ def training_sample_independence_evidence(
|
|||||||
"status": "unavailable",
|
"status": "unavailable",
|
||||||
"reason": "evaluation dataset summary is unavailable",
|
"reason": "evaluation dataset summary is unavailable",
|
||||||
"evaluation_summary_path": str(evaluation_summary),
|
"evaluation_summary_path": str(evaluation_summary),
|
||||||
"independent_for_all_supplied_training_corpora": False,
|
"independent_for_all_supplied_lineage_corpora": False,
|
||||||
}
|
}
|
||||||
|
|
||||||
try:
|
try:
|
||||||
@@ -134,18 +134,23 @@ def training_sample_independence_evidence(
|
|||||||
"reason": str(exc),
|
"reason": str(exc),
|
||||||
"evaluation_summary_path": str(evaluation_summary),
|
"evaluation_summary_path": str(evaluation_summary),
|
||||||
"evaluation_summary_sha256": sha256_file(evaluation_summary),
|
"evaluation_summary_sha256": sha256_file(evaluation_summary),
|
||||||
"independent_for_all_supplied_training_corpora": False,
|
"independent_for_all_supplied_lineage_corpora": False,
|
||||||
}
|
}
|
||||||
|
|
||||||
rows: list[dict[str, Any]] = []
|
rows: list[dict[str, Any]] = []
|
||||||
union_overlap: set[str] = set()
|
union_overlap: set[str] = set()
|
||||||
for raw_summary in training_summaries:
|
for raw_summary in lineage_summaries:
|
||||||
summary = raw_summary.expanduser().resolve(strict=True)
|
summary = raw_summary.expanduser().resolve(strict=True)
|
||||||
try:
|
try:
|
||||||
payload = json.loads(summary.read_text(encoding="utf-8"))
|
payload = json.loads(summary.read_text(encoding="utf-8"))
|
||||||
training_samples = _sample_slugs(payload, split="train")
|
all_samples = _sample_slugs(payload, split=None)
|
||||||
if not training_samples:
|
if not all_samples:
|
||||||
raise ValueError("training summary contains no training samples")
|
raise ValueError("lineage summary contains no samples")
|
||||||
|
roles_by_sample: dict[str, set[str]] = {}
|
||||||
|
for tile in payload["tiles"]:
|
||||||
|
sample_slug = str(tile["sample_slug"]).strip()
|
||||||
|
split = str(tile.get("split") or "unknown").strip()
|
||||||
|
roles_by_sample.setdefault(sample_slug, set()).add(split)
|
||||||
except (json.JSONDecodeError, OSError, ValueError) as exc:
|
except (json.JSONDecodeError, OSError, ValueError) as exc:
|
||||||
return {
|
return {
|
||||||
"status": "invalid",
|
"status": "invalid",
|
||||||
@@ -153,32 +158,35 @@ def training_sample_independence_evidence(
|
|||||||
"evaluation_summary_path": str(evaluation_summary),
|
"evaluation_summary_path": str(evaluation_summary),
|
||||||
"evaluation_summary_sha256": sha256_file(evaluation_summary),
|
"evaluation_summary_sha256": sha256_file(evaluation_summary),
|
||||||
"evaluation_samples": sorted(evaluation_samples),
|
"evaluation_samples": sorted(evaluation_samples),
|
||||||
"independent_for_all_supplied_training_corpora": False,
|
"independent_for_all_supplied_lineage_corpora": False,
|
||||||
}
|
}
|
||||||
overlap = evaluation_samples & training_samples
|
overlap = evaluation_samples & all_samples
|
||||||
union_overlap.update(overlap)
|
union_overlap.update(overlap)
|
||||||
rows.append(
|
rows.append(
|
||||||
{
|
{
|
||||||
"training_summary_path": str(summary),
|
"lineage_summary_path": str(summary),
|
||||||
"training_summary_sha256": sha256_file(summary),
|
"lineage_summary_sha256": sha256_file(summary),
|
||||||
"training_sample_count": len(training_samples),
|
"lineage_sample_count": len(all_samples),
|
||||||
"overlapping_evaluation_samples": sorted(overlap),
|
"overlapping_evaluation_samples": sorted(overlap),
|
||||||
|
"exposure_roles": {
|
||||||
|
sample: sorted(roles_by_sample[sample]) for sample in sorted(overlap)
|
||||||
|
},
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
|
|
||||||
if not rows:
|
if not rows:
|
||||||
return {
|
return {
|
||||||
"status": "unavailable",
|
"status": "unavailable",
|
||||||
"reason": "no training summaries were supplied",
|
"reason": "no complete model-lineage summaries were supplied",
|
||||||
"evaluation_summary_path": str(evaluation_summary),
|
"evaluation_summary_path": str(evaluation_summary),
|
||||||
"evaluation_summary_sha256": sha256_file(evaluation_summary),
|
"evaluation_summary_sha256": sha256_file(evaluation_summary),
|
||||||
"evaluation_samples": sorted(evaluation_samples),
|
"evaluation_samples": sorted(evaluation_samples),
|
||||||
"training_corpora": [],
|
"lineage_corpora": [],
|
||||||
"overlapping_evaluation_samples": [],
|
"overlapping_evaluation_samples": [],
|
||||||
"independent_for_all_supplied_training_corpora": False,
|
"independent_for_all_supplied_lineage_corpora": False,
|
||||||
"interpretation": (
|
"interpretation": (
|
||||||
"Training/evaluation independence cannot be established "
|
"Model-lineage/evaluation independence cannot be established "
|
||||||
"without exact training summaries."
|
"without every ancestral corpus summary."
|
||||||
),
|
),
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -188,29 +196,37 @@ def training_sample_independence_evidence(
|
|||||||
"evaluation_summary_path": str(evaluation_summary),
|
"evaluation_summary_path": str(evaluation_summary),
|
||||||
"evaluation_summary_sha256": sha256_file(evaluation_summary),
|
"evaluation_summary_sha256": sha256_file(evaluation_summary),
|
||||||
"evaluation_samples": sorted(evaluation_samples),
|
"evaluation_samples": sorted(evaluation_samples),
|
||||||
"training_corpora": rows,
|
"lineage_corpora": rows,
|
||||||
"overlapping_evaluation_samples": sorted(union_overlap),
|
"overlapping_evaluation_samples": sorted(union_overlap),
|
||||||
"independent_for_all_supplied_training_corpora": independent,
|
"independent_for_all_supplied_lineage_corpora": independent,
|
||||||
"interpretation": (
|
"interpretation": (
|
||||||
"No evaluation AOI occurs in the train split of any supplied corpus."
|
"No evaluation AOI occurs in any split of any supplied ancestral corpus."
|
||||||
if independent
|
if independent
|
||||||
else "At least one evaluation AOI occurs in a supplied train split; "
|
else "At least one evaluation AOI was exposed in an ancestral train, "
|
||||||
"the matrix is blocked before model loading."
|
"validation, calibration or other split; the matrix is blocked before "
|
||||||
|
"model loading."
|
||||||
),
|
),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def training_sample_independence_evidence(
|
||||||
|
dataset_yaml: Path, training_summaries: list[Path]
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Backward-compatible alias; evidence now checks every lineage split."""
|
||||||
|
return model_lineage_independence_evidence(dataset_yaml, training_summaries)
|
||||||
|
|
||||||
|
|
||||||
def write_blocked_manifest(
|
def write_blocked_manifest(
|
||||||
output: Path, dataset_yaml: Path, evidence: dict[str, Any]
|
output: Path, dataset_yaml: Path, evidence: dict[str, Any]
|
||||||
) -> None:
|
) -> None:
|
||||||
payload = {
|
payload = {
|
||||||
"schema_version": 2,
|
"schema_version": 3,
|
||||||
"generated_at": datetime.now(UTC).isoformat(),
|
"generated_at": datetime.now(UTC).isoformat(),
|
||||||
"status": "blocked_training_sample_overlap",
|
"status": "blocked_model_lineage_sample_exposure",
|
||||||
"claim_boundary": "No checkpoint ranking or release claim is permitted.",
|
"claim_boundary": "No checkpoint ranking or release claim is permitted.",
|
||||||
"dataset_yaml": str(dataset_yaml),
|
"dataset_yaml": str(dataset_yaml),
|
||||||
"dataset_yaml_sha256": sha256_file(dataset_yaml),
|
"dataset_yaml_sha256": sha256_file(dataset_yaml),
|
||||||
"training_sample_independence_evidence": evidence,
|
"model_lineage_independence_evidence": evidence,
|
||||||
"model_loading_attempted": False,
|
"model_loading_attempted": False,
|
||||||
"gpu_inference_attempted": False,
|
"gpu_inference_attempted": False,
|
||||||
}
|
}
|
||||||
@@ -261,8 +277,21 @@ def main() -> int:
|
|||||||
parser.add_argument("--output", type=Path, required=True)
|
parser.add_argument("--output", type=Path, required=True)
|
||||||
parser.add_argument("--background-prefix", action="append", required=True)
|
parser.add_argument("--background-prefix", action="append", required=True)
|
||||||
parser.add_argument("--background-confidence", type=float, default=0.15)
|
parser.add_argument("--background-confidence", type=float, default=0.15)
|
||||||
parser.add_argument("--training-summary", type=Path, action="append", default=[])
|
parser.add_argument(
|
||||||
parser.add_argument("--require-training-sample-independence", action="store_true")
|
"--lineage-summary",
|
||||||
|
"--training-summary",
|
||||||
|
dest="lineage_summary",
|
||||||
|
type=Path,
|
||||||
|
action="append",
|
||||||
|
default=[],
|
||||||
|
help="tile summary for every corpus in the complete model ancestry",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--require-lineage-sample-independence",
|
||||||
|
"--require-training-sample-independence",
|
||||||
|
dest="require_lineage_sample_independence",
|
||||||
|
action="store_true",
|
||||||
|
)
|
||||||
parser.add_argument("--device", default="cuda:0")
|
parser.add_argument("--device", default="cuda:0")
|
||||||
parser.add_argument("--imgsz", type=int, default=640)
|
parser.add_argument("--imgsz", type=int, default=640)
|
||||||
parser.add_argument("--batch", type=int, default=8)
|
parser.add_argument("--batch", type=int, default=8)
|
||||||
@@ -275,17 +304,19 @@ def main() -> int:
|
|||||||
dataset_yaml = args.dataset_yaml.expanduser().resolve(strict=True)
|
dataset_yaml = args.dataset_yaml.expanduser().resolve(strict=True)
|
||||||
images = validation_images(dataset_yaml)
|
images = validation_images(dataset_yaml)
|
||||||
overlap_evidence = dataset_overlap_evidence(dataset_yaml)
|
overlap_evidence = dataset_overlap_evidence(dataset_yaml)
|
||||||
independence_evidence = training_sample_independence_evidence(
|
independence_evidence = model_lineage_independence_evidence(
|
||||||
dataset_yaml, args.training_summary
|
dataset_yaml, args.lineage_summary
|
||||||
)
|
)
|
||||||
if args.require_training_sample_independence and not args.training_summary:
|
if args.require_lineage_sample_independence and not args.lineage_summary:
|
||||||
parser.error(
|
parser.error(
|
||||||
"--require-training-sample-independence requires at least one "
|
"--require-lineage-sample-independence requires every ancestral "
|
||||||
"--training-summary"
|
"--lineage-summary"
|
||||||
)
|
)
|
||||||
if (
|
if (
|
||||||
args.require_training_sample_independence
|
args.require_lineage_sample_independence
|
||||||
and not independence_evidence["independent_for_all_supplied_training_corpora"]
|
and not independence_evidence[
|
||||||
|
"independent_for_all_supplied_lineage_corpora"
|
||||||
|
]
|
||||||
):
|
):
|
||||||
write_blocked_manifest(args.output, dataset_yaml, independence_evidence)
|
write_blocked_manifest(args.output, dataset_yaml, independence_evidence)
|
||||||
return 3
|
return 3
|
||||||
@@ -374,14 +405,14 @@ def main() -> int:
|
|||||||
reverse=True,
|
reverse=True,
|
||||||
)
|
)
|
||||||
payload = {
|
payload = {
|
||||||
"schema_version": 2,
|
"schema_version": 3,
|
||||||
"generated_at": datetime.now(UTC).isoformat(),
|
"generated_at": datetime.now(UTC).isoformat(),
|
||||||
"status": "ok" if successful else "failed",
|
"status": "ok" if successful else "failed",
|
||||||
"claim_boundary": "Non-protected validation ranking only; no test, challenge or promotion claim.",
|
"claim_boundary": "Non-protected validation ranking only; no test, challenge or promotion claim.",
|
||||||
"dataset_yaml": str(dataset_yaml),
|
"dataset_yaml": str(dataset_yaml),
|
||||||
"dataset_yaml_sha256": sha256_file(dataset_yaml),
|
"dataset_yaml_sha256": sha256_file(dataset_yaml),
|
||||||
"dataset_overlap_evidence": overlap_evidence,
|
"dataset_overlap_evidence": overlap_evidence,
|
||||||
"training_sample_independence_evidence": independence_evidence,
|
"model_lineage_independence_evidence": independence_evidence,
|
||||||
"validation_image_count": len(images),
|
"validation_image_count": len(images),
|
||||||
"pure_background_prefixes": list(prefixes),
|
"pure_background_prefixes": list(prefixes),
|
||||||
"pure_background_confidence": args.background_confidence,
|
"pure_background_confidence": args.background_confidence,
|
||||||
|
|||||||
@@ -1,9 +1,11 @@
|
|||||||
import json
|
import json
|
||||||
|
import sys
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
from scripts.evaluate_yolo_checkpoint_matrix import (
|
from scripts.evaluate_yolo_checkpoint_matrix import (
|
||||||
dataset_overlap_evidence,
|
dataset_overlap_evidence,
|
||||||
training_sample_independence_evidence,
|
main,
|
||||||
|
model_lineage_independence_evidence,
|
||||||
write_blocked_manifest,
|
write_blocked_manifest,
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -67,7 +69,7 @@ def _write_summary(path: Path, rows: list[tuple[str, str]]) -> None:
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def test_training_sample_independence_detects_overlap_for_each_corpus(
|
def test_model_lineage_independence_detects_exposure_in_every_split(
|
||||||
tmp_path: Path,
|
tmp_path: Path,
|
||||||
) -> None:
|
) -> None:
|
||||||
evaluation = tmp_path / "evaluation"
|
evaluation = tmp_path / "evaluation"
|
||||||
@@ -80,24 +82,25 @@ def test_training_sample_independence_detects_overlap_for_each_corpus(
|
|||||||
)
|
)
|
||||||
active = tmp_path / "active.json"
|
active = tmp_path / "active.json"
|
||||||
challenger = tmp_path / "challenger.json"
|
challenger = tmp_path / "challenger.json"
|
||||||
_write_summary(active, [("postel_bos", "train"), ("mol", "train")])
|
_write_summary(active, [("postel_bos", "train"), ("turnhout", "val")])
|
||||||
_write_summary(
|
_write_summary(
|
||||||
challenger, [("postel_bos", "train"), ("dessel", "train")]
|
challenger, [("postel_bos", "train"), ("dessel", "train")]
|
||||||
)
|
)
|
||||||
|
|
||||||
evidence = training_sample_independence_evidence(
|
evidence = model_lineage_independence_evidence(
|
||||||
dataset_yaml, [active, challenger]
|
dataset_yaml, [active, challenger]
|
||||||
)
|
)
|
||||||
|
|
||||||
assert evidence["status"] == "overlap"
|
assert evidence["status"] == "overlap"
|
||||||
assert evidence["overlapping_evaluation_samples"] == ["postel_bos"]
|
assert evidence["overlapping_evaluation_samples"] == ["postel_bos", "turnhout"]
|
||||||
assert evidence["independent_for_all_supplied_training_corpora"] is False
|
assert evidence["independent_for_all_supplied_lineage_corpora"] is False
|
||||||
assert [
|
assert [
|
||||||
row["overlapping_evaluation_samples"] for row in evidence["training_corpora"]
|
row["overlapping_evaluation_samples"] for row in evidence["lineage_corpora"]
|
||||||
] == [["postel_bos"], ["postel_bos"]]
|
] == [["postel_bos", "turnhout"], ["postel_bos"]]
|
||||||
|
assert evidence["lineage_corpora"][0]["exposure_roles"]["turnhout"] == ["val"]
|
||||||
|
|
||||||
|
|
||||||
def test_training_sample_independence_accepts_disjoint_samples(tmp_path: Path) -> None:
|
def test_model_lineage_independence_accepts_disjoint_samples(tmp_path: Path) -> None:
|
||||||
evaluation = tmp_path / "evaluation"
|
evaluation = tmp_path / "evaluation"
|
||||||
evaluation.mkdir()
|
evaluation.mkdir()
|
||||||
dataset_yaml = evaluation / "dataset.yaml"
|
dataset_yaml = evaluation / "dataset.yaml"
|
||||||
@@ -108,14 +111,14 @@ def test_training_sample_independence_accepts_disjoint_samples(tmp_path: Path) -
|
|||||||
training = tmp_path / "training.json"
|
training = tmp_path / "training.json"
|
||||||
_write_summary(training, [("mol", "train")])
|
_write_summary(training, [("mol", "train")])
|
||||||
|
|
||||||
evidence = training_sample_independence_evidence(dataset_yaml, [training])
|
evidence = model_lineage_independence_evidence(dataset_yaml, [training])
|
||||||
|
|
||||||
assert evidence["status"] == "independent"
|
assert evidence["status"] == "independent"
|
||||||
assert evidence["overlapping_evaluation_samples"] == []
|
assert evidence["overlapping_evaluation_samples"] == []
|
||||||
assert evidence["independent_for_all_supplied_training_corpora"] is True
|
assert evidence["independent_for_all_supplied_lineage_corpora"] is True
|
||||||
|
|
||||||
|
|
||||||
def test_training_sample_independence_is_unavailable_without_training_corpus(
|
def test_model_lineage_independence_is_unavailable_without_ancestral_corpus(
|
||||||
tmp_path: Path,
|
tmp_path: Path,
|
||||||
) -> None:
|
) -> None:
|
||||||
evaluation = tmp_path / "evaluation"
|
evaluation = tmp_path / "evaluation"
|
||||||
@@ -126,11 +129,11 @@ def test_training_sample_independence_is_unavailable_without_training_corpus(
|
|||||||
evaluation / "yolo_tile_dataset_summary.json", [("turnhout", "val")]
|
evaluation / "yolo_tile_dataset_summary.json", [("turnhout", "val")]
|
||||||
)
|
)
|
||||||
|
|
||||||
evidence = training_sample_independence_evidence(dataset_yaml, [])
|
evidence = model_lineage_independence_evidence(dataset_yaml, [])
|
||||||
|
|
||||||
assert evidence["status"] == "unavailable"
|
assert evidence["status"] == "unavailable"
|
||||||
assert evidence["overlapping_evaluation_samples"] == []
|
assert evidence["overlapping_evaluation_samples"] == []
|
||||||
assert evidence["independent_for_all_supplied_training_corpora"] is False
|
assert evidence["independent_for_all_supplied_lineage_corpora"] is False
|
||||||
|
|
||||||
|
|
||||||
def test_blocked_manifest_records_that_models_and_gpu_were_not_used(
|
def test_blocked_manifest_records_that_models_and_gpu_were_not_used(
|
||||||
@@ -145,11 +148,51 @@ def test_blocked_manifest_records_that_models_and_gpu_were_not_used(
|
|||||||
dataset_yaml,
|
dataset_yaml,
|
||||||
{
|
{
|
||||||
"status": "overlap",
|
"status": "overlap",
|
||||||
"independent_for_all_supplied_training_corpora": False,
|
"independent_for_all_supplied_lineage_corpora": False,
|
||||||
},
|
},
|
||||||
)
|
)
|
||||||
|
|
||||||
payload = json.loads(output.read_text(encoding="utf-8"))
|
payload = json.loads(output.read_text(encoding="utf-8"))
|
||||||
assert payload["status"] == "blocked_training_sample_overlap"
|
assert payload["status"] == "blocked_model_lineage_sample_exposure"
|
||||||
assert payload["model_loading_attempted"] is False
|
assert payload["model_loading_attempted"] is False
|
||||||
assert payload["gpu_inference_attempted"] is False
|
assert payload["gpu_inference_attempted"] is False
|
||||||
|
|
||||||
|
|
||||||
|
def test_cli_blocks_lineage_validation_exposure_before_model_resolution(
|
||||||
|
tmp_path: Path, monkeypatch
|
||||||
|
) -> None:
|
||||||
|
evaluation = tmp_path / "evaluation"
|
||||||
|
images = evaluation / "images" / "val"
|
||||||
|
images.mkdir(parents=True)
|
||||||
|
(images / "turnhout_r0_c0.png").write_bytes(b"not opened before gate")
|
||||||
|
dataset_yaml = evaluation / "dataset.yaml"
|
||||||
|
dataset_yaml.write_text("path: .\nval: images/val\n", encoding="utf-8")
|
||||||
|
_write_summary(
|
||||||
|
evaluation / "yolo_tile_dataset_summary.json", [("turnhout", "val")]
|
||||||
|
)
|
||||||
|
ancestor = tmp_path / "ancestor.json"
|
||||||
|
_write_summary(ancestor, [("turnhout", "val"), ("mol", "train")])
|
||||||
|
output = tmp_path / "blocked.json"
|
||||||
|
monkeypatch.setattr(
|
||||||
|
sys,
|
||||||
|
"argv",
|
||||||
|
[
|
||||||
|
"evaluate_yolo_checkpoint_matrix.py",
|
||||||
|
"--dataset-yaml",
|
||||||
|
str(dataset_yaml),
|
||||||
|
"--model",
|
||||||
|
str(tmp_path / "model-is-never-resolved.pt"),
|
||||||
|
"--output",
|
||||||
|
str(output),
|
||||||
|
"--background-prefix",
|
||||||
|
"background",
|
||||||
|
"--lineage-summary",
|
||||||
|
str(ancestor),
|
||||||
|
"--require-lineage-sample-independence",
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
|
assert main() == 3
|
||||||
|
payload = json.loads(output.read_text(encoding="utf-8"))
|
||||||
|
assert payload["status"] == "blocked_model_lineage_sample_exposure"
|
||||||
|
assert payload["model_loading_attempted"] is False
|
||||||
|
|||||||
Reference in New Issue
Block a user