From 30f6f707c9bb60cf64b4ec8db6c3988a201c2fc9 Mon Sep 17 00:00:00 2001 From: Jens Date: Sun, 9 Aug 2026 14:31:43 +0200 Subject: [PATCH] audit building checkpoints and preserve production gate --- ...9-v68-checkpoint-and-threshold-review.json | 93 ++++++++ ...026-08-09-ai-assisted-checkpoint-review.md | 64 ++++++ scripts/README.md | 6 + .../evaluate_belgium_building_candidate.py | 5 +- scripts/evaluate_yolo_checkpoint_matrix.py | 206 ++++++++++++++++++ 5 files changed, 373 insertions(+), 1 deletion(-) create mode 100644 artifacts/evidence/accuracy/model-training/20260809-v68-checkpoint-and-threshold-review.json create mode 100644 docs/reviews/2026-08-09-ai-assisted-checkpoint-review.md create mode 100644 scripts/evaluate_yolo_checkpoint_matrix.py diff --git a/artifacts/evidence/accuracy/model-training/20260809-v68-checkpoint-and-threshold-review.json b/artifacts/evidence/accuracy/model-training/20260809-v68-checkpoint-and-threshold-review.json new file mode 100644 index 00000000..3670616e --- /dev/null +++ b/artifacts/evidence/accuracy/model-training/20260809-v68-checkpoint-and-threshold-review.json @@ -0,0 +1,93 @@ +{ + "schema_version": 1, + "generated_at": "2026-08-09T12:30:00Z", + "run_id": "building-kempen-v68-checkpoint-and-threshold-review-r1", + "status": "completed_no_production_change", + "review_type": "ai_assisted_visual_and_metric_review", + "claim_boundary": "Non-protected Kempen validation only. This is not a formal human sign-off, protected-test result, national accuracy claim or authorization to infer outside the configured geographic scope.", + "runtime": { + "device": "cuda:0", + "gpu": "NVIDIA GeForce RTX 4080 SUPER", + "torch": "2.11.0+cu128", + "ultralytics": "8.4.99" + }, + "checkpoint_matrix": { + "newer_family_manifest": { + "path": "/app/storage/training/building-kempen-v68-checkpoint-matrix-r1/checkpoint-matrix.json", + "sha256": "a17640f8190915bd38bed98b2b6d086e1020987842b48bffadec2e20163f8647" + }, + "original_family_manifest": { + "path": "/app/storage/training/building-kempen-v68-checkpoint-matrix-original-r1/checkpoint-matrix.json", + "sha256": "face91b3b0cb4d332dff1f176051dc46dd931275fceb91fe465ad010bbd675eb" + }, + "active": { + "model_sha256": "a9088b8491dfae36694b53e9e9406cb4e3511d334a5712fa34f75078a47759c1", + "precision": 0.541626, + "recall": 0.429757, + "map50": 0.345426, + "map50_95": 0.140719, + "pure_background_detections_at_confidence_0_15": 0 + }, + "best_validation_challenger": { + "model_sha256": "038f1f97a6afd534f29e1f392a730a58207b928ca01e31ab8d8fed6106705820", + "precision": 0.573691, + "recall": 0.454274, + "map50": 0.368177, + "map50_95": 0.155438, + "pure_background_detections_at_confidence_0_15": 2 + } + }, + "threshold_review": { + "split": "val", + "image_count": 36, + "positive_image_count": 18, + "pure_background_image_count": 18, + "match_iou": 0.25, + "active_manifest": { + "path": "/app/storage/training/building-kempen-v68-reviewedexp6-threshold-r1/active-thresholds.json", + "sha256": "3462c3f4be480e34976084901c8dff1141168d818e1c6a74f86d26ae5872391" + }, + "challenger_manifest": { + "path": "/app/storage/training/building-kempen-v68-reviewedexp6-threshold-r1/reviewedexp6-thresholds.json", + "sha256": "c810aa0372a2767d538c9454f6f9f7a54338e61650756e35fdd1750a57684756" + }, + "confidence_0_25": { + "active": {"tp": 3830, "fp": 2584, "fn": 1856, "precision": 0.597131, "recall": 0.673584, "f1": 0.633058, "pure_background_detections": 0}, + "challenger": {"tp": 4005, "fp": 2774, "fn": 1681, "precision": 0.590795, "recall": 0.704362, "f1": 0.642599, "pure_background_detections": 0} + }, + "confidence_0_30": { + "active": {"precision": 0.713823, "recall": 0.543088, "f1": 0.616860, "pure_background_detections": 0}, + "challenger": {"precision": 0.697588, "recall": 0.595146, "f1": 0.642308, "pure_background_detections": 0} + } + }, + "visual_review": { + "active_contact_sheet": { + "path": "/app/storage/training/building-kempen-v68-reviewedexp6-visual-r1/active/candidate_error_contact_sheet.png", + "sha256": "609cffab8d5f47d93ec482a3ac1b960dac35645f65906ce4344cd50e9742b5f3" + }, + "challenger_contact_sheet": { + "path": "/app/storage/training/building-kempen-v68-reviewedexp6-visual-r1/reviewedexp6/candidate_error_contact_sheet.png", + "sha256": "65ac7a24ee2b51bcbaa83d5972639602b4b4d5fd71a0a3aab2ef99510b5465f6" + }, + "observations": [ + "The challenger reduces missed-building pressure in both dense Turnhout and lower-density Westerlo examples.", + "Dense urban overlays still contain many low-IoU disagreements caused by visible roof position and shape versus GRB ground-level building footprints.", + "AI-assisted visual inspection cannot convert ambiguous roof-footprint disagreements into accepted human labels." + ] + }, + "production_gate": { + "attempted_action": "Run the existing seven-zone production-API benchmark with challenger confidence 0.25.", + "result": "blocked_fail_closed_before_inference", + "reason": "The historical challenger predates the mandatory immutable runtime-provenance sidecar and governed database snapshot binding.", + "unsafe_bypass_used": false, + "production_model_changed": false, + "active_model_sha256_after_decision": "a9088b8491dfae36694b53e9e9406cb4e3511d334a5712fa34f75078a47759c1", + "active_image_after_decision": "geointel-all-in-one:0209167cfd37-wip25e7de62cde9-ai", + "server_health_after_decision": "healthy" + }, + "decision": { + "v67_candidate": "rejected", + "reviewedexp6_at_confidence_0_25": "retain_as_unpromoted_validation_challenger", + "reason": "The challenger is measurably stronger on the non-protected validation split and clears its local background controls at 0.25, but production promotion remains blocked until its original training lineage is migrated into an exact sidecar and governed source snapshot and the existing seven-zone workflow is rerun." + } +} diff --git a/docs/reviews/2026-08-09-ai-assisted-checkpoint-review.md b/docs/reviews/2026-08-09-ai-assisted-checkpoint-review.md new file mode 100644 index 00000000..2a8bc5d0 --- /dev/null +++ b/docs/reviews/2026-08-09-ai-assisted-checkpoint-review.md @@ -0,0 +1,64 @@ +# AI-assisted building checkpoint review — 2026-08-09 + +## Outcome + +GeoIntel still runs the proven active building checkpoint. A new 12-epoch GPU +fine-tune was rejected because it reduced mAP and introduced three detections +on pure-background validation tiles. A wider checkpoint comparison identified +the older `reviewedexp6` checkpoint as the strongest non-protected validation +challenger, but it was not promoted. + +The production runtime was verified healthy after the review with: + +- image `geointel-all-in-one:0209167cfd37-wip25e7de62cde9-ai`; +- active weights SHA-256 + `a9088b8491dfae36694b53e9e9406cb4e3511d334a5712fa34f75078a47759c1`; +- NVIDIA RTX 4080 SUPER inference; and +- the existing Kempen validation-scope enforcement. + +## Metric review + +The checkpoint matrix used only the 36-image non-protected validation split: +18 positive images and 18 declared pure-background images. The active model +measured mAP50 `0.345426` and mAP50-95 `0.140719`. The `reviewedexp6` +challenger measured `0.368177` and `0.155438` respectively. + +At confidence `0.25` and match IoU `0.25`, the active model measured F1 +`0.633058`; the challenger measured `0.642599`. Both produced zero detections +on the 18 pure-background validation tiles at this threshold. At confidence +`0.15`, however, the challenger still reproduces the two background detections +that blocked its July promotion. + +## Visual review + +The error sheets were inspected at original resolution. Green denotes a +matched prediction, red a false positive, magenta a false negative and yellow +the reference footprint. The challenger reduces missed-building pressure, +particularly in Westerlo, but dense Turnhout tiles still show extensive +low-IoU disagreement. + +A major part of that disagreement is not safely resolved by additional epochs: +the aerial image shows a roof while GRB GBG represents the building footprint +at ground level. Perspective, roof overhang and acquisition-date differences +can therefore shift visible roofs relative to the authoritative footprint. +These cases remain review candidates; they were not relabelled automatically. + +This is explicitly an AI-assisted inspection, not a signed human review. No +human-review gate or label-acceptance record was fabricated. + +## Fail-closed production decision + +The existing seven-zone API benchmark was started with the challenger at +confidence `0.25`, but the current provenance gate stopped it before model +loading. The historical checkpoint predates the required neighbouring +`.geointel-model.json` runtime manifest and its governed database snapshot. +That control was not bypassed and no production inference from the unbound +checkpoint was accepted. + +The challenger may only be reconsidered after its original training evidence +is migrated without invented lineage, followed by the same seven positive +zones and three pure-background controls. Until then, the active checkpoint is +the only safe production choice. + +Machine-readable hashes, metrics, paths and the exact decision are in +`artifacts/evidence/accuracy/model-training/20260809-v68-checkpoint-and-threshold-review.json`. diff --git a/scripts/README.md b/scripts/README.md index 06640967..078e6098 100644 --- a/scripts/README.md +++ b/scripts/README.md @@ -1,5 +1,11 @@ # Scripts +`evaluate_yolo_checkpoint_matrix.py` compares an explicit list of local YOLO +weights on one declared, non-protected `val` split using CUDA. It records exact +model and dataset hashes, standard Ultralytics detection metrics and a separate +pure-background detection count. The output claim is validation ranking only; +the script neither reads protected test data nor promotes a model. + Setup-, import-, demo- en maintenance-scripts voor GeoIntel. ## WALOUS source provisioning diff --git a/scripts/evaluate_belgium_building_candidate.py b/scripts/evaluate_belgium_building_candidate.py index cb484bd1..9469de59 100644 --- a/scripts/evaluate_belgium_building_candidate.py +++ b/scripts/evaluate_belgium_building_candidate.py @@ -227,7 +227,10 @@ def main() -> int: summary = json.loads(args.summary.read_text(encoding="utf-8")) manifest = json.loads(args.corpus_manifest.read_text(encoding="utf-8")) - regions = {item["sample_slug"]: item["region"] for item in manifest["samples"]} + # Legacy Kempen manifests predate the national region field. They remain + # valid for an explicitly non-routed local validation run; regional routing + # below still fails closed when a requested region is absent. + regions = {item["sample_slug"]: str(item.get("region") or "unknown") for item in manifest["samples"]} pure_empty_slugs = { item["sample_slug"] for item in manifest["samples"] diff --git a/scripts/evaluate_yolo_checkpoint_matrix.py b/scripts/evaluate_yolo_checkpoint_matrix.py new file mode 100644 index 00000000..8539eae0 --- /dev/null +++ b/scripts/evaluate_yolo_checkpoint_matrix.py @@ -0,0 +1,206 @@ +#!/usr/bin/env python3 +"""Compare local detection checkpoints on one non-protected YOLO validation split.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import re +from datetime import UTC, datetime +from pathlib import Path +from typing import Any + + +def sha256_file(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def safe_name(path: Path) -> str: + value = re.sub(r"[^a-z0-9]+", "-", path.stem.casefold()).strip("-") + return (value or path.stem.casefold())[:80] + + +def validation_images(dataset_yaml: Path) -> list[Path]: + import yaml + + payload = yaml.safe_load(dataset_yaml.read_text(encoding="utf-8")) + if not isinstance(payload, dict): + raise ValueError("dataset YAML must contain a mapping") + source = payload.get("val") + root = Path(str(payload.get("path") or dataset_yaml.parent)).expanduser() + if not root.is_absolute(): + root = (dataset_yaml.parent / root).resolve(strict=False) + if not isinstance(source, str) or not source.strip(): + raise ValueError("dataset YAML requires one explicit val image directory") + directory = Path(source).expanduser() + if not directory.is_absolute(): + directory = root / directory + if not directory.is_dir(): + raise ValueError(f"validation image directory is unavailable: {directory}") + images = sorted( + path + for path in directory.iterdir() + if path.is_file() and path.suffix.casefold() in {".jpg", ".jpeg", ".png", ".tif", ".tiff"} + ) + if not images: + raise ValueError("validation image directory is empty") + return images + + +def metric_value(metrics: Any, attribute: str) -> float: + value = getattr(metrics.box, attribute) + return float(value) + + +def background_detection_count( + model: Any, + images: list[Path], + *, + prefixes: tuple[str, ...], + confidence: float, + image_size: int, + device: str, +) -> tuple[int, int]: + selected = [path for path in images if path.stem.casefold().startswith(prefixes)] + if not selected: + raise ValueError("no validation images match the declared pure-background prefixes") + count = 0 + for start in range(0, len(selected), 16): + results = model.predict( + [str(path) for path in selected[start : start + 16]], + conf=confidence, + iou=0.7, + max_det=1000, + imgsz=image_size, + device=device, + verbose=False, + ) + count += sum(len(result.boxes) for result in results) + return len(selected), count + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--dataset-yaml", type=Path, required=True) + parser.add_argument("--model", type=Path, action="append", required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--background-prefix", action="append", required=True) + parser.add_argument("--background-confidence", type=float, default=0.15) + parser.add_argument("--device", default="cuda:0") + parser.add_argument("--imgsz", type=int, default=640) + parser.add_argument("--batch", type=int, default=8) + args = parser.parse_args() + + if args.output.exists(): + parser.error(f"output already exists: {args.output}") + if not 0.0 < args.background_confidence < 1.0: + parser.error("--background-confidence must be between zero and one") + dataset_yaml = args.dataset_yaml.expanduser().resolve(strict=True) + images = validation_images(dataset_yaml) + prefixes = tuple(value.casefold() for value in args.background_prefix) + + import torch + from ultralytics import YOLO + + if not torch.cuda.is_available() or not args.device.casefold().startswith("cuda"): + raise SystemExit("checkpoint matrix requires the configured CUDA device") + + output_root = args.output.parent / f"{args.output.stem}-runs" + rows: list[dict[str, Any]] = [] + seen_hashes: set[str] = set() + for raw_model in args.model: + model_path = raw_model.expanduser().resolve(strict=True) + model_sha256 = sha256_file(model_path) + if model_sha256 in seen_hashes: + continue + seen_hashes.add(model_sha256) + row: dict[str, Any] = { + "model_path": str(model_path), + "model_sha256": model_sha256, + "size_bytes": model_path.stat().st_size, + } + model = None + try: + model = YOLO(str(model_path)) + metrics = model.val( + data=str(dataset_yaml), + split="val", + imgsz=args.imgsz, + batch=args.batch, + workers=0, + device=args.device, + project=str(output_root), + name=safe_name(model_path), + exist_ok=False, + plots=False, + save_json=False, + verbose=False, + ) + background_images, background_detections = background_detection_count( + model, + images, + prefixes=prefixes, + confidence=args.background_confidence, + image_size=args.imgsz, + device=args.device, + ) + row.update( + { + "status": "ok", + "precision": metric_value(metrics, "mp"), + "recall": metric_value(metrics, "mr"), + "map50": metric_value(metrics, "map50"), + "map50_95": metric_value(metrics, "map"), + "pure_background_image_count": background_images, + "pure_background_detection_count": background_detections, + } + ) + except Exception as exc: # preserve the complete attempted matrix + row.update({"status": "error", "error_type": type(exc).__name__, "error": str(exc)[:1000]}) + finally: + del model + if torch.cuda.is_available(): + torch.cuda.empty_cache() + rows.append(row) + print(json.dumps(row, sort_keys=True), flush=True) + + successful = [row for row in rows if row["status"] == "ok"] + successful.sort( + key=lambda row: ( + int(row["pure_background_detection_count"] == 0), + row["map50_95"], + row["map50"], + row["precision"], + row["recall"], + ), + reverse=True, + ) + payload = { + "schema_version": 1, + "generated_at": datetime.now(UTC).isoformat(), + "status": "ok" if successful else "failed", + "claim_boundary": "Non-protected validation ranking only; no test, challenge or promotion claim.", + "dataset_yaml": str(dataset_yaml), + "dataset_yaml_sha256": sha256_file(dataset_yaml), + "validation_image_count": len(images), + "pure_background_prefixes": list(prefixes), + "pure_background_confidence": args.background_confidence, + "device": args.device, + "torch_version": torch.__version__, + "candidate_count": len(rows), + "successful_candidate_count": len(successful), + "ranking": successful, + "attempts": rows, + } + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8") + return 0 if successful else 2 + + +if __name__ == "__main__": + raise SystemExit(main())