diff --git a/CHANGELOG.md b/CHANGELOG.md index 150dc42c..bc592403 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,14 @@ # Changelog +## Sprint 170 Persistent false-negative evidence audit (2026-07-12) + +- Added fixed-threshold evidence portfolio input generation so model comparisons use exactly one matching model/tile/threshold run per AOI. +- Added GIS-aware false-negative audit tooling with WGS84 geodetic areas, building-size buckets and persistent miss detection across multiple persisted QA portfolios. +- Missing, invalid or non-polygon evidence geometry now fails the operator audit explicitly. +- Added readiness compilation and focused regression coverage. +- No API contract, migration, model activation, provider fetching, fake QA output or model download behavior changed. + ## Sprint 169 Filtered YOLO candidate gate and operator hardening (2026-07-12) - Trained and fully gated inactive `geointel-building-yolov8s-aoi1024cleanpx12vis035lowvar512e50-pt` against seven positive AOIs and split pure-empty/sparse-context backgrounds. diff --git a/backend/tests/test_sprint170_detection_false_negative_audit.py b/backend/tests/test_sprint170_detection_false_negative_audit.py new file mode 100644 index 00000000..83515ca8 --- /dev/null +++ b/backend/tests/test_sprint170_detection_false_negative_audit.py @@ -0,0 +1,203 @@ +from __future__ import annotations + +import json +import subprocess +import sys +from pathlib import Path + + +ROOT = Path(__file__).resolve().parents[2] + + +def _polygon(x: float, y: float, size: float) -> dict: + return { + "type": "Polygon", + "coordinates": [ + [ + [x, y], + [x + size, y], + [x + size, y + size], + [x, y + size], + [x, y], + ] + ], + } + + +def _evidence_feature(role: str, source_id: str, geometry: dict) -> dict: + return { + "type": "Feature", + "id": f"{role}:{source_id}", + "properties": { + "qa_evidence_role": role, + "source_feature_id": source_id, + "reference_feature_id": source_id, + }, + "geometry": geometry, + } + + +def test_fixed_threshold_portfolio_inputs_select_one_comparable_run_per_aoi(tmp_path: Path) -> None: + script = ROOT / "scripts" / "build_fixed_threshold_evidence_portfolio_inputs.py" + assert script.exists() + + sample_summaries = [] + for slug in ("geel", "mol"): + summary_path = tmp_path / f"{slug}-quality-summary.json" + items = [ + { + "project_id": f"project-{slug}-low", + "quality_check_id": f"qc-{slug}-low", + "model_asset_id": "model-a", + "threshold": 0.05, + "quality_score": 0.2, + "f1_score": 0.2, + }, + { + "project_id": f"project-{slug}-fixed", + "quality_check_id": f"qc-{slug}-fixed", + "model_asset_id": "model-a", + "threshold": 0.15, + "quality_score": 0.3, + "f1_score": 0.3, + }, + ] + summary_path.write_text(json.dumps({"items": items}), encoding="utf-8") + sample_summaries.append( + {"sample_slug": slug, "summary_path": str(summary_path)} + ) + + multi_summary_path = tmp_path / "multi-sample.json" + multi_summary_path.write_text( + json.dumps( + { + "sample_count": 2, + "sample_summaries": sample_summaries, + "items": [], + } + ), + encoding="utf-8", + ) + output_dir = tmp_path / "fixed-inputs" + result = subprocess.run( + [ + sys.executable, + str(script), + "--multi-sample-summary", + str(multi_summary_path), + "--threshold", + "0.15", + "--model-asset-id", + "model-a", + "--model-sha256", + "abc123", + "--output-dir", + str(output_dir), + ], + cwd=ROOT, + check=True, + capture_output=True, + text=True, + ) + + manifest_path = output_dir / "calibration-evidence-portfolio-manifest.json" + manifest = json.loads(manifest_path.read_text(encoding="utf-8")) + assert manifest["model_asset_id"] == "model-a" + assert manifest["model_sha256"] == "abc123" + assert manifest["fixed_threshold"] == 0.15 + assert [sample["sample_slug"] for sample in manifest["samples"]] == ["geel", "mol"] + for sample in manifest["samples"]: + filtered = json.loads(Path(sample["summary_path"]).read_text(encoding="utf-8")) + assert len(filtered["items"]) == 1 + assert filtered["items"][0]["threshold"] == 0.15 + assert filtered["best_by_score"] == filtered["items"][0] + assert str(manifest_path) in result.stdout + + +def test_false_negative_audit_finds_persistent_reference_misses(tmp_path: Path) -> None: + script = ROOT / "scripts" / "audit_detection_false_negative_evidence.py" + readiness = (ROOT / "scripts" / "run_readiness_check.sh").read_text(encoding="utf-8") + assert script.exists() + assert "py_compile scripts/build_fixed_threshold_evidence_portfolio_inputs.py" in readiness + assert "py_compile scripts/audit_detection_false_negative_evidence.py" in readiness + + portfolio_paths = [] + for label, features in ( + ( + "active", + [ + _evidence_feature("false_negative", "persistent-small", _polygon(5.0, 51.2, 0.0001)), + _evidence_feature("false_negative", "recovered-large", _polygon(5.001, 51.2, 0.0003)), + _evidence_feature("match_reference", "matched", _polygon(5.002, 51.2, 0.0002)), + ], + ), + ( + "candidate", + [ + _evidence_feature("false_negative", "persistent-small", _polygon(5.0, 51.2, 0.0001)), + _evidence_feature("match_reference", "recovered-large", _polygon(5.001, 51.2, 0.0003)), + _evidence_feature("match_reference", "matched", _polygon(5.002, 51.2, 0.0002)), + ], + ), + ): + portfolio_dir = tmp_path / label + evidence_dir = portfolio_dir / "samples" / "geel" / "evidence" + evidence_dir.mkdir(parents=True) + evidence_path = evidence_dir / "calibration_evidence.geojson" + evidence_path.write_text( + json.dumps({"type": "FeatureCollection", "features": features}), + encoding="utf-8", + ) + portfolio_path = portfolio_dir / "calibration_evidence_portfolio.json" + portfolio_path.write_text( + json.dumps( + { + "model_asset_id": f"model-{label}", + "samples": [ + { + "sample_slug": "geel", + "evidence_geojson_path": str(evidence_path), + } + ], + } + ), + encoding="utf-8", + ) + portfolio_paths.append((label, portfolio_path)) + + output_dir = tmp_path / "audit" + subprocess.run( + [ + sys.executable, + str(script), + "--portfolio", + f"active={portfolio_paths[0][1]}", + "--portfolio", + f"candidate={portfolio_paths[1][1]}", + "--output-dir", + str(output_dir), + ], + cwd=ROOT, + check=True, + capture_output=True, + text=True, + ) + + report = json.loads( + (output_dir / "detection_false_negative_audit.json").read_text(encoding="utf-8") + ) + assert report["portfolio_count"] == 2 + sample = report["samples"][0] + assert sample["sample_slug"] == "geel" + assert sample["persistent_false_negative_count"] == 1 + assert sample["persistent_reference_ids"] == ["source:persistent-small"] + active = next(item for item in sample["portfolios"] if item["label"] == "active") + candidate = next(item for item in sample["portfolios"] if item["label"] == "candidate") + assert active["false_negative_count"] == 2 + assert active["matched_reference_count"] == 1 + assert active["false_negative_rate"] == 2 / 3 + assert candidate["false_negative_count"] == 1 + assert candidate["false_negative_rate"] == 1 / 3 + assert active["false_negative_area_m2"]["median"] > 0 + assert report["recommendations"] + assert (output_dir / "detection_false_negative_audit.md").is_file() diff --git a/scripts/README.md b/scripts/README.md index a1390216..a9f83f5a 100644 --- a/scripts/README.md +++ b/scripts/README.md @@ -684,6 +684,41 @@ QA evidence GeoJSON endpoint, writes `calibration_evidence.geojson`, matched references, false positives and false negatives. Set `CALIBRATION_EVIDENCE_MODE=best` to export only the `best_by_score` run. +Build fixed-threshold portfolio inputs when two model runs must be compared at +the same confidence threshold across every AOI: + +```bash +python scripts/build_fixed_threshold_evidence_portfolio_inputs.py \ + --multi-sample-summary artifacts/detection-quality-matrix/multi-sample//multi_sample_quality_summary.json \ + --threshold 0.35 \ + --model-asset-id geointel-building-yolov8s-aoi1024bg512r3e50-pt \ + --model-sha256 e0980572aac90e7efc514608eb16d7de5bfbf27a4bbec04e7bc1bc8c02f9601f \ + --tile-size 512 \ + --tile-overlap 64 \ + --output-dir artifacts/detection-false-negative-review/active-inputs +``` + +The builder selects exactly one persisted QA run per AOI and refuses ambiguous +model/tile/threshold matches. Pass its emitted manifest to +`assemble_detection_calibration_evidence_portfolio.sh` with +`CALIBRATION_EVIDENCE_MODE=all`; each filtered summary contains one run. + +Compare two or more downloaded evidence portfolios with geodetic WGS84 areas: + +```bash +python scripts/audit_detection_false_negative_evidence.py \ + --portfolio active=artifacts/detection-false-negative-review/active/calibration_evidence_portfolio.json \ + --portfolio candidate=artifacts/detection-false-negative-review/candidate/calibration_evidence_portfolio.json \ + --output-dir artifacts/detection-false-negative-review/audit +``` + +The audit reports false-negative rates and area buckets per AOI/model, plus +reference buildings missed by every compared portfolio. Stable +`source_feature_id` values are preferred; a normalized geometry fingerprint is +used only when source IDs are absent. Invalid or missing geometry fails the +audit instead of being silently skipped. The tools do not run inference, +create QA records, mutate model defaults or download data/models. + Docker images install only the GIS runtime by default. To build a local/Tower image with PyTorch/Ultralytics available for the configured-YOLO preflight and runtime path, set: diff --git a/scripts/audit_detection_false_negative_evidence.py b/scripts/audit_detection_false_negative_evidence.py new file mode 100644 index 00000000..7ac089d6 --- /dev/null +++ b/scripts/audit_detection_false_negative_evidence.py @@ -0,0 +1,359 @@ +from __future__ import annotations + +import argparse +import hashlib +import json +import math +import statistics +from datetime import datetime, timezone +from pathlib import Path +from typing import Any + + +AREA_BUCKETS = ( + ("tiny_lt_25_m2", 0.0, 25.0), + ("small_25_100_m2", 25.0, 100.0), + ("medium_100_500_m2", 100.0, 500.0), + ("large_gte_500_m2", 500.0, math.inf), +) + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser( + description="Audit persisted GeoIntel false-negative QA evidence across model portfolios." + ) + parser.add_argument( + "--portfolio", + action="append", + required=True, + help="Label and calibration_evidence_portfolio.json path as label=/path/file.json.", + ) + parser.add_argument("--output-dir", required=True, type=Path) + return parser.parse_args() + + +def load_json(path: Path) -> dict[str, Any]: + if not path.is_file(): + raise SystemExit(f"JSON input is not readable: {path}") + payload = json.loads(path.read_text(encoding="utf-8-sig")) + if not isinstance(payload, dict): + raise SystemExit(f"JSON input must be an object: {path}") + return payload + + +def parse_portfolio_arg(raw: str) -> tuple[str, Path]: + label, separator, path_raw = raw.partition("=") + label = label.strip() + if not separator or not label or not path_raw.strip(): + raise SystemExit("--portfolio must use label=/path/to/calibration_evidence_portfolio.json") + return label, Path(path_raw.strip()).expanduser().resolve() + + +def resolve_evidence_path(portfolio_path: Path, sample: dict[str, Any]) -> Path: + raw = str(sample.get("evidence_geojson_path") or "") + configured = Path(raw).expanduser() + candidates = [configured] + if raw and not configured.is_absolute(): + candidates.append(portfolio_path.parent / configured) + sample_slug = str(sample.get("sample_slug") or "").strip().lower() + candidates.append( + portfolio_path.parent + / "samples" + / sample_slug + / "evidence" + / "calibration_evidence.geojson" + ) + for candidate in candidates: + if candidate.is_file(): + return candidate.resolve() + raise SystemExit(f"Evidence GeoJSON is not readable for {sample_slug}: {raw}") + + +def percentile(values: list[float], fraction: float) -> float | None: + if not values: + return None + ordered = sorted(values) + position = (len(ordered) - 1) * fraction + lower = math.floor(position) + upper = math.ceil(position) + if lower == upper: + return ordered[lower] + return ordered[lower] + (ordered[upper] - ordered[lower]) * (position - lower) + + +def area_stats(values: list[float]) -> dict[str, float | int | None]: + return { + "count": len(values), + "min": min(values) if values else None, + "p10": percentile(values, 0.10), + "median": statistics.median(values) if values else None, + "p90": percentile(values, 0.90), + "max": max(values) if values else None, + } + + +def area_bucket(area_m2: float) -> str: + for label, minimum, maximum in AREA_BUCKETS: + if minimum <= area_m2 < maximum: + return label + raise AssertionError(f"No area bucket for {area_m2}") + + +def stable_reference_id(feature: dict[str, Any], geometry: Any) -> str: + properties = feature.get("properties") or {} + source_feature_id = properties.get("source_feature_id") + if source_feature_id not in (None, ""): + return f"source:{source_feature_id}" + reference_feature_id = properties.get("reference_feature_id") + if reference_feature_id not in (None, ""): + return f"reference:{reference_feature_id}" + normalized_wkb = getattr(geometry.normalize(), "wkb", geometry.wkb) + return f"geometry:{hashlib.sha256(normalized_wkb).hexdigest()}" + + +def audit_feature_collection(payload: dict[str, Any], geod: Any, shape: Any) -> dict[str, Any]: + if payload.get("type") != "FeatureCollection": + raise SystemExit("Evidence GeoJSON must be a FeatureCollection") + features = payload.get("features") or [] + false_negative_ids: set[str] = set() + false_negative_areas: list[float] = [] + matched_reference_areas: list[float] = [] + bucket_counts = { + label: {"false_negative": 0, "matched_reference": 0, "total_reference": 0, "false_negative_rate": None} + for label, _, _ in AREA_BUCKETS + } + + for feature in features: + if not isinstance(feature, dict): + raise SystemExit("Evidence GeoJSON contains a non-object feature") + role = str((feature.get("properties") or {}).get("qa_evidence_role") or "") + if role not in {"false_negative", "match_reference"}: + continue + geometry_payload = feature.get("geometry") + if not isinstance(geometry_payload, dict): + raise SystemExit(f"Evidence feature {feature.get('id')} has no geometry") + geometry = shape(geometry_payload) + if geometry.is_empty or not geometry.is_valid: + raise SystemExit(f"Evidence feature {feature.get('id')} has invalid geometry") + if geometry.geom_type not in {"Polygon", "MultiPolygon"}: + raise SystemExit( + f"Evidence feature {feature.get('id')} must be Polygon or MultiPolygon" + ) + area_m2 = abs(float(geod.geometry_area_perimeter(geometry)[0])) + bucket = area_bucket(area_m2) + bucket_role = "false_negative" if role == "false_negative" else "matched_reference" + bucket_counts[bucket][bucket_role] += 1 + bucket_counts[bucket]["total_reference"] += 1 + if role == "false_negative": + false_negative_ids.add(stable_reference_id(feature, geometry)) + false_negative_areas.append(area_m2) + else: + matched_reference_areas.append(area_m2) + + for values in bucket_counts.values(): + total = values["total_reference"] + values["false_negative_rate"] = values["false_negative"] / total if total else None + + total_reference = len(false_negative_areas) + len(matched_reference_areas) + return { + "false_negative_ids": false_negative_ids, + "false_negative_count": len(false_negative_areas), + "matched_reference_count": len(matched_reference_areas), + "total_reference_count": total_reference, + "false_negative_rate": len(false_negative_areas) / total_reference if total_reference else None, + "false_negative_area_m2": area_stats(false_negative_areas), + "matched_reference_area_m2": area_stats(matched_reference_areas), + "area_buckets": bucket_counts, + } + + +def build_recommendations(samples: list[dict[str, Any]]) -> list[str]: + recommendations: list[str] = [] + persistent_total = sum(sample["persistent_false_negative_count"] for sample in samples) + if persistent_total: + recommendations.append( + f"Prioritize targeted positive sampling for {persistent_total} reference buildings missed by every compared portfolio." + ) + ranked = sorted( + samples, + key=lambda sample: max( + (item.get("false_negative_rate") or 0.0 for item in sample["portfolios"]), + default=0.0, + ), + reverse=True, + ) + if ranked: + recommendations.append( + "Expand or oversample the weakest AOIs first: " + + ", ".join(sample["sample_slug"] for sample in ranked[:3]) + + "." + ) + small_bias_samples: list[str] = [] + for sample in samples: + for portfolio in sample["portfolios"]: + buckets = portfolio["area_buckets"] + small_total = sum( + buckets[key]["total_reference"] + for key in ("tiny_lt_25_m2", "small_25_100_m2") + ) + small_fn = sum( + buckets[key]["false_negative"] + for key in ("tiny_lt_25_m2", "small_25_100_m2") + ) + larger_total = sum( + buckets[key]["total_reference"] + for key in ("medium_100_500_m2", "large_gte_500_m2") + ) + larger_fn = sum( + buckets[key]["false_negative"] + for key in ("medium_100_500_m2", "large_gte_500_m2") + ) + small_rate = small_fn / small_total if small_total else 0.0 + larger_rate = larger_fn / larger_total if larger_total else 0.0 + if small_total >= 5 and small_rate >= larger_rate + 0.15: + small_bias_samples.append(sample["sample_slug"]) + break + if small_bias_samples: + recommendations.append( + "Review small-building label retention and pixel-size gates for: " + + ", ".join(sorted(set(small_bias_samples))) + + "." + ) + if not recommendations: + recommendations.append( + "No dominant false-negative pattern was found; preserve current data gates and collect more independent AOIs." + ) + return recommendations + + +def run_audit(portfolio_args: list[str], output_dir: Path) -> tuple[Path, Path]: + try: + from pyproj import Geod + from shapely.geometry import shape + except ImportError as exc: + raise SystemExit( + "False-negative GIS audit requires the existing GeoIntel GIS extras (pyproj and shapely)." + ) from exc + + parsed_portfolios = [parse_portfolio_arg(raw) for raw in portfolio_args] + labels = [label for label, _ in parsed_portfolios] + if len(labels) != len(set(labels)): + raise SystemExit("Every --portfolio label must be unique") + geod = Geod(ellps="WGS84") + portfolio_samples: dict[str, dict[str, dict[str, Any]]] = {} + portfolio_meta: list[dict[str, Any]] = [] + + for label, portfolio_path in parsed_portfolios: + portfolio = load_json(portfolio_path) + samples = portfolio.get("samples") or [] + if not isinstance(samples, list) or not samples: + raise SystemExit(f"Evidence portfolio has no samples: {portfolio_path}") + sample_results: dict[str, dict[str, Any]] = {} + for sample in samples: + sample_slug = str(sample.get("sample_slug") or "").strip().lower() + if not sample_slug: + raise SystemExit(f"Evidence portfolio has a sample without sample_slug: {portfolio_path}") + evidence_path = resolve_evidence_path(portfolio_path, sample) + result = audit_feature_collection(load_json(evidence_path), geod, shape) + result.update( + { + "label": label, + "model_asset_id": portfolio.get("model_asset_id"), + "evidence_geojson_path": str(evidence_path), + } + ) + sample_results[sample_slug] = result + portfolio_samples[label] = sample_results + portfolio_meta.append( + { + "label": label, + "path": str(portfolio_path), + "model_asset_id": portfolio.get("model_asset_id"), + "sample_count": len(sample_results), + } + ) + + expected_slugs = set(next(iter(portfolio_samples.values()))) + for label, samples in portfolio_samples.items(): + if set(samples) != expected_slugs: + raise SystemExit( + f"Portfolio {label} has different AOIs; fixed-threshold comparisons require identical samples" + ) + + sample_reports: list[dict[str, Any]] = [] + for sample_slug in sorted(expected_slugs): + portfolio_rows = [] + false_negative_sets = [] + for label, _ in parsed_portfolios: + raw = portfolio_samples[label][sample_slug] + false_negative_sets.append(raw["false_negative_ids"]) + portfolio_rows.append( + {key: value for key, value in raw.items() if key != "false_negative_ids"} + ) + persistent_ids = sorted(set.intersection(*false_negative_sets)) + sample_reports.append( + { + "sample_slug": sample_slug, + "persistent_false_negative_count": len(persistent_ids), + "persistent_reference_ids": persistent_ids, + "portfolios": portfolio_rows, + } + ) + + report = { + "generated_at": datetime.now(timezone.utc).isoformat(), + "schema_version": 1, + "input_crs": "EPSG:4326", + "area_method": "WGS84 geodesic area via pyproj.Geod", + "portfolio_count": len(portfolio_meta), + "portfolios": portfolio_meta, + "sample_count": len(sample_reports), + "samples": sample_reports, + "recommendations": build_recommendations(sample_reports), + } + output_dir = output_dir.expanduser().resolve() + output_dir.mkdir(parents=True, exist_ok=True) + json_path = output_dir / "detection_false_negative_audit.json" + json_path.write_text(json.dumps(report, indent=2, sort_keys=True), encoding="utf-8") + + lines = [ + "# Detection false-negative evidence audit", + "", + f"- Generated: {report['generated_at']}", + f"- Portfolios: {report['portfolio_count']}", + f"- AOIs: {report['sample_count']}", + f"- Area method: {report['area_method']}", + "", + "## AOI comparison", + "", + "| AOI | Persistent misses | " + + " | ".join(f"{label} FN rate" for label, _ in parsed_portfolios) + + " |", + "|---|---:|" + "---:|" * len(parsed_portfolios), + ] + for sample in sample_reports: + rates = [ + f"{(row['false_negative_rate'] or 0.0):.3f}" + for row in sample["portfolios"] + ] + lines.append( + f"| {sample['sample_slug']} | {sample['persistent_false_negative_count']} | " + + " | ".join(rates) + + " |" + ) + lines.extend(["", "## Recommended data actions", ""]) + lines.extend(f"- {item}" for item in report["recommendations"]) + markdown_path = output_dir / "detection_false_negative_audit.md" + markdown_path.write_text("\n".join(lines) + "\n", encoding="utf-8") + return json_path, markdown_path + + +def main() -> None: + args = parse_args() + json_path, markdown_path = run_audit(args.portfolio, args.output_dir) + print(f"False-negative audit JSON: {json_path}") + print(f"False-negative audit Markdown: {markdown_path}") + + +if __name__ == "__main__": + main() diff --git a/scripts/build_fixed_threshold_evidence_portfolio_inputs.py b/scripts/build_fixed_threshold_evidence_portfolio_inputs.py new file mode 100644 index 00000000..d66a4ea1 --- /dev/null +++ b/scripts/build_fixed_threshold_evidence_portfolio_inputs.py @@ -0,0 +1,168 @@ +from __future__ import annotations + +import argparse +import json +from pathlib import Path +from typing import Any + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser( + description="Build comparable fixed-threshold inputs for a GeoIntel calibration evidence portfolio." + ) + parser.add_argument("--multi-sample-summary", required=True, type=Path) + parser.add_argument("--threshold", required=True, type=float) + parser.add_argument("--model-asset-id") + parser.add_argument("--model-sha256") + parser.add_argument("--tile-size", type=int) + parser.add_argument("--tile-overlap", type=int) + parser.add_argument("--output-dir", required=True, type=Path) + return parser.parse_args() + + +def load_json(path: Path) -> dict[str, Any]: + if not path.is_file(): + raise SystemExit(f"JSON input is not readable: {path}") + payload = json.loads(path.read_text(encoding="utf-8-sig")) + if not isinstance(payload, dict): + raise SystemExit(f"JSON input must be an object: {path}") + return payload + + +def resolve_summary_path(raw: str, multi_summary_path: Path) -> Path: + path = Path(raw).expanduser() + candidates = [path] + if not path.is_absolute(): + candidates.extend((multi_summary_path.parent / path, Path.cwd() / path)) + for candidate in candidates: + if candidate.is_file(): + return candidate.resolve() + raise SystemExit(f"Sample summary is not readable: {raw}") + + +def matches_run( + item: dict[str, Any], + *, + threshold: float, + model_asset_id: str | None, + tile_size: int | None, + tile_overlap: int | None, +) -> bool: + try: + item_threshold = float(item.get("threshold")) + except (TypeError, ValueError): + return False + if abs(item_threshold - threshold) > 1e-9: + return False + if model_asset_id and item.get("model_asset_id") != model_asset_id: + return False + if tile_size is not None and item.get("tile_size") != tile_size: + return False + if tile_overlap is not None and item.get("tile_overlap") != tile_overlap: + return False + return True + + +def build_inputs(args: argparse.Namespace) -> Path: + multi_summary_path = args.multi_sample_summary.expanduser().resolve() + multi_summary = load_json(multi_summary_path) + sample_summaries = multi_summary.get("sample_summaries") or [] + if not isinstance(sample_summaries, list) or not sample_summaries: + raise SystemExit("Multi-sample summary has no sample_summaries") + + output_dir = args.output_dir.expanduser().resolve() + samples_dir = output_dir / "samples" + samples_dir.mkdir(parents=True, exist_ok=True) + manifest_samples: list[dict[str, Any]] = [] + selected_model_ids: set[str] = set() + + for sample in sample_summaries: + if not isinstance(sample, dict): + raise SystemExit("Every sample summary entry must be an object") + sample_slug = str(sample.get("sample_slug") or "").strip().lower() + if not sample_slug: + raise SystemExit("Sample summary entry is missing sample_slug") + source_summary_path = resolve_summary_path( + str(sample.get("summary_path") or ""), multi_summary_path + ) + source_summary = load_json(source_summary_path) + items = source_summary.get("items") or [] + matches = [ + item + for item in items + if isinstance(item, dict) + and matches_run( + item, + threshold=args.threshold, + model_asset_id=args.model_asset_id, + tile_size=args.tile_size, + tile_overlap=args.tile_overlap, + ) + ] + if len(matches) != 1: + raise SystemExit( + f"Expected exactly one fixed-threshold run for {sample_slug}; found {len(matches)}" + ) + selected = dict(matches[0]) + selected_model_id = str(selected.get("model_asset_id") or "").strip() + if not selected_model_id: + raise SystemExit(f"Selected run for {sample_slug} has no model_asset_id") + selected_model_ids.add(selected_model_id) + + filtered_summary = dict(source_summary) + filtered_summary.update( + { + "items": [selected], + "best_by_score": selected, + "best_by_recall": selected, + "best_by_precision": selected, + "fixed_threshold": args.threshold, + "source_summary_path": str(source_summary_path), + } + ) + sample_dir = samples_dir / sample_slug + sample_dir.mkdir(parents=True, exist_ok=True) + filtered_path = sample_dir / "quality_matrix_summary.json" + filtered_path.write_text( + json.dumps(filtered_summary, indent=2, sort_keys=True), encoding="utf-8" + ) + manifest_samples.append( + { + "sample_slug": sample_slug, + "aoi_label": sample_slug.replace("_", " ").title(), + "summary_path": str(filtered_path), + "operator_notes": ( + f"Fixed threshold {args.threshold:g}; source {source_summary_path}." + ), + } + ) + + if len(selected_model_ids) != 1: + raise SystemExit( + f"Fixed-threshold runs contain multiple model assets: {sorted(selected_model_ids)}" + ) + selected_model_id = next(iter(selected_model_ids)) + manifest = { + "portfolio_name": ( + f"GeoIntel fixed-threshold false-negative review: {selected_model_id} @ {args.threshold:g}" + ), + "model_asset_id": selected_model_id, + "model_sha256": args.model_sha256, + "fixed_threshold": args.threshold, + "notes": "Comparable per-AOI persisted QA evidence; no inference is run by this input builder.", + "samples": manifest_samples, + } + manifest_path = output_dir / "calibration-evidence-portfolio-manifest.json" + manifest_path.write_text( + json.dumps(manifest, indent=2, sort_keys=True), encoding="utf-8" + ) + return manifest_path + + +def main() -> None: + manifest_path = build_inputs(parse_args()) + print(f"Fixed-threshold portfolio manifest: {manifest_path}") + + +if __name__ == "__main__": + main() diff --git a/scripts/run_readiness_check.sh b/scripts/run_readiness_check.sh index 7c9ff9b5..d05e13e4 100755 --- a/scripts/run_readiness_check.sh +++ b/scripts/run_readiness_check.sh @@ -47,6 +47,8 @@ ${PYTHON_BIN} -m py_compile scripts/export_operator_yolo_tile_dataset.py ${PYTHON_BIN} -m py_compile scripts/audit_operator_yolo_dataset_quality.py ${PYTHON_BIN} -m py_compile scripts/build_detection_model_promotion_report.py ${PYTHON_BIN} -m py_compile scripts/build_background_corpus_split_report.py +${PYTHON_BIN} -m py_compile scripts/build_fixed_threshold_evidence_portfolio_inputs.py +${PYTHON_BIN} -m py_compile scripts/audit_detection_false_negative_evidence.py ${PYTHON_BIN} -m py_compile scripts/activate_promoted_yolo_candidate.py ${PYTHON_BIN} -m py_compile scripts/cleanup_demo_artifacts.py ${PYTHON_BIN} -m py_compile backend/scripts/cleanup_demo_artifacts.py