Add persistent false negative evidence audit
GeoIntel CI / docs-smoke (push) Has been cancelled
GeoIntel CI / contract-smoke (push) Has been cancelled

This commit is contained in:
Codex
2026-07-12 23:30:52 +02:00
parent 1a8e54e21a
commit 53cd38a5b2
6 changed files with 775 additions and 0 deletions
+8
View File
@@ -7,6 +7,14 @@
# Changelog
## Sprint 170 Persistent false-negative evidence audit (2026-07-12)
- Added fixed-threshold evidence portfolio input generation so model comparisons use exactly one matching model/tile/threshold run per AOI.
- Added GIS-aware false-negative audit tooling with WGS84 geodetic areas, building-size buckets and persistent miss detection across multiple persisted QA portfolios.
- Missing, invalid or non-polygon evidence geometry now fails the operator audit explicitly.
- Added readiness compilation and focused regression coverage.
- No API contract, migration, model activation, provider fetching, fake QA output or model download behavior changed.
## Sprint 169 Filtered YOLO candidate gate and operator hardening (2026-07-12)
- Trained and fully gated inactive `geointel-building-yolov8s-aoi1024cleanpx12vis035lowvar512e50-pt` against seven positive AOIs and split pure-empty/sparse-context backgrounds.
@@ -0,0 +1,203 @@
from __future__ import annotations
import json
import subprocess
import sys
from pathlib import Path
ROOT = Path(__file__).resolve().parents[2]
def _polygon(x: float, y: float, size: float) -> dict:
return {
"type": "Polygon",
"coordinates": [
[
[x, y],
[x + size, y],
[x + size, y + size],
[x, y + size],
[x, y],
]
],
}
def _evidence_feature(role: str, source_id: str, geometry: dict) -> dict:
return {
"type": "Feature",
"id": f"{role}:{source_id}",
"properties": {
"qa_evidence_role": role,
"source_feature_id": source_id,
"reference_feature_id": source_id,
},
"geometry": geometry,
}
def test_fixed_threshold_portfolio_inputs_select_one_comparable_run_per_aoi(tmp_path: Path) -> None:
script = ROOT / "scripts" / "build_fixed_threshold_evidence_portfolio_inputs.py"
assert script.exists()
sample_summaries = []
for slug in ("geel", "mol"):
summary_path = tmp_path / f"{slug}-quality-summary.json"
items = [
{
"project_id": f"project-{slug}-low",
"quality_check_id": f"qc-{slug}-low",
"model_asset_id": "model-a",
"threshold": 0.05,
"quality_score": 0.2,
"f1_score": 0.2,
},
{
"project_id": f"project-{slug}-fixed",
"quality_check_id": f"qc-{slug}-fixed",
"model_asset_id": "model-a",
"threshold": 0.15,
"quality_score": 0.3,
"f1_score": 0.3,
},
]
summary_path.write_text(json.dumps({"items": items}), encoding="utf-8")
sample_summaries.append(
{"sample_slug": slug, "summary_path": str(summary_path)}
)
multi_summary_path = tmp_path / "multi-sample.json"
multi_summary_path.write_text(
json.dumps(
{
"sample_count": 2,
"sample_summaries": sample_summaries,
"items": [],
}
),
encoding="utf-8",
)
output_dir = tmp_path / "fixed-inputs"
result = subprocess.run(
[
sys.executable,
str(script),
"--multi-sample-summary",
str(multi_summary_path),
"--threshold",
"0.15",
"--model-asset-id",
"model-a",
"--model-sha256",
"abc123",
"--output-dir",
str(output_dir),
],
cwd=ROOT,
check=True,
capture_output=True,
text=True,
)
manifest_path = output_dir / "calibration-evidence-portfolio-manifest.json"
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
assert manifest["model_asset_id"] == "model-a"
assert manifest["model_sha256"] == "abc123"
assert manifest["fixed_threshold"] == 0.15
assert [sample["sample_slug"] for sample in manifest["samples"]] == ["geel", "mol"]
for sample in manifest["samples"]:
filtered = json.loads(Path(sample["summary_path"]).read_text(encoding="utf-8"))
assert len(filtered["items"]) == 1
assert filtered["items"][0]["threshold"] == 0.15
assert filtered["best_by_score"] == filtered["items"][0]
assert str(manifest_path) in result.stdout
def test_false_negative_audit_finds_persistent_reference_misses(tmp_path: Path) -> None:
script = ROOT / "scripts" / "audit_detection_false_negative_evidence.py"
readiness = (ROOT / "scripts" / "run_readiness_check.sh").read_text(encoding="utf-8")
assert script.exists()
assert "py_compile scripts/build_fixed_threshold_evidence_portfolio_inputs.py" in readiness
assert "py_compile scripts/audit_detection_false_negative_evidence.py" in readiness
portfolio_paths = []
for label, features in (
(
"active",
[
_evidence_feature("false_negative", "persistent-small", _polygon(5.0, 51.2, 0.0001)),
_evidence_feature("false_negative", "recovered-large", _polygon(5.001, 51.2, 0.0003)),
_evidence_feature("match_reference", "matched", _polygon(5.002, 51.2, 0.0002)),
],
),
(
"candidate",
[
_evidence_feature("false_negative", "persistent-small", _polygon(5.0, 51.2, 0.0001)),
_evidence_feature("match_reference", "recovered-large", _polygon(5.001, 51.2, 0.0003)),
_evidence_feature("match_reference", "matched", _polygon(5.002, 51.2, 0.0002)),
],
),
):
portfolio_dir = tmp_path / label
evidence_dir = portfolio_dir / "samples" / "geel" / "evidence"
evidence_dir.mkdir(parents=True)
evidence_path = evidence_dir / "calibration_evidence.geojson"
evidence_path.write_text(
json.dumps({"type": "FeatureCollection", "features": features}),
encoding="utf-8",
)
portfolio_path = portfolio_dir / "calibration_evidence_portfolio.json"
portfolio_path.write_text(
json.dumps(
{
"model_asset_id": f"model-{label}",
"samples": [
{
"sample_slug": "geel",
"evidence_geojson_path": str(evidence_path),
}
],
}
),
encoding="utf-8",
)
portfolio_paths.append((label, portfolio_path))
output_dir = tmp_path / "audit"
subprocess.run(
[
sys.executable,
str(script),
"--portfolio",
f"active={portfolio_paths[0][1]}",
"--portfolio",
f"candidate={portfolio_paths[1][1]}",
"--output-dir",
str(output_dir),
],
cwd=ROOT,
check=True,
capture_output=True,
text=True,
)
report = json.loads(
(output_dir / "detection_false_negative_audit.json").read_text(encoding="utf-8")
)
assert report["portfolio_count"] == 2
sample = report["samples"][0]
assert sample["sample_slug"] == "geel"
assert sample["persistent_false_negative_count"] == 1
assert sample["persistent_reference_ids"] == ["source:persistent-small"]
active = next(item for item in sample["portfolios"] if item["label"] == "active")
candidate = next(item for item in sample["portfolios"] if item["label"] == "candidate")
assert active["false_negative_count"] == 2
assert active["matched_reference_count"] == 1
assert active["false_negative_rate"] == 2 / 3
assert candidate["false_negative_count"] == 1
assert candidate["false_negative_rate"] == 1 / 3
assert active["false_negative_area_m2"]["median"] > 0
assert report["recommendations"]
assert (output_dir / "detection_false_negative_audit.md").is_file()
+35
View File
@@ -684,6 +684,41 @@ QA evidence GeoJSON endpoint, writes `calibration_evidence.geojson`,
matched references, false positives and false negatives. Set
`CALIBRATION_EVIDENCE_MODE=best` to export only the `best_by_score` run.
Build fixed-threshold portfolio inputs when two model runs must be compared at
the same confidence threshold across every AOI:
```bash
python scripts/build_fixed_threshold_evidence_portfolio_inputs.py \
--multi-sample-summary artifacts/detection-quality-matrix/multi-sample/<run>/multi_sample_quality_summary.json \
--threshold 0.35 \
--model-asset-id geointel-building-yolov8s-aoi1024bg512r3e50-pt \
--model-sha256 e0980572aac90e7efc514608eb16d7de5bfbf27a4bbec04e7bc1bc8c02f9601f \
--tile-size 512 \
--tile-overlap 64 \
--output-dir artifacts/detection-false-negative-review/active-inputs
```
The builder selects exactly one persisted QA run per AOI and refuses ambiguous
model/tile/threshold matches. Pass its emitted manifest to
`assemble_detection_calibration_evidence_portfolio.sh` with
`CALIBRATION_EVIDENCE_MODE=all`; each filtered summary contains one run.
Compare two or more downloaded evidence portfolios with geodetic WGS84 areas:
```bash
python scripts/audit_detection_false_negative_evidence.py \
--portfolio active=artifacts/detection-false-negative-review/active/calibration_evidence_portfolio.json \
--portfolio candidate=artifacts/detection-false-negative-review/candidate/calibration_evidence_portfolio.json \
--output-dir artifacts/detection-false-negative-review/audit
```
The audit reports false-negative rates and area buckets per AOI/model, plus
reference buildings missed by every compared portfolio. Stable
`source_feature_id` values are preferred; a normalized geometry fingerprint is
used only when source IDs are absent. Invalid or missing geometry fails the
audit instead of being silently skipped. The tools do not run inference,
create QA records, mutate model defaults or download data/models.
Docker images install only the GIS runtime by default. To build a local/Tower
image with PyTorch/Ultralytics available for the configured-YOLO preflight and
runtime path, set:
@@ -0,0 +1,359 @@
from __future__ import annotations
import argparse
import hashlib
import json
import math
import statistics
from datetime import datetime, timezone
from pathlib import Path
from typing import Any
AREA_BUCKETS = (
("tiny_lt_25_m2", 0.0, 25.0),
("small_25_100_m2", 25.0, 100.0),
("medium_100_500_m2", 100.0, 500.0),
("large_gte_500_m2", 500.0, math.inf),
)
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Audit persisted GeoIntel false-negative QA evidence across model portfolios."
)
parser.add_argument(
"--portfolio",
action="append",
required=True,
help="Label and calibration_evidence_portfolio.json path as label=/path/file.json.",
)
parser.add_argument("--output-dir", required=True, type=Path)
return parser.parse_args()
def load_json(path: Path) -> dict[str, Any]:
if not path.is_file():
raise SystemExit(f"JSON input is not readable: {path}")
payload = json.loads(path.read_text(encoding="utf-8-sig"))
if not isinstance(payload, dict):
raise SystemExit(f"JSON input must be an object: {path}")
return payload
def parse_portfolio_arg(raw: str) -> tuple[str, Path]:
label, separator, path_raw = raw.partition("=")
label = label.strip()
if not separator or not label or not path_raw.strip():
raise SystemExit("--portfolio must use label=/path/to/calibration_evidence_portfolio.json")
return label, Path(path_raw.strip()).expanduser().resolve()
def resolve_evidence_path(portfolio_path: Path, sample: dict[str, Any]) -> Path:
raw = str(sample.get("evidence_geojson_path") or "")
configured = Path(raw).expanduser()
candidates = [configured]
if raw and not configured.is_absolute():
candidates.append(portfolio_path.parent / configured)
sample_slug = str(sample.get("sample_slug") or "").strip().lower()
candidates.append(
portfolio_path.parent
/ "samples"
/ sample_slug
/ "evidence"
/ "calibration_evidence.geojson"
)
for candidate in candidates:
if candidate.is_file():
return candidate.resolve()
raise SystemExit(f"Evidence GeoJSON is not readable for {sample_slug}: {raw}")
def percentile(values: list[float], fraction: float) -> float | None:
if not values:
return None
ordered = sorted(values)
position = (len(ordered) - 1) * fraction
lower = math.floor(position)
upper = math.ceil(position)
if lower == upper:
return ordered[lower]
return ordered[lower] + (ordered[upper] - ordered[lower]) * (position - lower)
def area_stats(values: list[float]) -> dict[str, float | int | None]:
return {
"count": len(values),
"min": min(values) if values else None,
"p10": percentile(values, 0.10),
"median": statistics.median(values) if values else None,
"p90": percentile(values, 0.90),
"max": max(values) if values else None,
}
def area_bucket(area_m2: float) -> str:
for label, minimum, maximum in AREA_BUCKETS:
if minimum <= area_m2 < maximum:
return label
raise AssertionError(f"No area bucket for {area_m2}")
def stable_reference_id(feature: dict[str, Any], geometry: Any) -> str:
properties = feature.get("properties") or {}
source_feature_id = properties.get("source_feature_id")
if source_feature_id not in (None, ""):
return f"source:{source_feature_id}"
reference_feature_id = properties.get("reference_feature_id")
if reference_feature_id not in (None, ""):
return f"reference:{reference_feature_id}"
normalized_wkb = getattr(geometry.normalize(), "wkb", geometry.wkb)
return f"geometry:{hashlib.sha256(normalized_wkb).hexdigest()}"
def audit_feature_collection(payload: dict[str, Any], geod: Any, shape: Any) -> dict[str, Any]:
if payload.get("type") != "FeatureCollection":
raise SystemExit("Evidence GeoJSON must be a FeatureCollection")
features = payload.get("features") or []
false_negative_ids: set[str] = set()
false_negative_areas: list[float] = []
matched_reference_areas: list[float] = []
bucket_counts = {
label: {"false_negative": 0, "matched_reference": 0, "total_reference": 0, "false_negative_rate": None}
for label, _, _ in AREA_BUCKETS
}
for feature in features:
if not isinstance(feature, dict):
raise SystemExit("Evidence GeoJSON contains a non-object feature")
role = str((feature.get("properties") or {}).get("qa_evidence_role") or "")
if role not in {"false_negative", "match_reference"}:
continue
geometry_payload = feature.get("geometry")
if not isinstance(geometry_payload, dict):
raise SystemExit(f"Evidence feature {feature.get('id')} has no geometry")
geometry = shape(geometry_payload)
if geometry.is_empty or not geometry.is_valid:
raise SystemExit(f"Evidence feature {feature.get('id')} has invalid geometry")
if geometry.geom_type not in {"Polygon", "MultiPolygon"}:
raise SystemExit(
f"Evidence feature {feature.get('id')} must be Polygon or MultiPolygon"
)
area_m2 = abs(float(geod.geometry_area_perimeter(geometry)[0]))
bucket = area_bucket(area_m2)
bucket_role = "false_negative" if role == "false_negative" else "matched_reference"
bucket_counts[bucket][bucket_role] += 1
bucket_counts[bucket]["total_reference"] += 1
if role == "false_negative":
false_negative_ids.add(stable_reference_id(feature, geometry))
false_negative_areas.append(area_m2)
else:
matched_reference_areas.append(area_m2)
for values in bucket_counts.values():
total = values["total_reference"]
values["false_negative_rate"] = values["false_negative"] / total if total else None
total_reference = len(false_negative_areas) + len(matched_reference_areas)
return {
"false_negative_ids": false_negative_ids,
"false_negative_count": len(false_negative_areas),
"matched_reference_count": len(matched_reference_areas),
"total_reference_count": total_reference,
"false_negative_rate": len(false_negative_areas) / total_reference if total_reference else None,
"false_negative_area_m2": area_stats(false_negative_areas),
"matched_reference_area_m2": area_stats(matched_reference_areas),
"area_buckets": bucket_counts,
}
def build_recommendations(samples: list[dict[str, Any]]) -> list[str]:
recommendations: list[str] = []
persistent_total = sum(sample["persistent_false_negative_count"] for sample in samples)
if persistent_total:
recommendations.append(
f"Prioritize targeted positive sampling for {persistent_total} reference buildings missed by every compared portfolio."
)
ranked = sorted(
samples,
key=lambda sample: max(
(item.get("false_negative_rate") or 0.0 for item in sample["portfolios"]),
default=0.0,
),
reverse=True,
)
if ranked:
recommendations.append(
"Expand or oversample the weakest AOIs first: "
+ ", ".join(sample["sample_slug"] for sample in ranked[:3])
+ "."
)
small_bias_samples: list[str] = []
for sample in samples:
for portfolio in sample["portfolios"]:
buckets = portfolio["area_buckets"]
small_total = sum(
buckets[key]["total_reference"]
for key in ("tiny_lt_25_m2", "small_25_100_m2")
)
small_fn = sum(
buckets[key]["false_negative"]
for key in ("tiny_lt_25_m2", "small_25_100_m2")
)
larger_total = sum(
buckets[key]["total_reference"]
for key in ("medium_100_500_m2", "large_gte_500_m2")
)
larger_fn = sum(
buckets[key]["false_negative"]
for key in ("medium_100_500_m2", "large_gte_500_m2")
)
small_rate = small_fn / small_total if small_total else 0.0
larger_rate = larger_fn / larger_total if larger_total else 0.0
if small_total >= 5 and small_rate >= larger_rate + 0.15:
small_bias_samples.append(sample["sample_slug"])
break
if small_bias_samples:
recommendations.append(
"Review small-building label retention and pixel-size gates for: "
+ ", ".join(sorted(set(small_bias_samples)))
+ "."
)
if not recommendations:
recommendations.append(
"No dominant false-negative pattern was found; preserve current data gates and collect more independent AOIs."
)
return recommendations
def run_audit(portfolio_args: list[str], output_dir: Path) -> tuple[Path, Path]:
try:
from pyproj import Geod
from shapely.geometry import shape
except ImportError as exc:
raise SystemExit(
"False-negative GIS audit requires the existing GeoIntel GIS extras (pyproj and shapely)."
) from exc
parsed_portfolios = [parse_portfolio_arg(raw) for raw in portfolio_args]
labels = [label for label, _ in parsed_portfolios]
if len(labels) != len(set(labels)):
raise SystemExit("Every --portfolio label must be unique")
geod = Geod(ellps="WGS84")
portfolio_samples: dict[str, dict[str, dict[str, Any]]] = {}
portfolio_meta: list[dict[str, Any]] = []
for label, portfolio_path in parsed_portfolios:
portfolio = load_json(portfolio_path)
samples = portfolio.get("samples") or []
if not isinstance(samples, list) or not samples:
raise SystemExit(f"Evidence portfolio has no samples: {portfolio_path}")
sample_results: dict[str, dict[str, Any]] = {}
for sample in samples:
sample_slug = str(sample.get("sample_slug") or "").strip().lower()
if not sample_slug:
raise SystemExit(f"Evidence portfolio has a sample without sample_slug: {portfolio_path}")
evidence_path = resolve_evidence_path(portfolio_path, sample)
result = audit_feature_collection(load_json(evidence_path), geod, shape)
result.update(
{
"label": label,
"model_asset_id": portfolio.get("model_asset_id"),
"evidence_geojson_path": str(evidence_path),
}
)
sample_results[sample_slug] = result
portfolio_samples[label] = sample_results
portfolio_meta.append(
{
"label": label,
"path": str(portfolio_path),
"model_asset_id": portfolio.get("model_asset_id"),
"sample_count": len(sample_results),
}
)
expected_slugs = set(next(iter(portfolio_samples.values())))
for label, samples in portfolio_samples.items():
if set(samples) != expected_slugs:
raise SystemExit(
f"Portfolio {label} has different AOIs; fixed-threshold comparisons require identical samples"
)
sample_reports: list[dict[str, Any]] = []
for sample_slug in sorted(expected_slugs):
portfolio_rows = []
false_negative_sets = []
for label, _ in parsed_portfolios:
raw = portfolio_samples[label][sample_slug]
false_negative_sets.append(raw["false_negative_ids"])
portfolio_rows.append(
{key: value for key, value in raw.items() if key != "false_negative_ids"}
)
persistent_ids = sorted(set.intersection(*false_negative_sets))
sample_reports.append(
{
"sample_slug": sample_slug,
"persistent_false_negative_count": len(persistent_ids),
"persistent_reference_ids": persistent_ids,
"portfolios": portfolio_rows,
}
)
report = {
"generated_at": datetime.now(timezone.utc).isoformat(),
"schema_version": 1,
"input_crs": "EPSG:4326",
"area_method": "WGS84 geodesic area via pyproj.Geod",
"portfolio_count": len(portfolio_meta),
"portfolios": portfolio_meta,
"sample_count": len(sample_reports),
"samples": sample_reports,
"recommendations": build_recommendations(sample_reports),
}
output_dir = output_dir.expanduser().resolve()
output_dir.mkdir(parents=True, exist_ok=True)
json_path = output_dir / "detection_false_negative_audit.json"
json_path.write_text(json.dumps(report, indent=2, sort_keys=True), encoding="utf-8")
lines = [
"# Detection false-negative evidence audit",
"",
f"- Generated: {report['generated_at']}",
f"- Portfolios: {report['portfolio_count']}",
f"- AOIs: {report['sample_count']}",
f"- Area method: {report['area_method']}",
"",
"## AOI comparison",
"",
"| AOI | Persistent misses | "
+ " | ".join(f"{label} FN rate" for label, _ in parsed_portfolios)
+ " |",
"|---|---:|" + "---:|" * len(parsed_portfolios),
]
for sample in sample_reports:
rates = [
f"{(row['false_negative_rate'] or 0.0):.3f}"
for row in sample["portfolios"]
]
lines.append(
f"| {sample['sample_slug']} | {sample['persistent_false_negative_count']} | "
+ " | ".join(rates)
+ " |"
)
lines.extend(["", "## Recommended data actions", ""])
lines.extend(f"- {item}" for item in report["recommendations"])
markdown_path = output_dir / "detection_false_negative_audit.md"
markdown_path.write_text("\n".join(lines) + "\n", encoding="utf-8")
return json_path, markdown_path
def main() -> None:
args = parse_args()
json_path, markdown_path = run_audit(args.portfolio, args.output_dir)
print(f"False-negative audit JSON: {json_path}")
print(f"False-negative audit Markdown: {markdown_path}")
if __name__ == "__main__":
main()
@@ -0,0 +1,168 @@
from __future__ import annotations
import argparse
import json
from pathlib import Path
from typing import Any
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Build comparable fixed-threshold inputs for a GeoIntel calibration evidence portfolio."
)
parser.add_argument("--multi-sample-summary", required=True, type=Path)
parser.add_argument("--threshold", required=True, type=float)
parser.add_argument("--model-asset-id")
parser.add_argument("--model-sha256")
parser.add_argument("--tile-size", type=int)
parser.add_argument("--tile-overlap", type=int)
parser.add_argument("--output-dir", required=True, type=Path)
return parser.parse_args()
def load_json(path: Path) -> dict[str, Any]:
if not path.is_file():
raise SystemExit(f"JSON input is not readable: {path}")
payload = json.loads(path.read_text(encoding="utf-8-sig"))
if not isinstance(payload, dict):
raise SystemExit(f"JSON input must be an object: {path}")
return payload
def resolve_summary_path(raw: str, multi_summary_path: Path) -> Path:
path = Path(raw).expanduser()
candidates = [path]
if not path.is_absolute():
candidates.extend((multi_summary_path.parent / path, Path.cwd() / path))
for candidate in candidates:
if candidate.is_file():
return candidate.resolve()
raise SystemExit(f"Sample summary is not readable: {raw}")
def matches_run(
item: dict[str, Any],
*,
threshold: float,
model_asset_id: str | None,
tile_size: int | None,
tile_overlap: int | None,
) -> bool:
try:
item_threshold = float(item.get("threshold"))
except (TypeError, ValueError):
return False
if abs(item_threshold - threshold) > 1e-9:
return False
if model_asset_id and item.get("model_asset_id") != model_asset_id:
return False
if tile_size is not None and item.get("tile_size") != tile_size:
return False
if tile_overlap is not None and item.get("tile_overlap") != tile_overlap:
return False
return True
def build_inputs(args: argparse.Namespace) -> Path:
multi_summary_path = args.multi_sample_summary.expanduser().resolve()
multi_summary = load_json(multi_summary_path)
sample_summaries = multi_summary.get("sample_summaries") or []
if not isinstance(sample_summaries, list) or not sample_summaries:
raise SystemExit("Multi-sample summary has no sample_summaries")
output_dir = args.output_dir.expanduser().resolve()
samples_dir = output_dir / "samples"
samples_dir.mkdir(parents=True, exist_ok=True)
manifest_samples: list[dict[str, Any]] = []
selected_model_ids: set[str] = set()
for sample in sample_summaries:
if not isinstance(sample, dict):
raise SystemExit("Every sample summary entry must be an object")
sample_slug = str(sample.get("sample_slug") or "").strip().lower()
if not sample_slug:
raise SystemExit("Sample summary entry is missing sample_slug")
source_summary_path = resolve_summary_path(
str(sample.get("summary_path") or ""), multi_summary_path
)
source_summary = load_json(source_summary_path)
items = source_summary.get("items") or []
matches = [
item
for item in items
if isinstance(item, dict)
and matches_run(
item,
threshold=args.threshold,
model_asset_id=args.model_asset_id,
tile_size=args.tile_size,
tile_overlap=args.tile_overlap,
)
]
if len(matches) != 1:
raise SystemExit(
f"Expected exactly one fixed-threshold run for {sample_slug}; found {len(matches)}"
)
selected = dict(matches[0])
selected_model_id = str(selected.get("model_asset_id") or "").strip()
if not selected_model_id:
raise SystemExit(f"Selected run for {sample_slug} has no model_asset_id")
selected_model_ids.add(selected_model_id)
filtered_summary = dict(source_summary)
filtered_summary.update(
{
"items": [selected],
"best_by_score": selected,
"best_by_recall": selected,
"best_by_precision": selected,
"fixed_threshold": args.threshold,
"source_summary_path": str(source_summary_path),
}
)
sample_dir = samples_dir / sample_slug
sample_dir.mkdir(parents=True, exist_ok=True)
filtered_path = sample_dir / "quality_matrix_summary.json"
filtered_path.write_text(
json.dumps(filtered_summary, indent=2, sort_keys=True), encoding="utf-8"
)
manifest_samples.append(
{
"sample_slug": sample_slug,
"aoi_label": sample_slug.replace("_", " ").title(),
"summary_path": str(filtered_path),
"operator_notes": (
f"Fixed threshold {args.threshold:g}; source {source_summary_path}."
),
}
)
if len(selected_model_ids) != 1:
raise SystemExit(
f"Fixed-threshold runs contain multiple model assets: {sorted(selected_model_ids)}"
)
selected_model_id = next(iter(selected_model_ids))
manifest = {
"portfolio_name": (
f"GeoIntel fixed-threshold false-negative review: {selected_model_id} @ {args.threshold:g}"
),
"model_asset_id": selected_model_id,
"model_sha256": args.model_sha256,
"fixed_threshold": args.threshold,
"notes": "Comparable per-AOI persisted QA evidence; no inference is run by this input builder.",
"samples": manifest_samples,
}
manifest_path = output_dir / "calibration-evidence-portfolio-manifest.json"
manifest_path.write_text(
json.dumps(manifest, indent=2, sort_keys=True), encoding="utf-8"
)
return manifest_path
def main() -> None:
manifest_path = build_inputs(parse_args())
print(f"Fixed-threshold portfolio manifest: {manifest_path}")
if __name__ == "__main__":
main()
+2
View File
@@ -47,6 +47,8 @@ ${PYTHON_BIN} -m py_compile scripts/export_operator_yolo_tile_dataset.py
${PYTHON_BIN} -m py_compile scripts/audit_operator_yolo_dataset_quality.py
${PYTHON_BIN} -m py_compile scripts/build_detection_model_promotion_report.py
${PYTHON_BIN} -m py_compile scripts/build_background_corpus_split_report.py
${PYTHON_BIN} -m py_compile scripts/build_fixed_threshold_evidence_portfolio_inputs.py
${PYTHON_BIN} -m py_compile scripts/audit_detection_false_negative_evidence.py
${PYTHON_BIN} -m py_compile scripts/activate_promoted_yolo_candidate.py
${PYTHON_BIN} -m py_compile scripts/cleanup_demo_artifacts.py
${PYTHON_BIN} -m py_compile backend/scripts/cleanup_demo_artifacts.py