diff --git a/CHANGELOG.md b/CHANGELOG.md index 543cfd9a..bc8912f8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,14 @@ # Changelog +## Sprint 175 Detection result scale and false-positive evidence review (2026-07-13) + +- Bounded Detection Lab table rendering with client-side 25/50/100-row pagination while preserving the complete persisted detection set for MapLibre and QA/QC. +- Added a strict read-only false-positive evidence audit with role-count drift checks, polygon validation, WGS84 geodesic areas, size buckets, AOI/class/tile summaries and combined review GeoJSON. +- Audited the active seven-AOI portfolio: 5,568 false positives among 13,613 candidates, median geometry area 184.5 m2, and 25.8% below 100 m2; Turnhout, Herentals and Geel carry the largest review volumes. +- Recorded that existing QA evidence has no per-detection confidence values; the audit reports zero confidence coverage and does not infer scores from the run threshold. +- Kept API contracts, database migrations, model activation, provider behavior and inference behavior unchanged. + ## Sprint 174 Focused small-building model promotion (2026-07-13) - Expanded the real operator corpus with four focused training AOIs and two independent validation AOIs, while keeping Turnhout, Retie and Westerlo outside the tile-training corpus as operation-level holdouts. diff --git a/backend/tests/test_sprint175_detection_review_hardening.py b/backend/tests/test_sprint175_detection_review_hardening.py new file mode 100644 index 00000000..933e3536 --- /dev/null +++ b/backend/tests/test_sprint175_detection_review_hardening.py @@ -0,0 +1,222 @@ +from __future__ import annotations + +import json +import subprocess +import sys +from pathlib import Path + + +ROOT = Path(__file__).resolve().parents[2] + + +def _polygon(x: float, y: float, size: float) -> dict: + return { + "type": "Polygon", + "coordinates": [ + [ + [x, y], + [x + size, y], + [x + size, y + size], + [x, y + size], + [x, y], + ] + ], + } + + +def _evidence_feature( + role: str, + feature_id: str, + geometry: dict, + *, + tile_index: int = 1, + confidence: float | None = None, +) -> dict: + properties = { + "qa_evidence_role": role, + "candidate_feature_id": feature_id, + "feature_class": "building", + "tile_index": tile_index, + "analysis_run_id": "run-1", + "quality_check_id": "quality-1", + "calibration_threshold": 0.15, + "calibration_model_asset_id": "model-a", + } + if confidence is not None: + properties["candidate_confidence"] = confidence + return { + "type": "Feature", + "id": f"{role}:{feature_id}", + "properties": properties, + "geometry": geometry, + } + + +def _write_portfolio( + tmp_path: Path, + features: list[dict], + *, + declared_false_positives: int, +) -> Path: + evidence_dir = tmp_path / "samples" / "geel" / "evidence" + evidence_dir.mkdir(parents=True) + evidence_path = evidence_dir / "calibration_evidence.geojson" + evidence_path.write_text( + json.dumps({"type": "FeatureCollection", "features": features}), + encoding="utf-8", + ) + portfolio_path = tmp_path / "calibration_evidence_portfolio.json" + portfolio_path.write_text( + json.dumps( + { + "model_asset_id": "model-a", + "model_sha256": "abc123", + "samples": [ + { + "sample_slug": "geel", + "role_counts": { + "false_positive": declared_false_positives, + "match_candidate": 1, + }, + "evidence_geojson_path": str(evidence_path), + } + ], + } + ), + encoding="utf-8", + ) + return portfolio_path + + +def test_detection_results_table_uses_bounded_local_pagination() -> None: + lab = (ROOT / "frontend" / "src" / "components" / "detection" / "DetectionLab.tsx").read_text( + encoding="utf-8" + ) + styles = (ROOT / "frontend" / "src" / "styles" / "app.css").read_text( + encoding="utf-8" + ) + + assert "DETECTION_PAGE_SIZE_OPTIONS" in lab + assert "visibleDetectionItems" in lab + assert "detectionItems.slice" in lab + assert "Detection result pagination" in lab + assert "Previous detection results page" in lab + assert "Next detection results page" in lab + assert "pagination-toolbar" in styles + assert "detectionItems.map((detection)" not in lab + + +def test_false_positive_audit_builds_reviewable_persisted_evidence(tmp_path: Path) -> None: + script = ROOT / "scripts" / "audit_detection_false_positive_evidence.py" + readiness = (ROOT / "scripts" / "run_readiness_check.sh").read_text(encoding="utf-8") + dockerfile = (ROOT / "deploy" / "unraid" / "Dockerfile.all-in-one").read_text( + encoding="utf-8" + ) + assert script.exists() + assert "py_compile scripts/audit_detection_false_positive_evidence.py" in readiness + assert "COPY scripts/audit_detection_false_positive_evidence.py" in dockerfile + + features = [ + _evidence_feature( + "false_positive", + "fp-small", + _polygon(5.0, 51.2, 0.0001), + tile_index=4, + confidence=0.27, + ), + _evidence_feature( + "false_positive", + "fp-large", + _polygon(5.001, 51.2, 0.0003), + tile_index=4, + confidence=0.81, + ), + _evidence_feature( + "match_candidate", + "matched", + _polygon(5.002, 51.2, 0.0002), + tile_index=7, + ), + ] + portfolio_path = _write_portfolio( + tmp_path / "portfolio", + features, + declared_false_positives=2, + ) + output_dir = tmp_path / "audit" + result = subprocess.run( + [ + sys.executable, + str(script), + "--portfolio", + str(portfolio_path), + "--output-dir", + str(output_dir), + ], + cwd=ROOT, + check=True, + capture_output=True, + text=True, + ) + + report = json.loads( + (output_dir / "detection_false_positive_audit.json").read_text(encoding="utf-8") + ) + assert report["model_asset_id"] == "model-a" + assert report["model_sha256"] == "abc123" + assert report["false_positive_count"] == 2 + assert report["candidate_count"] == 3 + assert report["false_positive_rate"] == 2 / 3 + assert report["confidence_coverage_count"] == 2 + assert report["confidence"]["median"] == 0.54 + assert report["source_tile_counts"] == {"geel:4": 2} + assert sum(bucket["count"] for bucket in report["area_buckets"].values()) == 2 + assert report["samples"][0]["false_positive_area_m2"]["median"] > 0 + assert report["recommendations"] + + geojson = json.loads( + (output_dir / "false_positives.geojson").read_text(encoding="utf-8") + ) + assert geojson["type"] == "FeatureCollection" + assert len(geojson["features"]) == 2 + assert all( + feature["properties"]["qa_evidence_role"] == "false_positive" + for feature in geojson["features"] + ) + assert all(feature["properties"]["sample_slug"] == "geel" for feature in geojson["features"]) + assert all(feature["properties"]["area_m2"] > 0 for feature in geojson["features"]) + assert "False-positive audit JSON" in result.stdout + assert (output_dir / "detection_false_positive_audit.md").is_file() + + +def test_false_positive_audit_rejects_declared_role_count_drift(tmp_path: Path) -> None: + script = ROOT / "scripts" / "audit_detection_false_positive_evidence.py" + portfolio_path = _write_portfolio( + tmp_path / "portfolio", + [ + _evidence_feature( + "false_positive", + "fp-one", + _polygon(5.0, 51.2, 0.0001), + ) + ], + declared_false_positives=2, + ) + + result = subprocess.run( + [ + sys.executable, + str(script), + "--portfolio", + str(portfolio_path), + "--output-dir", + str(tmp_path / "audit"), + ], + cwd=ROOT, + check=False, + capture_output=True, + text=True, + ) + + assert result.returncode != 0 + assert "declares 2 false positives but evidence contains 1" in result.stderr diff --git a/deploy/unraid/Dockerfile.all-in-one b/deploy/unraid/Dockerfile.all-in-one index e0e58005..478ac386 100644 --- a/deploy/unraid/Dockerfile.all-in-one +++ b/deploy/unraid/Dockerfile.all-in-one @@ -83,6 +83,7 @@ COPY scripts/export_detection_calibration_evidence.sh /app/scripts/export_detect COPY scripts/assemble_detection_calibration_evidence_portfolio.sh /app/scripts/assemble_detection_calibration_evidence_portfolio.sh COPY scripts/build_fixed_threshold_evidence_portfolio_inputs.py /app/scripts/build_fixed_threshold_evidence_portfolio_inputs.py COPY scripts/audit_detection_false_negative_evidence.py /app/scripts/audit_detection_false_negative_evidence.py +COPY scripts/audit_detection_false_positive_evidence.py /app/scripts/audit_detection_false_positive_evidence.py COPY scripts/run_operator_hard_negative_detection_matrix.sh /app/scripts/run_operator_hard_negative_detection_matrix.sh COPY scripts/run_background_corpus_split_matrix.sh /app/scripts/run_background_corpus_split_matrix.sh COPY scripts/build_background_corpus_split_report.py /app/scripts/build_background_corpus_split_report.py diff --git a/docs/AI_PIPELINES.md b/docs/AI_PIPELINES.md index c860b026..880cae38 100644 --- a/docs/AI_PIPELINES.md +++ b/docs/AI_PIPELINES.md @@ -352,6 +352,16 @@ remains available as a higher-precision legacy `0.15` choice. The older default-promotion blocker. Every production-like run still requires persisted QA/QC against suitable reference data. +The persisted seven-AOI evidence for this profile contains 5,568 false +positives among 13,613 candidate detections. The read-only audit command in +`scripts/README.md` reports the largest review volumes in Turnhout, Herentals +and Geel, a median false-positive geometry area of about 184.5 m2, and 25.8% +tiny/small geometry below 100 m2. Current evidence does not include +per-detection confidence, so model-review reports must retain confidence +coverage as zero rather than treating threshold `0.15` as an observed score. +Combined false-positive GeoJSON is evidence for operator review only; a feature +must be visually confirmed before it is used as a hard-negative label. + To update a Tower/Unraid `.env` from a promoted report, use the guarded activation helper. It validates the exact report candidate key, verifies that the candidate has `promotion_status=promote_candidate`, resolves the local model diff --git a/docs/CODEX_EXECUTION_LOG.md b/docs/CODEX_EXECUTION_LOG.md index 38128ae4..851010e5 100644 --- a/docs/CODEX_EXECUTION_LOG.md +++ b/docs/CODEX_EXECUTION_LOG.md @@ -7138,3 +7138,41 @@ Open: ## Next recommended pass - After redeploy, verify the active model SHA, local model-load preflight, live PostGIS migration smoke and browser profile selection. Then review false-positive evidence and the remaining 5,838 persistent misses before any further training. + +# Sprint 175 - Detection result scale and false-positive evidence review + +## UI hardening + +- Confirmed that Detection Lab rendered every persisted detection row at once; Westerlo alone produced 1,172 body rows in the browser. +- Added local 25/50/100-row pagination with a default of 50 rows, bounded page controls and automatic page-one reset after run/filter/result changes. +- Kept the full persisted collection unchanged for the existing MapLibre GeoJSON overlay and detection QA/QC. No endpoint, response envelope or persistence contract changed. + +## Persisted false-positive evidence + +- Added `scripts/audit_detection_false_positive_evidence.py` as a read-only evidence consumer. +- The audit validates FeatureCollection/polygon geometry, compares declared portfolio role counts with actual evidence, computes WGS84 geodesic areas, preserves original feature provenance and emits combined `false_positives.geojson`. +- Source tile summaries are qualified by AOI because `tile_index` is local to each raster manifest. +- The active fixed-threshold seven-AOI portfolio produced: + - 5,568 false positives among 13,613 candidate detections (`0.4090` false-positive share); + - median false-positive geometry area `184.5 m2`, p90 `607.7 m2`; + - 52 below 25 m2, 1,382 between 25-100 m2, 3,412 between 100-500 m2 and 722 at or above 500 m2; + - largest AOI review volumes: Turnhout `1,102`, Herentals `917`, Geel `913`; + - largest AOI-qualified tile hotspot: `turnhout:0` with 236 false positives. +- Existing persisted QA evidence carries run threshold and tile index but no per-detection confidence. The audit reports confidence coverage `0/5,568` and does not invent confidence statistics. +- Added focused regression coverage, readiness compilation and all-in-one image inclusion. No training, inference, provider fetch, model download or activation occurred. + +## Local validation + +- `python -m compileall backend/app`: passed. +- `python -m pytest`: 475 passed. +- `python -m ruff check` for the new audit/test modules: passed. +- `npm run typecheck`: passed. +- `npm run build`: passed; app bundle `216.83 kB`, MapLibre bundle `801.82 kB` before gzip. +- `bash scripts/run_readiness_check.sh`: passed with 475 tests and all release-critical syntax gates. +- `python -m alembic heads`: one head, `202606120900`. +- `python -m alembic upgrade head --sql`: complete migration chain rendered successfully. +- Local `docker compose config` could not run because Docker CLI is not installed on the Windows host; live image/PostGIS validation is delegated to the Docker-enabled Tower deployment. + +## Next recommended pass + +- Redeploy and verify bounded table rendering against the live 1,172-detection Westerlo run. Then visually classify a stratified sample from Turnhout, Herentals and Geel before deciding whether any confirmed false positives belong in a new hard-negative corpus. diff --git a/docs/TODO.md b/docs/TODO.md index 181c7d60..cda9be30 100644 --- a/docs/TODO.md +++ b/docs/TODO.md @@ -138,7 +138,8 @@ This file now starts with the current implementation status. Older preparation/b - [x] Review per-AOI false-negative evidence, expand positive sample/label coverage and verify the resulting candidate improves false-negative rate in every validated AOI. - [x] Complete rebuild/restart and browser/runtime smoke for the guarded promoted V1 building detector activation. - [x] Expand focused small-building training evidence after reviewing persistent false negatives, train one inactive candidate and pass it through positive, pure-empty and fixed-reference promotion evidence before guarded activation. -- [ ] Review the remaining 5,838 persistent false negatives and the increased false-positive load before any further model training; do not start another blind run. +- [x] Audit the promoted model's increased false-positive load from persisted seven-AOI QA evidence, including geodetic area buckets, AOI-qualified tile hotspots and combined review GeoJSON. +- [ ] Visually classify representative false-positive evidence from Turnhout, Herentals and Geel and review the remaining 5,838 persistent false negatives before any further model training; do not start another blind run. ## Sprint 8 status diff --git a/frontend/README.md b/frontend/README.md index c7ff0dae..5ad1a6b5 100644 --- a/frontend/README.md +++ b/frontend/README.md @@ -155,6 +155,13 @@ AI Lab run controls explicitly explain when no raster dataset is available, inst - `App.tsx` still owns shared state orchestration and API calls; extracted components receive the same state and callbacks as props. - Existing MapLibre overlay behavior, dataset/reference flows, Detection Lab flows and Segmentation Lab flows are unchanged. +## Detection result scale and review + +- Detection Lab keeps the complete persisted detection collection available to the existing MapLibre overlay and QA/QC flows. +- The results table renders 50 rows by default and provides 25/50/100 row sizes plus previous/next controls. This bounds DOM work for dense AOIs without discarding or resampling detections. +- Selecting another run, changing a class/confidence filter or loading a new result collection resets the visible table to page one. +- Pagination is intentionally client-side over the canonical persisted response; detection API contracts and GeoJSON output are unchanged. + ## Sprint 15 additions - Added a Projects panel action to load the explicit offline demo workflow. diff --git a/frontend/src/components/detection/DetectionLab.tsx b/frontend/src/components/detection/DetectionLab.tsx index 7c47704a..9c31b581 100644 --- a/frontend/src/components/detection/DetectionLab.tsx +++ b/frontend/src/components/detection/DetectionLab.tsx @@ -1,3 +1,4 @@ +import { useEffect, useState } from 'react' import type { DatasetCreateResponse, DetectionModelCapability, @@ -12,6 +13,9 @@ import type { import type { DetectionCalibrationRunRow } from '../../hooks/useDetectionWorkflow' import { DETECTION_OPERATOR_PROFILES, type DetectionOperatorProfile } from './detectionProfiles' +const DETECTION_PAGE_SIZE_OPTIONS = [25, 50, 100] as const +const DEFAULT_DETECTION_PAGE_SIZE = 50 + interface CalibrationRow { analysisRunId: string qualityCheckId: string @@ -152,6 +156,18 @@ export function DetectionLab({ const bestF1Candidate = bestCalibrationRow(calibrationRows, 'f1') const bestPrecisionCandidate = bestCalibrationRow(calibrationRows, 'precision') const lowestFalsePositivePressureCandidate = bestLowestCalibrationRow(calibrationRows, 'falsePositives') + const [detectionResultPage, setDetectionResultPage] = useState(1) + const [detectionPageSize, setDetectionPageSize] = useState(DEFAULT_DETECTION_PAGE_SIZE) + const detectionPageCount = Math.max(1, Math.ceil(detectionItems.length / detectionPageSize)) + const currentDetectionPage = Math.min(detectionResultPage, detectionPageCount) + const detectionPageStart = (currentDetectionPage - 1) * detectionPageSize + const detectionPageEnd = Math.min(detectionPageStart + detectionPageSize, detectionItems.length) + const visibleDetectionItems = detectionItems.slice(detectionPageStart, detectionPageEnd) + + useEffect(() => { + setDetectionResultPage(1) + }, [selectedDetectionRunId, detectionClassFilter, detectionMinConfidenceFilter, detectionItems]) + const detectionHasTileManifest = !detectionRequiresTileManifest || detectionTileManifestPath.trim().length > 0 const detectionRunReady = @@ -744,28 +760,73 @@ export function DetectionLab({ {detectionItems.length > 0 ? ( -
- - - - - - - - - - - {detectionItems.map((detection) => ( - - - - - + <> +
+

+ {detectionPageStart + 1}-{detectionPageEnd} + of {detectionItems.length} +

+ +
+ + Page {currentDetectionPage} of {detectionPageCount} + +
+
+
+
ClassConfidenceModelSource tile
{detection.class_name}{detection.confidence.toFixed(2)}{detection.model_name}{detection.source_tile_path || 'n/a'}
+ + + + + + - ))} - -
ClassConfidenceModelSource tile
-
+ + + {visibleDetectionItems.map((detection) => ( + + {detection.class_name} + {detection.confidence.toFixed(2)} + {detection.model_name} + {detection.source_tile_path || 'n/a'} + + ))} + + + + ) : null} diff --git a/frontend/src/styles/app.css b/frontend/src/styles/app.css index 490ed509..47f68440 100644 --- a/frontend/src/styles/app.css +++ b/frontend/src/styles/app.css @@ -998,6 +998,51 @@ section th { overflow-wrap: anywhere; } +.pagination-toolbar { + display: flex; + flex-wrap: wrap; + gap: 0.65rem 1rem; + align-items: center; + justify-content: space-between; + margin-top: 0.85rem; + border: 1px solid var(--line); + border-radius: 7px; + padding: 0.55rem 0.65rem; + background: #f8fbf9; +} + +.pagination-summary, +.pagination-actions, +.pagination-page-size { + display: flex; + gap: 0.45rem; + align-items: center; + margin: 0; +} + +.pagination-summary span, +.pagination-actions span, +.pagination-page-size { + color: var(--muted); + font-size: 0.82rem; +} + +.pagination-page-size select { + width: auto; + min-width: 4.5rem; + margin: 0; + padding: 0.35rem 1.65rem 0.35rem 0.45rem; +} + +.pagination-button { + width: 2rem; + min-width: 2rem; + height: 2rem; + padding: 0; + font-size: 1rem; + line-height: 1; +} + .workbench-inspector .job-result, .workbench-inspector pre { max-height: 16rem; diff --git a/scripts/README.md b/scripts/README.md index 86588764..af2c49d8 100644 --- a/scripts/README.md +++ b/scripts/README.md @@ -793,6 +793,24 @@ used only when source IDs are absent. Invalid or missing geometry fails the audit instead of being silently skipped. The tools do not run inference, create QA records, mutate model defaults or download data/models. +Audit the false-positive review load of one persisted evidence portfolio before +turning detections into hard-negative training input: + +```bash +python scripts/audit_detection_false_positive_evidence.py \ + --portfolio artifacts/model-review/small-building-candidate/evidence-portfolio/calibration_evidence_portfolio.json \ + --output-dir artifacts/model-review/small-building-candidate/false-positive-audit +``` + +The command validates the portfolio role counts against each persisted evidence +GeoJSON, rejects invalid/non-polygon geometry, computes WGS84 geodesic area and +size buckets, and reports false-positive pressure per AOI, class and +AOI-qualified source tile. It writes `detection_false_positive_audit.json`, a +Markdown handoff and combined `false_positives.geojson` for map review. Original +evidence properties and geometry are preserved. Confidence statistics are only +computed when confidence is actually present in persisted evidence; missing +coverage is reported explicitly and never inferred from the run threshold. + Docker images install only the GIS runtime by default. To build a local/Tower image with PyTorch/Ultralytics available for the configured-YOLO preflight and runtime path, set: diff --git a/scripts/audit_detection_false_positive_evidence.py b/scripts/audit_detection_false_positive_evidence.py new file mode 100644 index 00000000..a6855169 --- /dev/null +++ b/scripts/audit_detection_false_positive_evidence.py @@ -0,0 +1,361 @@ +from __future__ import annotations + +import argparse +import json +import math +from collections import Counter +from datetime import datetime, timezone +from pathlib import Path +from typing import Any + +from audit_detection_false_negative_evidence import ( + AREA_BUCKETS, + area_bucket, + area_stats, + load_json, + resolve_evidence_path, +) + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser( + description="Audit persisted GeoIntel false-positive QA evidence for one model portfolio." + ) + parser.add_argument( + "--portfolio", + required=True, + type=Path, + help="Path to calibration_evidence_portfolio.json.", + ) + parser.add_argument("--output-dir", required=True, type=Path) + return parser.parse_args() + + +def _confidence(properties: dict[str, Any], feature_id: Any) -> float | None: + for key in ("candidate_confidence", "confidence"): + value = properties.get(key) + if value is None: + continue + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise SystemExit(f"Evidence feature {feature_id} has non-numeric {key}") + numeric = float(value) + if not math.isfinite(numeric) or not 0.0 <= numeric <= 1.0: + raise SystemExit(f"Evidence feature {feature_id} has invalid {key}: {value}") + return numeric + return None + + +def _source_tile(properties: dict[str, Any]) -> str: + source_tile_path = properties.get("source_tile_path") + if source_tile_path not in (None, ""): + return str(source_tile_path) + tile_index = properties.get("tile_index") + if tile_index not in (None, ""): + return str(tile_index) + return "unknown" + + +def _feature_class(properties: dict[str, Any]) -> str: + value = properties.get("feature_class") or properties.get("class_name") + return str(value) if value not in (None, "") else "unknown" + + +def _bucket_summary(areas: list[float]) -> dict[str, dict[str, float | int]]: + counts = Counter(area_bucket(area) for area in areas) + total = len(areas) + return { + label: { + "count": counts[label], + "share": counts[label] / total if total else 0.0, + } + for label, _, _ in AREA_BUCKETS + } + + +def audit_sample( + payload: dict[str, Any], + *, + sample_slug: str, + geod: Any, + shape: Any, +) -> tuple[dict[str, Any], list[dict[str, Any]], list[float], list[float]]: + if payload.get("type") != "FeatureCollection": + raise SystemExit(f"Evidence for {sample_slug} must be a FeatureCollection") + features = payload.get("features") + if not isinstance(features, list): + raise SystemExit(f"Evidence for {sample_slug} must contain a features list") + + false_positive_features: list[dict[str, Any]] = [] + areas: list[float] = [] + confidences: list[float] = [] + role_counts: Counter[str] = Counter() + source_tile_counts: Counter[str] = Counter() + class_counts: Counter[str] = Counter() + + for feature in features: + if not isinstance(feature, dict): + raise SystemExit(f"Evidence for {sample_slug} contains a non-object feature") + properties = feature.get("properties") or {} + if not isinstance(properties, dict): + raise SystemExit(f"Evidence feature {feature.get('id')} has invalid properties") + role = str(properties.get("qa_evidence_role") or "") + role_counts[role] += 1 + if role != "false_positive": + continue + + geometry_payload = feature.get("geometry") + if not isinstance(geometry_payload, dict): + raise SystemExit(f"Evidence feature {feature.get('id')} has no geometry") + geometry = shape(geometry_payload) + if geometry.is_empty or not geometry.is_valid: + raise SystemExit(f"Evidence feature {feature.get('id')} has invalid geometry") + if geometry.geom_type not in {"Polygon", "MultiPolygon"}: + raise SystemExit( + f"Evidence feature {feature.get('id')} must be Polygon or MultiPolygon" + ) + + area_m2 = abs(float(geod.geometry_area_perimeter(geometry)[0])) + confidence = _confidence(properties, feature.get("id")) + source_tile = _source_tile(properties) + feature_class = _feature_class(properties) + augmented_properties = dict(properties) + augmented_properties.update( + { + "sample_slug": sample_slug, + "area_m2": area_m2, + "area_bucket": area_bucket(area_m2), + "review_source_tile": source_tile, + "review_confidence_available": confidence is not None, + } + ) + false_positive_features.append( + { + "type": "Feature", + "id": f"{sample_slug}:{feature.get('id')}", + "properties": augmented_properties, + "geometry": geometry_payload, + } + ) + areas.append(area_m2) + if confidence is not None: + confidences.append(confidence) + source_tile_counts[source_tile] += 1 + class_counts[feature_class] += 1 + + false_positive_count = role_counts["false_positive"] + matched_candidate_count = role_counts["match_candidate"] + candidate_count = false_positive_count + matched_candidate_count + report = { + "sample_slug": sample_slug, + "false_positive_count": false_positive_count, + "matched_candidate_count": matched_candidate_count, + "candidate_count": candidate_count, + "false_positive_rate": false_positive_count / candidate_count if candidate_count else None, + "false_positive_area_m2": area_stats(areas), + "area_buckets": _bucket_summary(areas), + "confidence_coverage_count": len(confidences), + "confidence": area_stats(confidences), + "source_tile_counts": dict(sorted(source_tile_counts.items())), + "class_counts": dict(sorted(class_counts.items())), + } + return report, false_positive_features, areas, confidences + + +def _merge_counts(rows: list[dict[str, Any]], key: str) -> dict[str, int]: + counts: Counter[str] = Counter() + for row in rows: + counts.update(row[key]) + return dict(sorted(counts.items())) + + +def _merge_source_tile_counts(rows: list[dict[str, Any]]) -> dict[str, int]: + counts: Counter[str] = Counter() + for row in rows: + counts.update( + { + f"{row['sample_slug']}:{source_tile}": count + for source_tile, count in row["source_tile_counts"].items() + } + ) + return dict(sorted(counts.items())) + + +def build_recommendations(report: dict[str, Any]) -> list[str]: + samples = report["samples"] + recommendations: list[str] = [] + ranked = sorted( + samples, + key=lambda sample: ( + sample["false_positive_count"], + sample["false_positive_rate"] or 0.0, + ), + reverse=True, + ) + if ranked: + recommendations.append( + "Review the highest false-positive volumes first: " + + ", ".join( + f"{sample['sample_slug']} ({sample['false_positive_count']})" + for sample in ranked[:3] + ) + + "." + ) + + area_buckets = report["area_buckets"] + small_count = sum( + area_buckets[key]["count"] + for key in ("tiny_lt_25_m2", "small_25_100_m2") + ) + if report["false_positive_count"]: + small_share = small_count / report["false_positive_count"] + recommendations.append( + f"Inspect tiny/small candidate geometry first when sampling hard negatives; it represents {small_share:.1%} of persisted false positives." + ) + + if report["confidence_coverage_count"] == 0: + recommendations.append( + "Persisted QA evidence has no per-detection confidence values; use the recorded run threshold and geometry/tile evidence without inventing confidence bands." + ) + elif report["confidence_coverage_count"] < report["false_positive_count"]: + recommendations.append( + "Confidence coverage is partial; do not infer a portfolio-wide confidence distribution from the covered subset." + ) + + tile_counts = report["source_tile_counts"] + if tile_counts: + top_tile, top_count = max(tile_counts.items(), key=lambda item: item[1]) + recommendations.append( + f"Start spatial review with source tile {top_tile} ({top_count} false positives), then confirm examples visually before adding any hard-negative labels." + ) + return recommendations + + +def run_audit(portfolio_path: Path, output_dir: Path) -> tuple[Path, Path]: + try: + from pyproj import Geod + from shapely.geometry import shape + except ImportError as exc: + raise SystemExit( + "False-positive GIS audit requires the existing GeoIntel GIS extras (pyproj and shapely)." + ) from exc + + portfolio_path = portfolio_path.expanduser().resolve() + portfolio = load_json(portfolio_path) + samples = portfolio.get("samples") + if not isinstance(samples, list) or not samples: + raise SystemExit(f"Evidence portfolio has no samples: {portfolio_path}") + + geod = Geod(ellps="WGS84") + sample_reports: list[dict[str, Any]] = [] + review_features: list[dict[str, Any]] = [] + all_areas: list[float] = [] + all_confidences: list[float] = [] + seen_slugs: set[str] = set() + + for sample in samples: + if not isinstance(sample, dict): + raise SystemExit("Evidence portfolio contains a non-object sample") + sample_slug = str(sample.get("sample_slug") or "").strip().lower() + if not sample_slug or sample_slug in seen_slugs: + raise SystemExit(f"Evidence portfolio has an invalid or duplicate sample_slug: {sample_slug}") + seen_slugs.add(sample_slug) + evidence_path = resolve_evidence_path(portfolio_path, sample) + sample_report, sample_features, areas, confidences = audit_sample( + load_json(evidence_path), + sample_slug=sample_slug, + geod=geod, + shape=shape, + ) + declared = (sample.get("role_counts") or {}).get("false_positive") + if declared is not None and declared != sample_report["false_positive_count"]: + raise SystemExit( + f"Sample {sample_slug} declares {declared} false positives but evidence contains {sample_report['false_positive_count']}" + ) + sample_report["evidence_geojson_path"] = str(evidence_path) + sample_reports.append(sample_report) + review_features.extend(sample_features) + all_areas.extend(areas) + all_confidences.extend(confidences) + + false_positive_count = sum(row["false_positive_count"] for row in sample_reports) + candidate_count = sum(row["candidate_count"] for row in sample_reports) + output_dir = output_dir.expanduser().resolve() + output_dir.mkdir(parents=True, exist_ok=True) + geojson_path = output_dir / "false_positives.geojson" + geojson_path.write_text( + json.dumps( + {"type": "FeatureCollection", "features": review_features}, + indent=2, + sort_keys=True, + ), + encoding="utf-8", + ) + + report = { + "generated_at": datetime.now(timezone.utc).isoformat(), + "schema_version": 1, + "portfolio_path": str(portfolio_path), + "model_asset_id": portfolio.get("model_asset_id"), + "model_sha256": portfolio.get("model_sha256"), + "input_crs": "EPSG:4326", + "area_method": "WGS84 geodesic area via pyproj.Geod", + "sample_count": len(sample_reports), + "false_positive_count": false_positive_count, + "candidate_count": candidate_count, + "false_positive_rate": false_positive_count / candidate_count if candidate_count else None, + "false_positive_area_m2": area_stats(all_areas), + "area_buckets": _bucket_summary(all_areas), + "confidence_coverage_count": len(all_confidences), + "confidence": area_stats(all_confidences), + "source_tile_counts": _merge_source_tile_counts(sample_reports), + "class_counts": _merge_counts(sample_reports, "class_counts"), + "samples": sample_reports, + "false_positive_geojson_path": str(geojson_path), + } + report["recommendations"] = build_recommendations(report) + + json_path = output_dir / "detection_false_positive_audit.json" + json_path.write_text(json.dumps(report, indent=2, sort_keys=True), encoding="utf-8") + + lines = [ + "# Detection false-positive evidence audit", + "", + f"- Generated: {report['generated_at']}", + f"- Model asset: {report['model_asset_id'] or 'not recorded'}", + f"- Model SHA256: {report['model_sha256'] or 'not recorded'}", + f"- Persisted false positives: {false_positive_count}", + f"- Candidate detections represented: {candidate_count}", + f"- Confidence coverage: {len(all_confidences)}/{false_positive_count}", + f"- Area method: {report['area_method']}", + "", + "## AOI review pressure", + "", + "| AOI | False positives | Candidate detections | FP rate | Median area m2 | Confidence coverage |", + "|---|---:|---:|---:|---:|---:|", + ] + for sample in sample_reports: + rate = sample["false_positive_rate"] + rate_text = f"{rate:.3f}" if rate is not None else "n/a" + median_area = sample["false_positive_area_m2"]["median"] + median_text = f"{median_area:.1f}" if median_area is not None else "n/a" + lines.append( + f"| {sample['sample_slug']} | {sample['false_positive_count']} | " + f"{sample['candidate_count']} | {rate_text} | {median_text} | " + f"{sample['confidence_coverage_count']}/{sample['false_positive_count']} |" + ) + lines.extend(["", "## Recommended review actions", ""]) + lines.extend(f"- {item}" for item in report["recommendations"]) + markdown_path = output_dir / "detection_false_positive_audit.md" + markdown_path.write_text("\n".join(lines) + "\n", encoding="utf-8") + return json_path, markdown_path + + +def main() -> None: + args = parse_args() + json_path, markdown_path = run_audit(args.portfolio, args.output_dir) + print(f"False-positive audit JSON: {json_path}") + print(f"False-positive audit Markdown: {markdown_path}") + + +if __name__ == "__main__": + main() diff --git a/scripts/run_readiness_check.sh b/scripts/run_readiness_check.sh index d05e13e4..b6464c58 100755 --- a/scripts/run_readiness_check.sh +++ b/scripts/run_readiness_check.sh @@ -49,6 +49,7 @@ ${PYTHON_BIN} -m py_compile scripts/build_detection_model_promotion_report.py ${PYTHON_BIN} -m py_compile scripts/build_background_corpus_split_report.py ${PYTHON_BIN} -m py_compile scripts/build_fixed_threshold_evidence_portfolio_inputs.py ${PYTHON_BIN} -m py_compile scripts/audit_detection_false_negative_evidence.py +${PYTHON_BIN} -m py_compile scripts/audit_detection_false_positive_evidence.py ${PYTHON_BIN} -m py_compile scripts/activate_promoted_yolo_candidate.py ${PYTHON_BIN} -m py_compile scripts/cleanup_demo_artifacts.py ${PYTHON_BIN} -m py_compile backend/scripts/cleanup_demo_artifacts.py