diff --git a/.env.example b/.env.example
index f981ad33..28c76bf6 100644
--- a/.env.example
+++ b/.env.example
@@ -109,6 +109,7 @@ GEOINTEL_FRONTEND_PORT=1202
GEOINTEL_BACKEND_PORT=8000
GEOINTEL_INSTALL_AI=false
GEOINTEL_STORAGE_PATH=./storage
+GEOINTEL_BACKUPS_PATH=./backups
GEOINTEL_MODELS_PATH=./models
GEOINTEL_POSTGIS_DATA_PATH=./postgres-data
GEOINTEL_POSTGRES_DB=geointel
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 16c7602b..2e3ac552 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -66,6 +66,16 @@
the live three-viewport audit passed with zero horizontal overflow,
console errors or failed requests and verified truthful delayed loading,
keyboard behavior, accessible control names and visible timing feedback.
+- Added an RC-10 read-only data-operations audit covering storage lifecycle,
+ disk pressure, persisted-path integrity, old failed work and
+ national/regional/maritime source families.
+- Added dry-run-first orphan/cache cleanup with fail-closed path categories,
+ an explicit delete ceiling, exact confirmation token and a recent
+ checksum-verified database plus SHA-256 storage-backup requirement.
+- Mounted release backups read-only at `/app/backups` and applied the same
+ backup/confirmation guard to the older demo-export cleanup path.
+- Added a live RC-10 audit that proves critical table counts remain unchanged
+ across the storage report and cleanup dry run.
- Added a read-only release-evidence manifest command with Git, migration,
dependency, configuration checksum and optional live endpoint evidence.
- Replaced the obsolete pre-build status with the current implemented
diff --git a/backend/README.md b/backend/README.md
index a8f7c3ca..5c76d8a8 100644
--- a/backend/README.md
+++ b/backend/README.md
@@ -1920,3 +1920,17 @@ file. It never imports provider data or writes directly to database tables.
Explicit demo seeding also reactivates its own archived technical project.
This keeps the opt-in fixture workflow selectable without changing the normal
active-project lifecycle.
+
+## Data operations and retention
+
+The runtime packages `audit_data_operations.py`,
+`cleanup_storage_artifacts.py` and the shared release-backup guard. The audit
+is read-only and combines disk pressure, storage lifecycle, persisted path
+integrity, failed-work counts and national/regional/maritime source-family
+inventory. Cleanup is limited to old unreferenced derived/cache/export files.
+
+Unknown paths, official source material, uploads, models and release/operator
+evidence are protected by default. Apply mode requires an exact confirmation,
+an explicit candidate ceiling and a recent checksum-verified database plus
+SHA-256 storage backup mounted read-only under `/app/backups`. See
+`docs/DATA_OPERATIONS_RUNBOOK.md`.
diff --git a/backend/scripts/cleanup_demo_artifacts.py b/backend/scripts/cleanup_demo_artifacts.py
index 127af7ae..1b377db2 100644
--- a/backend/scripts/cleanup_demo_artifacts.py
+++ b/backend/scripts/cleanup_demo_artifacts.py
@@ -9,13 +9,19 @@ from typing import Any
BACKEND_ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(BACKEND_ROOT))
+REPOSITORY_ROOT = BACKEND_ROOT.parent
+SCRIPTS_ROOT = REPOSITORY_ROOT / "scripts"
+if SCRIPTS_ROOT.is_dir():
+ sys.path.insert(0, str(SCRIPTS_ROOT))
from app.core.config import get_settings
from app.db.session import SessionLocal
from app.models import Export, Project
+from release_backup_guard import require_confirmation, verify_current_backup
DEMO_PROJECT_NAME = "GeoIntel Demo - Building QA"
+DELETE_CONFIRMATION = "DELETE_DEMO_EXPORTS"
def is_within_storage_root(path: Path, storage_root: Path) -> bool:
@@ -189,6 +195,21 @@ def build_parser() -> argparse.ArgumentParser:
help="Restrict cleanup to an export_type. Repeat for multiple types.",
)
parser.add_argument("--apply", action="store_true", help="Delete selected export rows and files.")
+ parser.add_argument(
+ "--backup-dir",
+ type=Path,
+ help="Recent checksum-verified release backup mounted read-only in the runtime.",
+ )
+ parser.add_argument(
+ "--backup-max-age-hours",
+ type=float,
+ default=24.0,
+ help="Maximum age accepted for the required release backup.",
+ )
+ parser.add_argument(
+ "--confirm",
+ help=f"Exact destructive-maintenance confirmation token: {DELETE_CONFIRMATION}",
+ )
return parser
@@ -199,6 +220,17 @@ def main() -> int:
parser.error("--keep-latest must be greater than or equal to zero")
if args.max_delete < 0:
parser.error("--max-delete must be greater than or equal to zero")
+ if args.apply:
+ try:
+ require_confirmation(args.confirm, DELETE_CONFIRMATION)
+ if args.backup_dir is None:
+ raise RuntimeError("--backup-dir is required with --apply")
+ verify_current_backup(
+ args.backup_dir,
+ max_age_hours=args.backup_max_age_hours,
+ )
+ except (RuntimeError, ValueError) as exc:
+ parser.error(str(exc))
summary = cleanup_demo_exports(
project_name=args.project_name,
diff --git a/backend/tests/test_rc10_data_operations.py b/backend/tests/test_rc10_data_operations.py
new file mode 100644
index 00000000..7b10998d
--- /dev/null
+++ b/backend/tests/test_rc10_data_operations.py
@@ -0,0 +1,259 @@
+from __future__ import annotations
+
+import hashlib
+import importlib.util
+import json
+import sys
+from datetime import datetime, timedelta, timezone
+from pathlib import Path
+from types import SimpleNamespace
+from uuid import uuid4
+
+import pytest
+
+
+ROOT = Path(__file__).resolve().parents[2]
+SCRIPTS = ROOT / "scripts"
+
+
+def load_script(name: str):
+ path = SCRIPTS / name
+ spec = importlib.util.spec_from_file_location(f"rc10_{path.stem}", path)
+ assert spec is not None and spec.loader is not None
+ module = importlib.util.module_from_spec(spec)
+ sys.modules[spec.name] = module
+ spec.loader.exec_module(module)
+ return module
+
+
+def write_backup(root: Path, *, created_at: datetime, inventory_mode: str = "sha256") -> None:
+ root.mkdir(parents=True)
+ manifest = {
+ "schema_version": 1,
+ "release_id": "rc10-test",
+ "created_at": created_at.isoformat(),
+ "read_only_source": True,
+ "database_password_secure": True,
+ "inventory_mode": inventory_mode,
+ "storage_inventory_requested": True,
+ "git_commit": "0123456789abcdef",
+ }
+ files = {
+ "manifest.json": json.dumps(manifest),
+ "database.dump": "database",
+ "database.list": "list",
+ "database-metadata.tsv": "alembic_head\t202607160001",
+ "table-counts.tsv": "datasets\t1",
+ "storage-manifest.tsv": "relative_path\tsize_bytes\tmtime_ns\tsha256",
+ }
+ for name, content in files.items():
+ (root / name).write_text(content, encoding="utf-8")
+ checksums = []
+ for name in sorted(files):
+ digest = hashlib.sha256((root / name).read_bytes()).hexdigest()
+ checksums.append(f"{digest} {name}")
+ (root / "CHECKSUMS.sha256").write_text("\n".join(checksums) + "\n", encoding="utf-8")
+
+
+def test_backup_guard_requires_recent_complete_sha256_storage_backup(tmp_path: Path) -> None:
+ guard = load_script("release_backup_guard.py")
+ now = datetime(2026, 7, 18, 12, tzinfo=timezone.utc)
+ backup = tmp_path / "backup"
+ write_backup(backup, created_at=now - timedelta(hours=2))
+
+ verified = guard.verify_current_backup(backup, now=now)
+
+ assert verified.release_id == "rc10-test"
+ assert verified.age_hours == pytest.approx(2)
+
+
+def test_backup_guard_rejects_stale_or_tampered_backup(tmp_path: Path) -> None:
+ guard = load_script("release_backup_guard.py")
+ now = datetime(2026, 7, 18, 12, tzinfo=timezone.utc)
+ stale = tmp_path / "stale"
+ write_backup(stale, created_at=now - timedelta(hours=30))
+ with pytest.raises(RuntimeError, match="maximum allowed age"):
+ guard.verify_current_backup(stale, now=now)
+
+ current = tmp_path / "tampered"
+ write_backup(current, created_at=now)
+ (current / "database.dump").write_text("tampered", encoding="utf-8")
+ with pytest.raises(RuntimeError, match="checksum mismatch"):
+ guard.verify_current_backup(current, now=now)
+
+
+def test_storage_lifecycle_is_fail_closed_and_protects_release_evidence() -> None:
+ audit = load_script("audit_data_operations.py")
+
+ assert audit.classify_relative_path("release-evidence/rc11/manifest.json") == (
+ "release-evidence",
+ True,
+ False,
+ )
+ assert audit.classify_relative_path("operator-evidence/source/raw.json")[1:] == (True, False)
+ assert audit.classify_relative_path("uploads/project/data.geojson")[1:] == (True, False)
+ assert audit.classify_relative_path("exports/project/old.json")[1:] == (False, True)
+ assert audit.classify_relative_path("unknown/value.bin")[1:] == (True, False)
+
+
+def test_storage_audit_only_selects_old_unreferenced_allowlisted_files(
+ tmp_path: Path,
+ monkeypatch: pytest.MonkeyPatch,
+) -> None:
+ audit = load_script("audit_data_operations.py")
+ storage = tmp_path / "storage"
+ old_orphan = storage / "exports" / "project" / "old.json"
+ referenced = storage / "exports" / "project" / "kept.json"
+ protected = storage / "release-evidence" / "rc" / "manifest.json"
+ unknown = storage / "misc" / "unknown.bin"
+ for path in (old_orphan, referenced, protected, unknown):
+ path.parent.mkdir(parents=True, exist_ok=True)
+ path.write_text(path.name, encoding="utf-8")
+ old_timestamp = (datetime.now(timezone.utc) - timedelta(days=30)).timestamp()
+ for path in (old_orphan, referenced, protected, unknown):
+ path.touch()
+ Path(path).chmod(0o644)
+ import os
+
+ os.utime(path, (old_timestamp, old_timestamp))
+
+ monkeypatch.setattr(
+ audit,
+ "collect_database_state",
+ lambda _db, _root: {
+ "references": {referenced.resolve()},
+ "counts": {},
+ "source_families": {"national": [], "regional": [], "maritime": []},
+ },
+ )
+ monkeypatch.setattr(
+ audit,
+ "disk_pressure",
+ lambda _root: {
+ "status": "ok",
+ "total_bytes": 100,
+ "used_bytes": 50,
+ "free_bytes": 50,
+ "free_percent": 50.0,
+ "acquisition_allowed": True,
+ },
+ )
+
+ report, candidates = audit.build_report(storage, SimpleNamespace(), minimum_age_days=7)
+
+ assert [candidate.relative_path for candidate in candidates] == ["exports/project/old.json"]
+ assert report["cleanup"]["candidate_count"] == 1
+ assert "release-evidence" in report["cleanup"]["protected_prefixes"]
+
+
+def test_source_family_report_covers_national_regional_and_maritime(tmp_path: Path) -> None:
+ audit = load_script("audit_data_operations.py")
+ national_id = uuid4()
+ regional_id = uuid4()
+ now = datetime.now(timezone.utc)
+ rows = {
+ audit.Project: [
+ SimpleNamespace(id=national_id, name=audit.NATIONAL_PROJECT_NAME, status="active"),
+ SimpleNamespace(id=regional_id, name="Wallonia operator", status="active"),
+ ],
+ audit.Dataset: [
+ SimpleNamespace(
+ project_id=national_id,
+ source_name="ngi_adminvector",
+ source="ngi",
+ name="Belgium boundary",
+ source_metadata={"coverage_zones": ["belgium"]},
+ source_version="2026",
+ imported_at=now,
+ status="ready",
+ storage_path=None,
+ metadata_json=None,
+ provenance_metadata=None,
+ ),
+ SimpleNamespace(
+ project_id=national_id,
+ source_name="rbins_marine_reporting_units",
+ source="rbins",
+ name="Belgian North Sea",
+ source_metadata={"coverage_zones": ["belgian_north_sea"]},
+ source_version="2024",
+ imported_at=now,
+ status="ready",
+ storage_path=None,
+ metadata_json=None,
+ provenance_metadata=None,
+ ),
+ SimpleNamespace(
+ project_id=regional_id,
+ source_name="wallonia_manual",
+ source="manual",
+ name="Wallonia source",
+ source_metadata={},
+ source_version="1",
+ imported_at=now,
+ status="ready",
+ storage_path=None,
+ metadata_json=None,
+ provenance_metadata=None,
+ ),
+ ],
+ audit.DatasetVersion: [],
+ audit.Export: [],
+ audit.Detection: [],
+ audit.Segmentation: [],
+ audit.Job: [],
+ audit.AnalysisRun: [],
+ }
+
+ class Query:
+ def __init__(self, values):
+ self.values = values
+
+ def all(self):
+ return self.values
+
+ class Session:
+ def query(self, model):
+ return Query(rows[model])
+
+ state = audit.collect_database_state(Session(), tmp_path)
+
+ assert {item["source_name"] for item in state["source_families"]["national"]} == {
+ "ngi_adminvector",
+ "rbins_marine_reporting_units",
+ }
+ assert {item["source_name"] for item in state["source_families"]["maritime"]} == {
+ "rbins_marine_reporting_units"
+ }
+ assert {item["source_name"] for item in state["source_families"]["regional"]} == {
+ "wallonia_manual"
+ }
+
+
+def test_cleanup_commands_require_backup_confirmation_and_read_only_mount() -> None:
+ generic = (SCRIPTS / "cleanup_storage_artifacts.py").read_text(encoding="utf-8")
+ demo = (ROOT / "backend/scripts/cleanup_demo_artifacts.py").read_text(encoding="utf-8")
+ compose = (ROOT / "docker-compose.unraid.yml").read_text(encoding="utf-8")
+ dockerman = (ROOT / "deploy/unraid/run-dockerman-container.sh").read_text(encoding="utf-8")
+ dockerfile = (ROOT / "deploy/unraid/Dockerfile.all-in-one").read_text(encoding="utf-8")
+ readiness = (SCRIPTS / "run_readiness_check.sh").read_text(encoding="utf-8")
+ live_audit = (SCRIPTS / "run_rc10_data_operations_audit.sh").read_text(encoding="utf-8")
+
+ assert "DELETE_STORAGE_ARTIFACTS" in generic
+ assert "verify_current_backup" in generic
+ assert "DELETE_DEMO_EXPORTS" in demo
+ assert "verify_current_backup" in demo
+ assert "/app/backups:ro" in compose
+ assert '/app/backups:ro"' in dockerman
+ for name in (
+ "release_backup_guard.py",
+ "audit_data_operations.py",
+ "cleanup_storage_artifacts.py",
+ ):
+ assert f"COPY scripts/{name}" in dockerfile
+ assert f"py_compile scripts/{name}" in readiness
+ assert "bash -n scripts/run_rc10_data_operations_audit.sh" in readiness
+ assert "--apply" not in live_audit
+ assert "table-counts-before.tsv" in live_audit
+ assert "table-counts-after.tsv" in live_audit
+ assert "deleted_count" in live_audit
diff --git a/backend/tests/test_sprint31_unraid_template.py b/backend/tests/test_sprint31_unraid_template.py
index 727547ed..7d40f2de 100644
--- a/backend/tests/test_sprint31_unraid_template.py
+++ b/backend/tests/test_sprint31_unraid_template.py
@@ -65,6 +65,7 @@ def test_unraid_template_exposes_every_operator_owned_runtime_setting() -> None:
template_variables = set(re.findall(r'Target="([A-Z][A-Z0-9_]+)"', template))
bridged_or_internal = {
"GEOINTEL_FRONTEND_PORT",
+ "GEOINTEL_BACKUPS_PATH",
"GEOINTEL_IMAGE",
"GEOINTEL_MODELS_PATH",
"GEOINTEL_POSTGIS_DATA_PATH",
diff --git a/deploy/unraid/Dockerfile.all-in-one b/deploy/unraid/Dockerfile.all-in-one
index 77b828be..2e1d022e 100644
--- a/deploy/unraid/Dockerfile.all-in-one
+++ b/deploy/unraid/Dockerfile.all-in-one
@@ -142,6 +142,9 @@ COPY scripts/run_split_background_promotion_workflow.sh /app/scripts/run_split_b
COPY scripts/activate_promoted_yolo_candidate.py /app/scripts/activate_promoted_yolo_candidate.py
COPY scripts/archive_technical_projects.py /app/scripts/archive_technical_projects.py
COPY scripts/runtime_state_report.py /app/scripts/runtime_state_report.py
+COPY scripts/release_backup_guard.py /app/scripts/release_backup_guard.py
+COPY scripts/audit_data_operations.py /app/scripts/audit_data_operations.py
+COPY scripts/cleanup_storage_artifacts.py /app/scripts/cleanup_storage_artifacts.py
COPY deploy/unraid/nginx-all-in-one.conf /etc/nginx/conf.d/default.conf
COPY deploy/unraid/all-in-one-start.sh /usr/local/bin/geointel-all-in-one-start
COPY --from=frontend-build /frontend/dist/ /usr/share/nginx/html/
@@ -171,7 +174,7 @@ LABEL org.opencontainers.image.title="GeoIntel" \
org.opencontainers.image.created="${GEOINTEL_BUILD_TIME}" \
io.geointel.ai.enabled="${GEOINTEL_INSTALL_AI}"
-VOLUME ["/var/lib/postgresql/data", "/app/storage"]
+VOLUME ["/var/lib/postgresql/data", "/app/storage", "/app/backups"]
EXPOSE 80
diff --git a/deploy/unraid/README.md b/deploy/unraid/README.md
index ba4e443d..59f46b0a 100644
--- a/deploy/unraid/README.md
+++ b/deploy/unraid/README.md
@@ -293,6 +293,24 @@ The configured upload limit is shared by FastAPI and the generated nginx
runtime configuration. Values outside `1..2048` MiB are rejected before the
active application is replaced.
+## Data operations
+
+`GEOINTEL_BACKUPS_PATH` defaults to
+`/mnt/user/appdata/geointel/backups` and is mounted read-only at
+`/app/backups`. Explicit cleanup commands can therefore verify a recent
+backup without permission to alter it.
+
+These commands are non-mutating:
+
+```bash
+docker exec geointel python /app/scripts/audit_data_operations.py
+docker exec geointel python /app/scripts/cleanup_storage_artifacts.py
+```
+
+The full backup, confirmation, candidate-limit and apply sequence is in
+`docs/DATA_OPERATIONS_RUNBOOK.md`. GeoIntel installs no automatic cleanup
+schedule.
+
## Safe cleanup
Safe cache cleanup if Docker build cache fills the Unraid Docker image:
diff --git a/deploy/unraid/geointel-unraid-template.xml b/deploy/unraid/geointel-unraid-template.xml
index d3db68db..2f0daf00 100644
--- a/deploy/unraid/geointel-unraid-template.xml
+++ b/deploy/unraid/geointel-unraid-template.xml
@@ -25,6 +25,7 @@
1202
/mnt/user/appdata/geointel/storage
/mnt/user/appdata/geointel/models
+ /mnt/user/appdata/geointel/backups
/mnt/user/appdata/geointel/postgres-data
geointel
geointel
diff --git a/deploy/unraid/geointel.env.example b/deploy/unraid/geointel.env.example
index a84cda1c..4c3af412 100644
--- a/deploy/unraid/geointel.env.example
+++ b/deploy/unraid/geointel.env.example
@@ -10,6 +10,9 @@ GEOINTEL_STORAGE_PATH=/mnt/user/appdata/geointel/storage
# Local AI model files mounted into the container as /app/models.
GEOINTEL_MODELS_PATH=/mnt/user/appdata/geointel/models
+# Checksum-verified release backups mounted read-only for cleanup guards.
+GEOINTEL_BACKUPS_PATH=/mnt/user/appdata/geointel/backups
+
# Embedded PostGIS data directory for the all-in-one container.
GEOINTEL_POSTGIS_DATA_PATH=/mnt/user/appdata/geointel/postgres-data
diff --git a/deploy/unraid/run-dockerman-container.sh b/deploy/unraid/run-dockerman-container.sh
index 27dbcabf..cced617b 100644
--- a/deploy/unraid/run-dockerman-container.sh
+++ b/deploy/unraid/run-dockerman-container.sh
@@ -15,6 +15,7 @@ GEOINTEL_FRONTEND_PORT="${GEOINTEL_FRONTEND_PORT:-1202}"
GEOINTEL_IMAGE="${GEOINTEL_IMAGE:-geointel-all-in-one:latest}"
GEOINTEL_STORAGE_PATH="${GEOINTEL_STORAGE_PATH:-/mnt/user/appdata/geointel/storage}"
GEOINTEL_MODELS_PATH="${GEOINTEL_MODELS_PATH:-/mnt/user/appdata/geointel/models}"
+GEOINTEL_BACKUPS_PATH="${GEOINTEL_BACKUPS_PATH:-/mnt/user/appdata/geointel/backups}"
GEOINTEL_POSTGIS_DATA_PATH="${GEOINTEL_POSTGIS_DATA_PATH:-/mnt/user/appdata/geointel/postgres-data}"
GEOINTEL_POSTGRES_DB="${GEOINTEL_POSTGRES_DB:-geointel}"
GEOINTEL_POSTGRES_USER="${GEOINTEL_POSTGRES_USER:-geointel}"
@@ -168,7 +169,7 @@ if docker ps -a --format '{{.Names}}' | grep -qx geointel; then
docker rm -f geointel
fi
-mkdir -p "$GEOINTEL_STORAGE_PATH" "$GEOINTEL_MODELS_PATH" "$GEOINTEL_POSTGIS_DATA_PATH"
+mkdir -p "$GEOINTEL_STORAGE_PATH" "$GEOINTEL_MODELS_PATH" "$GEOINTEL_BACKUPS_PATH" "$GEOINTEL_POSTGIS_DATA_PATH"
migrate_compose_volume_if_needed
docker run -d \
@@ -268,6 +269,7 @@ docker run -d \
-v "${GEOINTEL_POSTGIS_DATA_PATH}:/var/lib/postgresql/data" \
-v "${GEOINTEL_STORAGE_PATH}:/app/storage" \
-v "${GEOINTEL_MODELS_PATH}:/app/models" \
+ -v "${GEOINTEL_BACKUPS_PATH}:/app/backups:ro" \
"$GEOINTEL_IMAGE"
docker ps --filter name=geointel
diff --git a/docker-compose.unraid.yml b/docker-compose.unraid.yml
index 0e03a5d2..0a014cdf 100644
--- a/docker-compose.unraid.yml
+++ b/docker-compose.unraid.yml
@@ -89,6 +89,7 @@ services:
- ${GEOINTEL_POSTGIS_DATA_PATH:-geointel_postgis}:/var/lib/postgresql/data
- ${GEOINTEL_STORAGE_PATH:-./storage}:/app/storage
- ${GEOINTEL_MODELS_PATH:-./models}:/app/models
+ - ${GEOINTEL_BACKUPS_PATH:-./backups}:/app/backups:ro
extra_hosts:
- "host.docker.internal:host-gateway"
restart: unless-stopped
diff --git a/docker-compose.yml b/docker-compose.yml
index c6c4935d..e4fc00b3 100644
--- a/docker-compose.yml
+++ b/docker-compose.yml
@@ -119,6 +119,7 @@ services:
volumes:
- ${GEOINTEL_STORAGE_PATH:-./storage}:/app/storage
- ${GEOINTEL_MODELS_PATH:-./models}:/app/models
+ - ${GEOINTEL_BACKUPS_PATH:-./backups}:/app/backups:ro
- ./fixtures:/app/fixtures:ro
extra_hosts:
- "host.docker.internal:host-gateway"
diff --git a/docs/DATA_OPERATIONS_RUNBOOK.md b/docs/DATA_OPERATIONS_RUNBOOK.md
new file mode 100644
index 00000000..0156e721
--- /dev/null
+++ b/docs/DATA_OPERATIONS_RUNBOOK.md
@@ -0,0 +1,145 @@
+# Data Operations and Retention Runbook
+
+## Purpose
+
+This runbook governs persistent GeoIntel data for Belgium and the Belgian
+North Sea. It covers raw, normalized, derived, export, AI and release-evidence
+artifacts. No command in this runbook schedules itself and no source is
+downloaded implicitly.
+
+## Lifecycle classes
+
+| Class | Paths | Retention |
+|---|---|---|
+| Raw and normalized source | `originals/`, `uploads/`, `operator-data/` | Protected; replace no official edition in place |
+| Derived and cache | `derived/`, `rasters/derived/`, `previews/`, `tiles/`, `masks/` | Eligible only when old and not referenced by persisted provenance |
+| Exports | `exports/` | Retain persisted exports; unreferenced old files may be candidates |
+| AI models | `/app/models` and `storage/models/` | Protected; activation and promotion evidence govern removal |
+| Immutable evidence | `release-evidence/`, `operator-evidence/` | Never a cleanup candidate |
+| Unknown | every unclassified path | Protected by default |
+
+Dataset, DatasetVersion, Export, Detection, Segmentation, Job and AnalysisRun
+paths and nested provenance are treated as references. Failed work remains
+audit provenance. Project archival remains reversible and does not physically
+delete its datasets or files.
+
+## Read-only audit
+
+Run the complete local storage, database-reference, disk-pressure and
+national/regional/maritime source-family report:
+
+```bash
+docker exec geointel python /app/scripts/audit_data_operations.py \
+ --minimum-age-days 7 \
+ --output /app/storage/release-evidence/rc-current/data-operations.json
+```
+
+The command:
+
+- inventories storage without following symlinks;
+- reports bytes and file counts per lifecycle class;
+- reports free disk bytes and percentage before new acquisition;
+- reports missing persisted path references;
+- identifies only old, unreferenced files inside the explicit cleanup
+ allowlist;
+- summarizes national, regional and maritime source names, versions and latest
+ import dates;
+- reports old failed Jobs and AnalysisRuns without deleting them.
+
+Use `--fail-on-pressure warning` for a stricter acquisition preflight and
+`--fail-on-missing-reference` for release gates. `critical` disk pressure
+blocks acquisition by default. The thresholds are 10 GiB or 5% free for
+critical and 25 GiB or 10% free for warning.
+
+The existing project-level freshness policy remains authoritative:
+
+```bash
+docker exec geointel python /app/scripts/audit_source_freshness.py \
+ --project-id \
+ --api-url http://127.0.0.1/api/v1 \
+ --json \
+ --output /app/storage/release-evidence/rc-current/source-freshness.json
+```
+
+Together, these reports cover the national workbench, regional source
+families and Belgian North Sea datasets without performing an external
+refresh. Catalog probes remain separate, explicit and read-only.
+
+## Cleanup dry run
+
+```bash
+docker exec geointel python /app/scripts/cleanup_storage_artifacts.py \
+ --minimum-age-days 7 \
+ --max-delete 25
+```
+
+This is always a dry run unless `--apply` is present. Review every candidate.
+Release evidence, operator evidence, official source material, uploads,
+models, unknown paths and every persisted path reference are excluded.
+
+Technical benchmark projects can be archived reversibly:
+
+```bash
+docker exec geointel python /app/scripts/archive_technical_projects.py --show-names
+```
+
+The normal project lifecycle archive path does not need destructive
+confirmation because it changes only `status=archived` and preserves all
+data.
+
+## Destructive apply gate
+
+First create a fresh backup with a SHA-256 storage inventory on the host:
+
+```bash
+bash scripts/backup_release_state.sh \
+ --container geointel \
+ --release-id rc10-before-cleanup-$(date -u +%Y%m%dT%H%M%SZ) \
+ --output-root /mnt/user/appdata/geointel/backups \
+ --storage-path /mnt/user/appdata/geointel/storage \
+ --models-path /mnt/user/appdata/geointel/models \
+ --inventory-mode sha256
+```
+
+Verify it read-only:
+
+```bash
+bash scripts/verify_release_backup.sh \
+ --backup-dir /mnt/user/appdata/geointel/backups/ \
+ --container geointel
+```
+
+The Unraid runtime mounts `GEOINTEL_BACKUPS_PATH` read-only at `/app/backups`.
+Only after reviewing the dry run may an operator execute:
+
+```bash
+docker exec geointel python /app/scripts/cleanup_storage_artifacts.py \
+ --minimum-age-days 7 \
+ --max-delete \
+ --backup-dir /app/backups/ \
+ --backup-max-age-hours 24 \
+ --confirm DELETE_STORAGE_ARTIFACTS \
+ --apply
+```
+
+The command re-runs the audit immediately before deletion. It refuses the
+operation when the exact token is absent, the candidate count exceeds the
+operator limit, the backup is stale/incomplete, checksums differ, the storage
+inventory is not SHA-256, or the path is outside the cleanup allowlist.
+
+The older demo-export cleanup has the same gate and uses confirmation token
+`DELETE_DEMO_EXPORTS`.
+
+## Failure meaning
+
+- `disk_pressure=critical`: stop new source acquisition and resolve capacity.
+- `missing_referenced_path_count>0`: persisted provenance points to absent
+ files; investigate before cleanup or release.
+- checksum or backup-age failure: create and verify a new backup.
+- candidate count above `--max-delete`: keep dry-run mode and review; never
+ increase the limit blindly.
+- symlink listed as skipped: inspect manually; the audit will not traverse or
+ delete it.
+
+No automatic cron or background cleanup is installed. Operators may schedule
+the read-only audit externally, but destructive commands must remain manual.
diff --git a/docs/STORAGE_ARCHITECTURE.md b/docs/STORAGE_ARCHITECTURE.md
index a7e73375..50238f57 100644
--- a/docs/STORAGE_ARCHITECTURE.md
+++ b/docs/STORAGE_ARCHITECTURE.md
@@ -321,6 +321,21 @@ The smoke never passes `--apply`. It fails if the cleanup summary is not a
dry-run, if any export/file deletion is reported, or if the dry-run candidate
fields are missing.
+## RC-10 data operations
+
+`docs/DATA_OPERATIONS_RUNBOOK.md` is the executable retention source of truth.
+`scripts/audit_data_operations.py` reports storage growth, disk pressure,
+persisted path integrity, old failed work and national/regional/maritime
+source families without mutation. `scripts/cleanup_storage_artifacts.py` can
+only remove old unreferenced artifacts from the explicit derived/cache/export
+allowlist.
+
+Every unknown category, upload, original, model, operator evidence and release
+evidence path is protected by default. Apply mode requires an exact
+confirmation token and a recent checksum-verified release backup containing a
+SHA-256 storage inventory. The backup root is mounted read-only at
+`/app/backups`; no cleanup is scheduled implicitly.
+
## Governed cross-domain raster and soil evidence
MercatorNet thematic products use the ordinary raster Dataset and immutable
diff --git a/scripts/README.md b/scripts/README.md
index 22c439b0..79d7c19c 100644
--- a/scripts/README.md
+++ b/scripts/README.md
@@ -2037,6 +2037,30 @@ The command atomically updates the operator-owned `.env`, changes the matching
PostgreSQL role and recreates the container. A failed role change restores the
previous environment file. The generated secret is never printed.
+## RC-10 data operations and retention
+
+Run the read-only storage, provenance, disk-pressure and source-family audit:
+
+```bash
+docker exec geointel python /app/scripts/audit_data_operations.py \
+ --minimum-age-days 7 \
+ --output /app/storage/release-evidence/rc-current/data-operations.json
+```
+
+Preview old unreferenced derived/cache/export candidates without deletion:
+
+```bash
+docker exec geointel python /app/scripts/cleanup_storage_artifacts.py \
+ --minimum-age-days 7 \
+ --max-delete 25
+```
+
+Apply requires a reviewed candidate count, the exact
+`DELETE_STORAGE_ARTIFACTS` token and a backup no older than 24 hours with a
+checksum-verified database dump and SHA-256 storage inventory. The host backup
+root is mounted read-only at `/app/backups`. See
+`docs/DATA_OPERATIONS_RUNBOOK.md`. No cleanup is scheduled by GeoIntel.
+
## RC-8 Belgium/North Sea release journeys
Preview the seven release areas without mutating GeoIntel:
diff --git a/scripts/audit_data_operations.py b/scripts/audit_data_operations.py
new file mode 100644
index 00000000..743291e9
--- /dev/null
+++ b/scripts/audit_data_operations.py
@@ -0,0 +1,446 @@
+#!/usr/bin/env python3
+"""Read-only storage, provenance and source-family audit for GeoIntel."""
+
+from __future__ import annotations
+
+import argparse
+import json
+import os
+import shutil
+import sys
+from collections import defaultdict
+from dataclasses import dataclass
+from datetime import datetime, timedelta, timezone
+from pathlib import Path
+from typing import Any, Iterable
+
+
+ROOT = Path(__file__).resolve().parents[1]
+BACKEND_ROOT = ROOT / "backend" if (ROOT / "backend" / "app").is_dir() else ROOT
+if str(BACKEND_ROOT) not in sys.path:
+ sys.path.insert(0, str(BACKEND_ROOT))
+
+from app.core.config import get_settings
+from app.db.session import SessionLocal
+from app.models import AnalysisRun, Dataset, DatasetVersion, Detection, Export, Job, Project, Segmentation
+
+
+NATIONAL_PROJECT_NAME = "Belgium and North Sea Workbench"
+PROTECTED_PREFIXES = (
+ "release-evidence",
+ "operator-evidence",
+ "operator-data",
+ "originals",
+ "uploads",
+ "models",
+)
+CLEANUP_PREFIXES = (
+ "exports",
+ "previews",
+ "tiles",
+ "masks",
+ "derived",
+ "rasters/derived",
+)
+IGNORED_FILENAMES = frozenset({".gitkeep", "README.md"})
+PATH_METADATA_FIELDS = (
+ "metadata_json",
+ "source_metadata",
+ "provenance_metadata",
+ "parameters_json",
+ "result_json",
+ "properties_json",
+ "provenance_json",
+ "bbox_json",
+)
+
+
+@dataclass(frozen=True)
+class FileRecord:
+ path: Path
+ relative_path: str
+ category: str
+ size_bytes: int
+ modified_at: datetime
+ protected: bool
+ cleanup_eligible: bool
+
+
+def utc_now() -> datetime:
+ return datetime.now(timezone.utc)
+
+
+def _prefix_match(relative_path: str, prefixes: Iterable[str]) -> str | None:
+ normalized = relative_path.strip("/")
+ matches = [
+ prefix
+ for prefix in prefixes
+ if normalized == prefix or normalized.startswith(f"{prefix}/")
+ ]
+ return max(matches, key=len) if matches else None
+
+
+def classify_relative_path(relative_path: str) -> tuple[str, bool, bool]:
+ protected = _prefix_match(relative_path, PROTECTED_PREFIXES)
+ if protected:
+ return protected, True, False
+ cleanup = _prefix_match(relative_path, CLEANUP_PREFIXES)
+ if cleanup:
+ return cleanup, False, True
+ category = relative_path.split("/", 1)[0] if relative_path else "."
+ return category, True, False
+
+
+def inventory_storage(storage_root: Path) -> tuple[list[FileRecord], list[str]]:
+ root = storage_root.resolve()
+ records: list[FileRecord] = []
+ skipped_symlinks: list[str] = []
+ for current, directories, filenames in os.walk(root, followlinks=False):
+ current_path = Path(current)
+ retained_directories: list[str] = []
+ for directory in directories:
+ child = current_path / directory
+ if child.is_symlink():
+ skipped_symlinks.append(child.relative_to(root).as_posix())
+ else:
+ retained_directories.append(directory)
+ directories[:] = retained_directories
+ for filename in filenames:
+ path = current_path / filename
+ if path.is_symlink():
+ skipped_symlinks.append(path.relative_to(root).as_posix())
+ continue
+ try:
+ stat = path.stat()
+ except OSError:
+ continue
+ relative = path.relative_to(root).as_posix()
+ category, protected, cleanup_eligible = classify_relative_path(relative)
+ records.append(
+ FileRecord(
+ path=path.resolve(),
+ relative_path=relative,
+ category=category,
+ size_bytes=stat.st_size,
+ modified_at=datetime.fromtimestamp(stat.st_mtime, tz=timezone.utc),
+ protected=protected,
+ cleanup_eligible=cleanup_eligible,
+ )
+ )
+ return records, sorted(skipped_symlinks)
+
+
+def _iter_strings(value: Any) -> Iterable[str]:
+ if isinstance(value, str):
+ yield value
+ elif isinstance(value, dict):
+ for nested in value.values():
+ yield from _iter_strings(nested)
+ elif isinstance(value, (list, tuple)):
+ for nested in value:
+ yield from _iter_strings(nested)
+
+
+def normalize_storage_reference(value: str | None, storage_root: Path) -> Path | None:
+ if not value:
+ return None
+ stripped = value.strip()
+ if not stripped or "://" in stripped or stripped.startswith("/vsi"):
+ return None
+ candidate = Path(stripped)
+ if not candidate.is_absolute():
+ normalized = stripped.replace("\\", "/")
+ if normalized.startswith("storage/"):
+ normalized = normalized.removeprefix("storage/")
+ if _prefix_match(normalized, PROTECTED_PREFIXES + CLEANUP_PREFIXES) is None:
+ return None
+ candidate = storage_root / normalized
+ try:
+ resolved = candidate.resolve()
+ resolved.relative_to(storage_root.resolve())
+ except (OSError, ValueError):
+ return None
+ return resolved
+
+
+def _add_row_references(target: set[Path], row: Any, storage_root: Path, direct_fields: Iterable[str]) -> None:
+ for field in direct_fields:
+ normalized = normalize_storage_reference(getattr(row, field, None), storage_root)
+ if normalized:
+ target.add(normalized)
+ for field in PATH_METADATA_FIELDS:
+ for value in _iter_strings(getattr(row, field, None)):
+ normalized = normalize_storage_reference(value, storage_root)
+ if normalized:
+ target.add(normalized)
+
+
+def collect_database_state(db: Any, storage_root: Path) -> dict[str, Any]:
+ projects = db.query(Project).all()
+ project_names = {project.id: project.name for project in projects}
+ datasets = db.query(Dataset).all()
+ versions = db.query(DatasetVersion).all()
+ exports = db.query(Export).all()
+ detections = db.query(Detection).all()
+ segmentations = db.query(Segmentation).all()
+ jobs = db.query(Job).all()
+ analysis_runs = db.query(AnalysisRun).all()
+
+ references: set[Path] = set()
+ for row in datasets:
+ _add_row_references(references, row, storage_root, ("storage_path",))
+ for row in versions:
+ _add_row_references(references, row, storage_root, ("storage_path",))
+ for row in exports:
+ _add_row_references(references, row, storage_root, ("storage_path",))
+ for row in detections:
+ _add_row_references(references, row, storage_root, ("source_tile_path",))
+ for row in segmentations:
+ _add_row_references(references, row, storage_root, ("mask_path", "source_tile_path"))
+ for row in jobs:
+ _add_row_references(references, row, storage_root, ())
+ for row in analysis_runs:
+ _add_row_references(references, row, storage_root, ())
+
+ source_families: dict[str, dict[str, dict[str, Any]]] = {
+ "national": {},
+ "regional": {},
+ "maritime": {},
+ }
+ maritime_tokens = ("marine", "maritime", "north_sea", "bathymetry", "rbin", "mdk", "msp")
+ for dataset in datasets:
+ source_name = (dataset.source_name or dataset.source or "unknown").strip().lower()
+ metadata = dataset.source_metadata or {}
+ zones = metadata.get("coverage_zones") or metadata.get("coverage_zone") or []
+ if isinstance(zones, str):
+ zones = [zones]
+ families: set[str] = set()
+ project_name = project_names.get(dataset.project_id, "")
+ if project_name == NATIONAL_PROJECT_NAME or "belgium" in zones:
+ families.add("national")
+ else:
+ families.add("regional")
+ searchable = " ".join([source_name, dataset.name.lower(), *(str(zone).lower() for zone in zones)])
+ if "belgian_north_sea" in zones or any(token in searchable for token in maritime_tokens):
+ families.add("maritime")
+ for family in families:
+ item = source_families[family].setdefault(
+ source_name,
+ {
+ "source_name": source_name,
+ "dataset_count": 0,
+ "ready_count": 0,
+ "latest_imported_at": None,
+ "source_versions": set(),
+ },
+ )
+ item["dataset_count"] += 1
+ item["ready_count"] += int(dataset.status == "ready")
+ if dataset.source_version:
+ item["source_versions"].add(dataset.source_version)
+ if dataset.imported_at and (
+ item["latest_imported_at"] is None or dataset.imported_at > item["latest_imported_at"]
+ ):
+ item["latest_imported_at"] = dataset.imported_at
+
+ serialized_families: dict[str, list[dict[str, Any]]] = {}
+ for family, sources in source_families.items():
+ serialized_families[family] = []
+ for source in sorted(sources.values(), key=lambda item: item["source_name"]):
+ serialized_families[family].append(
+ {
+ **source,
+ "latest_imported_at": (
+ source["latest_imported_at"].isoformat()
+ if source["latest_imported_at"] is not None
+ else None
+ ),
+ "source_versions": sorted(source["source_versions"]),
+ }
+ )
+
+ failed_cutoff = utc_now() - timedelta(days=7)
+ return {
+ "references": references,
+ "counts": {
+ "projects": len(projects),
+ "active_projects": sum(project.status == "active" for project in projects),
+ "archived_projects": sum(project.status == "archived" for project in projects),
+ "datasets": len(datasets),
+ "dataset_versions": len(versions),
+ "exports": len(exports),
+ "detections": len(detections),
+ "segmentations": len(segmentations),
+ "jobs": len(jobs),
+ "analysis_runs": len(analysis_runs),
+ "failed_jobs_older_than_7d": sum(
+ job.status == "failed" and job.created_at and job.created_at < failed_cutoff for job in jobs
+ ),
+ "failed_runs_older_than_7d": sum(
+ run.status == "failed" and run.created_at and run.created_at < failed_cutoff
+ for run in analysis_runs
+ ),
+ },
+ "source_families": serialized_families,
+ }
+
+
+def disk_pressure(storage_root: Path) -> dict[str, Any]:
+ usage = shutil.disk_usage(storage_root)
+ free_percent = (usage.free / usage.total * 100) if usage.total else 0.0
+ if usage.free < 10 * 1024**3 or free_percent < 5:
+ status = "critical"
+ elif usage.free < 25 * 1024**3 or free_percent < 10:
+ status = "warning"
+ else:
+ status = "ok"
+ return {
+ "status": status,
+ "total_bytes": usage.total,
+ "used_bytes": usage.used,
+ "free_bytes": usage.free,
+ "free_percent": round(free_percent, 2),
+ "acquisition_allowed": status != "critical",
+ }
+
+
+def build_report(
+ storage_root: Path,
+ db: Any,
+ *,
+ minimum_age_days: int = 7,
+ max_candidate_records: int = 500,
+) -> tuple[dict[str, Any], list[FileRecord]]:
+ if minimum_age_days < 1:
+ raise ValueError("minimum_age_days must be at least one")
+ root = storage_root.expanduser().resolve()
+ if not root.is_dir():
+ raise RuntimeError(f"Storage root is not a directory: {root}")
+ records, skipped_symlinks = inventory_storage(root)
+ database = collect_database_state(db, root)
+ references: set[Path] = database.pop("references")
+ cutoff = utc_now() - timedelta(days=minimum_age_days)
+
+ categories: dict[str, dict[str, Any]] = defaultdict(
+ lambda: {"file_count": 0, "size_bytes": 0, "protected": False, "cleanup_eligible": False}
+ )
+ for record in records:
+ item = categories[record.category]
+ item["file_count"] += 1
+ item["size_bytes"] += record.size_bytes
+ item["protected"] = item["protected"] or record.protected
+ item["cleanup_eligible"] = item["cleanup_eligible"] or record.cleanup_eligible
+
+ existing_paths = {record.path for record in records}
+ missing_references = sorted(
+ path.relative_to(root).as_posix()
+ for path in references
+ if path not in existing_paths and not path.is_dir()
+ )
+ candidates = sorted(
+ (
+ record
+ for record in records
+ if record.cleanup_eligible
+ and record.path not in references
+ and record.modified_at <= cutoff
+ and record.path.name not in IGNORED_FILENAMES
+ ),
+ key=lambda item: (item.modified_at, item.relative_path),
+ )
+ report = {
+ "schema_version": 1,
+ "generated_at": utc_now().isoformat(),
+ "mode": "read-only",
+ "storage_root": str(root),
+ "minimum_age_days": minimum_age_days,
+ "disk_pressure": disk_pressure(root),
+ "lifecycle": {
+ "raw_and_normalized": ["originals", "uploads", "operator-data"],
+ "derived_and_cache": ["derived", "rasters/derived", "previews", "tiles", "masks"],
+ "exports": ["exports"],
+ "ai_models": ["models"],
+ "immutable_evidence": ["release-evidence", "operator-evidence"],
+ },
+ "categories": [
+ {"category": category, **values}
+ for category, values in sorted(categories.items())
+ ],
+ "database": database,
+ "integrity": {
+ "referenced_path_count": len(references),
+ "missing_referenced_path_count": len(missing_references),
+ "missing_referenced_paths": missing_references[:max_candidate_records],
+ "skipped_symlinks": skipped_symlinks[:max_candidate_records],
+ },
+ "cleanup": {
+ "candidate_count": len(candidates),
+ "candidate_bytes": sum(item.size_bytes for item in candidates),
+ "candidate_records_truncated": len(candidates) > max_candidate_records,
+ "candidates": [
+ {
+ "relative_path": item.relative_path,
+ "category": item.category,
+ "size_bytes": item.size_bytes,
+ "modified_at": item.modified_at.isoformat(),
+ }
+ for item in candidates[:max_candidate_records]
+ ],
+ "protected_prefixes": list(PROTECTED_PREFIXES),
+ "eligible_prefixes": list(CLEANUP_PREFIXES),
+ "apply_requires": [
+ "exact confirmation token",
+ "recent checksum-verified database and SHA-256 storage backup",
+ "explicit maximum delete count",
+ ],
+ },
+ "limitations": [
+ "The audit never downloads or refreshes an official source.",
+ "Failed jobs and analysis runs remain provenance and are reported, not deleted.",
+ "Unknown storage categories are protected by default.",
+ "No cleanup is scheduled implicitly.",
+ ],
+ }
+ return report, candidates
+
+
+def parse_args() -> argparse.Namespace:
+ parser = argparse.ArgumentParser(description=__doc__)
+ parser.add_argument("--storage-root", type=Path)
+ parser.add_argument("--minimum-age-days", type=int, default=7)
+ parser.add_argument("--max-candidate-records", type=int, default=500)
+ parser.add_argument("--output", type=Path)
+ parser.add_argument("--fail-on-pressure", choices=("never", "critical", "warning"), default="critical")
+ parser.add_argument("--fail-on-missing-reference", action="store_true")
+ return parser.parse_args()
+
+
+def main() -> int:
+ args = parse_args()
+ storage_root = args.storage_root or Path(get_settings().storage_root)
+ with SessionLocal() as db:
+ report, _ = build_report(
+ storage_root,
+ db,
+ minimum_age_days=args.minimum_age_days,
+ max_candidate_records=args.max_candidate_records,
+ )
+ serialized = json.dumps(report, indent=2, sort_keys=True)
+ print(serialized)
+ if args.output:
+ args.output.parent.mkdir(parents=True, exist_ok=True)
+ temporary = args.output.with_suffix(f"{args.output.suffix}.partial")
+ temporary.write_text(serialized + "\n", encoding="utf-8")
+ temporary.replace(args.output)
+ pressure = report["disk_pressure"]["status"]
+ if args.fail_on_pressure == "critical" and pressure == "critical":
+ return 1
+ if args.fail_on_pressure == "warning" and pressure in {"warning", "critical"}:
+ return 1
+ if args.fail_on_missing_reference and report["integrity"]["missing_referenced_path_count"]:
+ return 1
+ return 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/scripts/cleanup_demo_artifacts.py b/scripts/cleanup_demo_artifacts.py
index 8532e16c..e2a18bb4 100644
--- a/scripts/cleanup_demo_artifacts.py
+++ b/scripts/cleanup_demo_artifacts.py
@@ -18,6 +18,7 @@ def load_backend_script() -> ModuleType:
_impl = load_backend_script()
DEMO_PROJECT_NAME = _impl.DEMO_PROJECT_NAME
+DELETE_CONFIRMATION = _impl.DELETE_CONFIRMATION
is_within_storage_root = _impl.is_within_storage_root
export_created_at = _impl.export_created_at
select_cleanup_candidates = _impl.select_cleanup_candidates
diff --git a/scripts/cleanup_storage_artifacts.py b/scripts/cleanup_storage_artifacts.py
new file mode 100644
index 00000000..166a0ece
--- /dev/null
+++ b/scripts/cleanup_storage_artifacts.py
@@ -0,0 +1,95 @@
+#!/usr/bin/env python3
+"""Dry-run-first cleanup for old unreferenced derived/cache artifacts."""
+
+from __future__ import annotations
+
+import argparse
+import json
+from pathlib import Path
+
+from audit_data_operations import build_report
+from release_backup_guard import require_confirmation, verify_current_backup
+
+from app.core.config import get_settings
+from app.db.session import SessionLocal
+
+
+CONFIRMATION = "DELETE_STORAGE_ARTIFACTS"
+
+
+def parse_args() -> argparse.Namespace:
+ parser = argparse.ArgumentParser(description=__doc__)
+ parser.add_argument("--storage-root", type=Path)
+ parser.add_argument("--minimum-age-days", type=int, default=7)
+ parser.add_argument("--max-delete", type=int, default=25)
+ parser.add_argument("--apply", action="store_true")
+ parser.add_argument("--confirm")
+ parser.add_argument("--backup-dir", type=Path)
+ parser.add_argument("--backup-max-age-hours", type=float, default=24.0)
+ return parser.parse_args()
+
+
+def main() -> int:
+ args = parse_args()
+ if args.max_delete < 0:
+ raise SystemExit("--max-delete must be greater than or equal to zero")
+ storage_root = (args.storage_root or Path(get_settings().storage_root)).resolve()
+ with SessionLocal() as db:
+ report, candidates = build_report(
+ storage_root,
+ db,
+ minimum_age_days=args.minimum_age_days,
+ max_candidate_records=max(args.max_delete, 500),
+ )
+
+ blocked_reason = None
+ backup = None
+ deleted: list[str] = []
+ if args.apply:
+ require_confirmation(args.confirm, CONFIRMATION)
+ if args.backup_dir is None:
+ raise RuntimeError("--backup-dir is required with --apply")
+ backup = verify_current_backup(
+ args.backup_dir,
+ max_age_hours=args.backup_max_age_hours,
+ )
+ if len(candidates) > args.max_delete:
+ blocked_reason = (
+ f"candidate_count {len(candidates)} exceeds --max-delete {args.max_delete}; "
+ "review the dry run and raise the explicit limit"
+ )
+ else:
+ for candidate in candidates:
+ candidate.path.unlink()
+ deleted.append(candidate.relative_path)
+
+ payload = {
+ "schema_version": 1,
+ "mode": "apply" if args.apply else "dry-run",
+ "storage_root": str(storage_root),
+ "minimum_age_days": args.minimum_age_days,
+ "max_delete": args.max_delete,
+ "candidate_count": len(candidates),
+ "candidate_bytes": sum(item.size_bytes for item in candidates),
+ "candidates": [item.relative_path for item in candidates],
+ "deleted_count": len(deleted),
+ "deleted": deleted,
+ "blocked_reason": blocked_reason,
+ "protected_prefixes": report["cleanup"]["protected_prefixes"],
+ "backup": (
+ {
+ "release_id": backup.release_id,
+ "created_at": backup.created_at.isoformat(),
+ "age_hours": round(backup.age_hours, 3),
+ "git_commit": backup.git_commit,
+ }
+ if backup
+ else None
+ ),
+ }
+ print(json.dumps(payload, indent=2, sort_keys=True))
+ return 1 if blocked_reason else 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/scripts/release_backup_guard.py b/scripts/release_backup_guard.py
new file mode 100644
index 00000000..3d98ead0
--- /dev/null
+++ b/scripts/release_backup_guard.py
@@ -0,0 +1,126 @@
+#!/usr/bin/env python3
+"""Verification guard shared by destructive GeoIntel operator commands."""
+
+from __future__ import annotations
+
+import hashlib
+import json
+from dataclasses import dataclass
+from datetime import datetime, timezone
+from pathlib import Path
+
+
+@dataclass(frozen=True)
+class VerifiedBackup:
+ backup_dir: Path
+ release_id: str
+ created_at: datetime
+ age_hours: float
+ git_commit: str
+
+
+def _sha256(path: Path) -> str:
+ digest = hashlib.sha256()
+ with path.open("rb") as handle:
+ for chunk in iter(lambda: handle.read(1024 * 1024), b""):
+ digest.update(chunk)
+ return digest.hexdigest()
+
+
+def _created_at(value: object) -> datetime:
+ if not isinstance(value, str):
+ raise RuntimeError("Backup manifest does not contain created_at")
+ parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
+ if parsed.tzinfo is None:
+ parsed = parsed.replace(tzinfo=timezone.utc)
+ return parsed.astimezone(timezone.utc)
+
+
+def verify_current_backup(
+ backup_dir: str | Path,
+ *,
+ max_age_hours: float = 24.0,
+ now: datetime | None = None,
+) -> VerifiedBackup:
+ """Verify checksums and release metadata without mutating the backup."""
+ if max_age_hours <= 0:
+ raise ValueError("max_age_hours must be greater than zero")
+
+ root = Path(backup_dir).expanduser().resolve()
+ if not root.is_dir():
+ raise RuntimeError(f"Backup directory does not exist: {root}")
+
+ required = {
+ "manifest.json",
+ "CHECKSUMS.sha256",
+ "database.dump",
+ "database.list",
+ "database-metadata.tsv",
+ "table-counts.tsv",
+ "storage-manifest.tsv",
+ }
+ missing = sorted(name for name in required if not (root / name).is_file())
+ if missing:
+ raise RuntimeError(f"Backup is incomplete; missing: {', '.join(missing)}")
+
+ checksum_lines = (root / "CHECKSUMS.sha256").read_text(encoding="utf-8").splitlines()
+ checked: set[str] = set()
+ for line in checksum_lines:
+ if not line.strip():
+ continue
+ try:
+ expected, name = line.split(maxsplit=1)
+ except ValueError as exc:
+ raise RuntimeError("Backup checksum file has an invalid line") from exc
+ name = name.lstrip("*")
+ if "/" in name or "\\" in name or name in {".", ".."}:
+ raise RuntimeError(f"Backup checksum contains an unsafe path: {name}")
+ target = root / name
+ if not target.is_file():
+ raise RuntimeError(f"Backup checksum target is missing: {name}")
+ if _sha256(target) != expected.lower():
+ raise RuntimeError(f"Backup checksum mismatch: {name}")
+ checked.add(name)
+
+ unchecked = sorted((required - {"CHECKSUMS.sha256"}) - checked)
+ if unchecked:
+ raise RuntimeError(f"Backup checksum coverage is incomplete: {', '.join(unchecked)}")
+
+ manifest = json.loads((root / "manifest.json").read_text(encoding="utf-8"))
+ if manifest.get("schema_version") != 1 or manifest.get("read_only_source") is not True:
+ raise RuntimeError("Backup manifest schema or read-only marker is invalid")
+ if manifest.get("database_password_secure") is not True:
+ raise RuntimeError("Backup was made from an insecure database configuration")
+ if manifest.get("inventory_mode") != "sha256" or manifest.get("storage_inventory_requested") is not True:
+ raise RuntimeError("Destructive maintenance requires a SHA-256 storage inventory backup")
+
+ created = _created_at(manifest.get("created_at"))
+ current = now or datetime.now(timezone.utc)
+ if current.tzinfo is None:
+ current = current.replace(tzinfo=timezone.utc)
+ age_hours = (current.astimezone(timezone.utc) - created).total_seconds() / 3600
+ if age_hours < -0.1:
+ raise RuntimeError("Backup timestamp is in the future")
+ if age_hours > max_age_hours:
+ raise RuntimeError(
+ f"Backup is {age_hours:.1f} hours old; maximum allowed age is {max_age_hours:.1f} hours"
+ )
+
+ release_id = manifest.get("release_id")
+ git_commit = manifest.get("git_commit")
+ if not isinstance(release_id, str) or not release_id:
+ raise RuntimeError("Backup release id is missing")
+ if not isinstance(git_commit, str) or len(git_commit) < 7:
+ raise RuntimeError("Backup Git commit is missing")
+ return VerifiedBackup(
+ backup_dir=root,
+ release_id=release_id,
+ created_at=created,
+ age_hours=age_hours,
+ git_commit=git_commit,
+ )
+
+
+def require_confirmation(actual: str | None, expected: str) -> None:
+ if actual != expected:
+ raise RuntimeError(f"Refusing destructive maintenance; pass --confirm {expected}")
diff --git a/scripts/run_rc10_data_operations_audit.sh b/scripts/run_rc10_data_operations_audit.sh
new file mode 100644
index 00000000..eb7cf371
--- /dev/null
+++ b/scripts/run_rc10_data_operations_audit.sh
@@ -0,0 +1,102 @@
+#!/usr/bin/env bash
+set -euo pipefail
+
+ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
+cd "$ROOT"
+
+CONTAINER="${RC10_CONTAINER:-geointel}"
+OUTPUT_DIR="${1:-artifacts/rc10-data-operations}"
+MINIMUM_AGE_DAYS="${RC10_MINIMUM_AGE_DAYS:-7}"
+mkdir -p "$OUTPUT_DIR"
+
+if [ "$(docker inspect -f '{{.State.Running}}' "$CONTAINER" 2>/dev/null || true)" != "true" ]; then
+ echo "Container '$CONTAINER' is not running." >&2
+ exit 2
+fi
+
+read_counts() {
+ docker exec "$CONTAINER" sh -c '
+ db="${POSTGRES_DB:-${GEOINTEL_POSTGRES_DB:-geointel}}"
+ user="${POSTGRES_USER:-${GEOINTEL_POSTGRES_USER:-geointel}}"
+ psql -X -v ON_ERROR_STOP=1 -U "$user" -d "$db" -AtF $'"'"'\t'"'"' \
+ -c "SELECT table_name,
+ CASE table_name
+ WHEN '\''projects'\'' THEN (SELECT count(*) FROM projects)
+ WHEN '\''datasets'\'' THEN (SELECT count(*) FROM datasets)
+ WHEN '\''dataset_versions'\'' THEN (SELECT count(*) FROM dataset_versions)
+ WHEN '\''jobs'\'' THEN (SELECT count(*) FROM jobs)
+ WHEN '\''analysis_runs'\'' THEN (SELECT count(*) FROM analysis_runs)
+ WHEN '\''exports'\'' THEN (SELECT count(*) FROM exports)
+ WHEN '\''detections'\'' THEN (SELECT count(*) FROM detections)
+ WHEN '\''segmentations'\'' THEN (SELECT count(*) FROM segmentations)
+ WHEN '\''quality_checks'\'' THEN (SELECT count(*) FROM quality_checks)
+ END
+ FROM (VALUES
+ ('\''projects'\''), ('\''datasets'\''), ('\''dataset_versions'\''),
+ ('\''jobs'\''), ('\''analysis_runs'\''), ('\''exports'\''),
+ ('\''detections'\''), ('\''segmentations'\''), ('\''quality_checks'\'')
+ ) AS critical(table_name)
+ ORDER BY table_name;"
+ '
+}
+
+read_counts > "$OUTPUT_DIR/table-counts-before.tsv"
+docker exec "$CONTAINER" python /app/scripts/audit_data_operations.py \
+ --minimum-age-days "$MINIMUM_AGE_DAYS" \
+ --fail-on-pressure never \
+ > "$OUTPUT_DIR/data-operations.json"
+docker exec "$CONTAINER" python /app/scripts/cleanup_storage_artifacts.py \
+ --minimum-age-days "$MINIMUM_AGE_DAYS" \
+ --max-delete 25 \
+ > "$OUTPUT_DIR/cleanup-dry-run.json"
+read_counts > "$OUTPUT_DIR/table-counts-after.tsv"
+
+python3 - "$OUTPUT_DIR" <<'PY'
+import json
+import pathlib
+import sys
+
+root = pathlib.Path(sys.argv[1])
+audit = json.loads((root / "data-operations.json").read_text(encoding="utf-8"))
+cleanup = json.loads((root / "cleanup-dry-run.json").read_text(encoding="utf-8"))
+before = (root / "table-counts-before.tsv").read_text(encoding="utf-8")
+after = (root / "table-counts-after.tsv").read_text(encoding="utf-8")
+
+if before != after:
+ raise SystemExit("RC10 read-only audit changed one or more critical table counts")
+if audit.get("mode") != "read-only":
+ raise SystemExit("Data operations audit did not report read-only mode")
+if cleanup.get("mode") != "dry-run" or cleanup.get("deleted_count") != 0:
+ raise SystemExit("Storage cleanup audit was not a zero-delete dry run")
+if "release-evidence" not in audit.get("cleanup", {}).get("protected_prefixes", []):
+ raise SystemExit("Release evidence is not protected")
+families = audit.get("database", {}).get("source_families", {})
+if set(families) != {"national", "regional", "maritime"}:
+ raise SystemExit("National/regional/maritime source-family report is incomplete")
+if audit.get("integrity", {}).get("missing_referenced_path_count", 0):
+ raise SystemExit("Persisted storage references are missing")
+
+manifest = {
+ "schema_version": 1,
+ "status": "passed",
+ "disk_pressure": audit.get("disk_pressure"),
+ "storage_category_count": len(audit.get("categories", [])),
+ "cleanup_candidate_count": cleanup.get("candidate_count", 0),
+ "cleanup_candidate_bytes": cleanup.get("candidate_bytes", 0),
+ "critical_table_counts_unchanged": True,
+ "source_family_counts": {
+ family: len(items) for family, items in families.items()
+ },
+ "missing_referenced_path_count": 0,
+}
+(root / "manifest.json").write_text(
+ json.dumps(manifest, indent=2, sort_keys=True) + "\n",
+ encoding="utf-8",
+)
+print(
+ "RC10 data operations audit passed: "
+ f"pressure={manifest['disk_pressure']['status']}, "
+ f"candidates={manifest['cleanup_candidate_count']}, "
+ f"evidence={root / 'manifest.json'}"
+)
+PY
diff --git a/scripts/run_readiness_check.sh b/scripts/run_readiness_check.sh
index b9d80969..842b656a 100755
--- a/scripts/run_readiness_check.sh
+++ b/scripts/run_readiness_check.sh
@@ -119,6 +119,9 @@ ${PYTHON_BIN} -m py_compile scripts/validate_detection_false_negative_review_dec
${PYTHON_BIN} -m py_compile scripts/activate_promoted_yolo_candidate.py
${PYTHON_BIN} -m py_compile scripts/cleanup_demo_artifacts.py
${PYTHON_BIN} -m py_compile scripts/archive_technical_projects.py
+${PYTHON_BIN} -m py_compile scripts/release_backup_guard.py
+${PYTHON_BIN} -m py_compile scripts/audit_data_operations.py
+${PYTHON_BIN} -m py_compile scripts/cleanup_storage_artifacts.py
${PYTHON_BIN} -m py_compile backend/scripts/cleanup_demo_artifacts.py
${PYTHON_BIN} -m compileall backend/app
(cd backend && ${PYTHON_BIN} -m pytest -W error::DeprecationWarning)
@@ -162,4 +165,5 @@ bash -n scripts/verify_demo_cleanup_dry_run.sh
bash -n scripts/capture_workbench_screenshots.sh
bash -n scripts/run_rc8_release_journeys.sh
bash -n scripts/run_rc9_ux_audit.sh
+bash -n scripts/run_rc10_data_operations_audit.sh
echo "== Run readiness check passed =="