diff --git a/CHANGELOG.md b/CHANGELOG.md index d5237190..57da0c10 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,12 @@ # Changelog +## Sprint 172 CPU AI image build hardening (2026-07-12) + +- Reordered the Unraid all-in-one Docker build so backend source changes reuse the Python/GIS/AI dependency layer. +- Pinned the opt-in CPU runtime to PyTorch 2.13.0 and torchvision 0.28.0 from the official CPU wheel index, avoiding unused CUDA runtime packages while preserving the currently validated framework versions. +- Kept AI dependencies opt-in and model weights local-only; no API, migration, model activation or inference contract changed. + ## Sprint 171 Positive AOI expansion and split safety (2026-07-12) - Added four explicit, real-reference Kempen training AOIs: Olen, Lille, Oud-Turnhout and Kasterlee center. diff --git a/backend/tests/test_docker_runtime_config.py b/backend/tests/test_docker_runtime_config.py index 7d7c15f0..878def58 100644 --- a/backend/tests/test_docker_runtime_config.py +++ b/backend/tests/test_docker_runtime_config.py @@ -265,6 +265,21 @@ def test_backend_docker_context_contains_gis_import_smoke_script() -> None: assert (ROOT / "backend" / "scripts" / "gis_import_smoke.py").exists() +def test_all_in_one_dockerfile_caches_dependencies_and_uses_cpu_torch_for_ai_runtime() -> None: + dockerfile = (ROOT / "deploy" / "unraid" / "Dockerfile.all-in-one").read_text(encoding="utf-8") + + metadata_copy_index = dockerfile.index("COPY backend/pyproject.toml backend/README.md /app/") + dependency_install_index = dockerfile.index('RUN /usr/bin/python3.11 -m venv /opt/geointel/venv \\') + backend_copy_index = dockerfile.index("COPY backend/ /app/") + smoke_index = dockerfile.index("RUN python scripts/gis_import_smoke.py") + + assert metadata_copy_index < dependency_install_index < backend_copy_index < smoke_index + assert "GEOINTEL_TORCH_INDEX_URL=https://download.pytorch.org/whl/cpu" in dockerfile + assert "GEOINTEL_TORCH_VERSION=2.13.0" in dockerfile + assert "GEOINTEL_TORCHVISION_VERSION=0.28.0" in dockerfile + assert '--index-url "$GEOINTEL_TORCH_INDEX_URL"' in dockerfile + + def test_unraid_deploy_passes_ai_build_arg_and_yolo_runtime_env() -> None: deploy_ps1 = (ROOT / "scripts" / "deploy_tower.ps1").read_text(encoding="utf-8") deploy_sh = (ROOT / "scripts" / "deploy_tower.sh").read_text(encoding="utf-8") diff --git a/deploy/unraid/Dockerfile.all-in-one b/deploy/unraid/Dockerfile.all-in-one index 6249bd35..2ff176e7 100644 --- a/deploy/unraid/Dockerfile.all-in-one +++ b/deploy/unraid/Dockerfile.all-in-one @@ -9,6 +9,9 @@ RUN npm run build FROM postgres:16-bookworm AS runtime ARG GEOINTEL_INSTALL_AI=false +ARG GEOINTEL_TORCH_INDEX_URL=https://download.pytorch.org/whl/cpu +ARG GEOINTEL_TORCH_VERSION=2.13.0 +ARG GEOINTEL_TORCHVISION_VERSION=0.28.0 ENV GEOINTEL_ENV=production \ GEOINTEL_API_PREFIX=/api/v1 \ @@ -44,15 +47,25 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ WORKDIR /app -COPY backend/ /app/ -COPY fixtures/ /app/fixtures/ +COPY backend/pyproject.toml backend/README.md /app/ +COPY backend/app/__init__.py /app/app/__init__.py RUN /usr/bin/python3.11 -m venv /opt/geointel/venv \ && pip install --no-cache-dir --upgrade pip setuptools \ + && if [ "$GEOINTEL_INSTALL_AI" = "true" ]; then \ + pip install --no-cache-dir \ + --index-url "$GEOINTEL_TORCH_INDEX_URL" \ + "torch==$GEOINTEL_TORCH_VERSION" \ + "torchvision==$GEOINTEL_TORCHVISION_VERSION"; \ + fi \ && extras=".[gis]" \ && if [ "$GEOINTEL_INSTALL_AI" = "true" ]; then extras=".[gis,ai]"; fi \ - && pip install --no-cache-dir "$extras" \ - && python scripts/gis_import_smoke.py \ + && pip install --no-cache-dir "$extras" + +COPY backend/ /app/ +COPY fixtures/ /app/fixtures/ + +RUN python scripts/gis_import_smoke.py \ && python scripts/yolo_preflight.py --json >/tmp/geointel-yolo-preflight.json \ && rm -f /etc/nginx/sites-enabled/default \ && mkdir -p /app/storage /run/nginx /var/log/nginx diff --git a/deploy/unraid/README.md b/deploy/unraid/README.md index 12fdae3a..66c675af 100644 --- a/deploy/unraid/README.md +++ b/deploy/unraid/README.md @@ -96,6 +96,11 @@ GIS-only image. Set `GEOINTEL_INSTALL_AI=true`, mount models through when you have a local model file. The AI-enabled image installs PyTorch/Ultralytics plus the native OpenCV runtime libraries needed for Ultralytics imports; it still never downloads model weights. +The documented CPU runtime installs pinned `torch==2.13.0` and +`torchvision==0.28.0` from `https://download.pytorch.org/whl/cpu`, avoiding the +unused CUDA runtime wheels included by the general Linux package index. The +Dockerfile copies dependency metadata before backend source, so normal code-only +redeploys can reuse the expensive dependency layer. `YOLO_CONFIG_DIR` defaults to `/app/storage/ultralytics`, a writable persistent path, so Ultralytics settings do not fall back to root user config directories. diff --git a/docs/CODEX_EXECUTION_LOG.md b/docs/CODEX_EXECUTION_LOG.md index 9682de8f..22ec06b2 100644 --- a/docs/CODEX_EXECUTION_LOG.md +++ b/docs/CODEX_EXECUTION_LOG.md @@ -7015,3 +7015,21 @@ Open: ## Next recommended pass - Finish the inactive candidate, run fixed-threshold positive evidence and split pure-empty/sparse-context background matrices, and preserve the current production default unless the promotion report passes every gate. +# Sprint 172 - CPU AI image build hardening + +## What changed + +- Reordered `deploy/unraid/Dockerfile.all-in-one` so `pyproject.toml` and minimal package metadata are installed before the complete backend source is copied. +- Code-only backend changes can now reuse the expensive GIS/AI dependency layer; dependency metadata changes still invalidate it. +- The opt-in CPU AI build now installs the same validated PyTorch `2.13.0` / torchvision `0.28.0` versions from the official CPU wheel index before installing the `ai` extra. +- Kept the full GIS and YOLO import/preflight smoke after the complete backend source copy. + +## Validation so far + +- RED: the new Docker ordering/CPU-wheel regression test failed against the old Dockerfile. +- GREEN: `python -m pytest backend/tests/test_docker_runtime_config.py -q`: 26 passed. +- `python -m pip index versions` confirmed `torch 2.13.0+cpu` and `torchvision 0.28.0+cpu` are available from the configured CPU index for the workstation platform. + +## Remaining validation + +- Build the AI-enabled all-in-one image on Tower after the current inactive model training run finishes, verify Torch reports a CPU build and rerun live migration/browser smokes before replacing the runtime. diff --git a/docs/DEPENDENCY_POLICY.md b/docs/DEPENDENCY_POLICY.md index 3bb15a8f..506ba9fa 100644 --- a/docs/DEPENDENCY_POLICY.md +++ b/docs/DEPENDENCY_POLICY.md @@ -43,6 +43,18 @@ AI dependencies remain separate in the `ai` optional dependency group and must not be installed by the default Docker backend image unless an explicit AI image or profile is introduced later. +The opt-in Unraid all-in-one AI build is CPU-oriented because its documented +runtime sets `YOLO_DEVICE=cpu`. It installs the pinned PyTorch/torchvision pair +from PyTorch's CPU wheel index before installing the `ai` extra. This avoids +shipping unused CUDA runtime libraries. The index and versions remain explicit +Docker build arguments so a future, separately validated GPU image can override +them without changing the base dependency group. + +Docker dependency metadata is copied before application source. Backend source +changes therefore reuse the dependency layer while changes to `pyproject.toml` +still invalidate it correctly. Runtime GIS and YOLO import/preflight smokes run +after the complete backend source is copied. + ## Approved AI - PyTorch