Files
geointel/backend/tests/test_sprint194_regional_timeseries.py
T
NuklearRabbit edb762219c
Managed validation / full (pull_request) Canceled after 0s
GeoIntel release gates / Compile, test, contracts and builds (pull_request) Canceled after 0s
GeoIntel release gates / Python and npm vulnerability policy (pull_request) Canceled after 0s
GeoIntel release gates / GIS image, SBOM and container scan (pull_request) Canceled after 0s
GeoIntel release gates / Python and npm vulnerability policy (push) Failing after 22s
GeoIntel release gates / Compile, test, contracts and builds (push) Waiting to run
GeoIntel release gates / GIS image, SBOM and container scan (push) Failing after 45s
fix(ci): restore release gates and contract baseline (fixes #4)
2026-08-29 02:00:23 +02:00

362 lines
13 KiB
Python

from __future__ import annotations
import argparse
import importlib.util
import io
import json
from pathlib import Path
import sys
import zipfile
from types import SimpleNamespace
from uuid import uuid4
import numpy as np
import rasterio
from rasterio.transform import from_origin
from app.models import Dataset
from app.services.vector_feature_service import VectorFeatureService
from tests.frontend_contract import read_feature
ROOT = Path(__file__).resolve().parents[2]
SCRIPTS = ROOT / "scripts"
if str(SCRIPTS) not in sys.path:
sys.path.insert(0, str(SCRIPTS))
def load_script(name: str):
path = SCRIPTS / name
module_name = f"test_{path.stem}"
spec = importlib.util.spec_from_file_location(module_name, path)
assert spec is not None
assert spec.loader is not None
module = importlib.util.module_from_spec(spec)
sys.modules[module_name] = module
spec.loader.exec_module(module)
return module
def population_archive() -> bytes:
text = "\n".join(
(
"CD_REFNIS|CD_SECTOR|TOTAL|TX_DESCR_SECTOR_NL|TX_DESCR_NL",
"13025|13025A00-|120|Mol centrum|Mol",
"13008|13008A00-|240|Geel centrum|Geel",
"11002|11002A00-|360|Antwerpen centrum|Antwerpen",
)
)
buffer = io.BytesIO()
with zipfile.ZipFile(buffer, "w") as archive:
archive.writestr("population.csv", text)
return buffer.getvalue()
def test_population_operator_filters_to_the_approved_scope() -> None:
module = load_script("provision_mol_population_history.py")
regional = module.GEOGRAPHIC_SCOPES["kempen-transport-region"]
mol = module.GEOGRAPHIC_SCOPES["mol"]
belgium = module.GEOGRAPHIC_SCOPES["belgium"]
regional_rows = module.population_rows(population_archive(), regional)
mol_rows = module.population_rows(population_archive(), mol)
national_rows = module.population_rows(population_archive(), belgium)
assert set(regional_rows) == {"13025A00-", "13008A00-"}
assert regional_rows["13008A00-"]["municipality"] == "Geel"
assert regional_rows["13008A00-"]["nis_code"] == "13008"
assert set(mol_rows) == {"13025A00-"}
assert set(national_rows) == {"13025A00-", "13008A00-", "11002A00-"}
assert national_rows["11002A00-"]["municipality"] == "Antwerpen"
assert belgium.all_municipalities is True
assert module.series_key(regional) == "statbel:population-statistical-sector:kempen-transport-region"
assert module.series_key(mol) == "statbel:population-statistical-sector:mol"
assert module.series_key(belgium) == "statbel:population-statistical-sector:belgium"
def test_population_operator_resolves_the_persisted_scope_boundary(tmp_path: Path) -> None:
module = load_script("provision_mol_population_history.py")
scope = module.GEOGRAPHIC_SCOPES["kempen-transport-region"]
scope_dir = tmp_path / scope.key
scope_dir.mkdir(parents=True)
boundary = scope_dir / "boundary.geojson"
boundary.write_text('{"type":"FeatureCollection","features":[]}', encoding="utf-8")
manifest = scope_dir / "kempen_transport_region_scope_manifest.json"
manifest.write_text(
json.dumps(
{
"status": "complete",
"scope_key": scope.key,
"boundary_filename": boundary.name,
}
),
encoding="utf-8",
)
args = argparse.Namespace(boundary_path=None, scope_output_root=tmp_path)
assert module.resolve_boundary_path(args, scope) == boundary
def test_population_operator_resolves_checksum_verified_belgium_boundary(tmp_path: Path) -> None:
module = load_script("provision_mol_population_history.py")
scope = module.GEOGRAPHIC_SCOPES["belgium"]
scope_dir = tmp_path / "belgium-north-sea"
scope_dir.mkdir(parents=True)
boundary = scope_dir / "belgium_land_boundary.geojson"
boundary.write_text(
json.dumps(
{
"type": "FeatureCollection",
"features": [
{
"type": "Feature",
"geometry": {
"type": "Polygon",
"coordinates": [[[2.5, 49.5], [6.4, 49.5], [6.4, 51.5], [2.5, 51.5], [2.5, 49.5]]],
},
"properties": {},
}
],
}
),
encoding="utf-8",
)
(scope_dir / "manifest.json").write_text(
json.dumps(
{
"scope": "belgium-and-belgian-north-sea",
"artifacts": {
"belgium_land_boundary": {
"sha256": module.sha256_path(boundary),
}
},
}
),
encoding="utf-8",
)
args = argparse.Namespace(boundary_path=None, scope_output_root=tmp_path)
assert module.resolve_boundary_path(args, scope) == boundary
def test_regional_coordinator_builds_explicit_population_and_landuse_commands(tmp_path: Path) -> None:
module = load_script("provision_regional_timeseries.py")
scope = module.GEOGRAPHIC_SCOPES["kempen-transport-region"]
args = argparse.Namespace(
output_root=tmp_path / "time-series",
scope_output_root=tmp_path / "scopes",
fetch_only=False,
force=False,
skip_population=False,
skip_landuse=False,
base_url="http://backend:8000",
population_years="2021,2025",
landuse_years="2013,2025",
historical_years="1778,1873,1969",
historical_themes="buildings,water,roads",
request_timeout=300,
import_timeout=3600,
max_landuse_features=500000,
max_historical_features=500000,
skip_historical=False,
)
members_path = tmp_path / "municipalities.geojson"
commands = dict(module.build_operator_commands(args, scope, tmp_path / "boundary.geojson", members_path))
assert set(commands) == {"population", "forest", "historical_landuse"}
assert commands["population"][0] == sys.executable
assert "--scope" in commands["population"]
assert "kempen-transport-region" in commands["population"]
assert scope.project_name in commands["population"]
assert commands["forest"][0] == sys.executable
assert "--max-features" in commands["forest"]
assert "--partition-boundaries-path" in commands["forest"]
assert str(members_path) in commands["forest"]
assert ",".join(scope.nis_codes) in commands["forest"]
assert "--force" not in commands["population"]
assert "--fetch-only" not in commands["forest"]
assert commands["historical_landuse"][0] == sys.executable
assert "provision_regional_historical_landuse.py" in commands["historical_landuse"][1]
assert "buildings,water,roads" in commands["historical_landuse"]
assert "--scope-output-root" in commands["historical_landuse"]
def test_regional_forest_provenance_does_not_claim_one_municipality() -> None:
module = load_script("provision_official_landuse_timeseries.py")
regional = module.scope_identity("Kempen (28 gemeenten)", "13001,13008,13025")
municipal = module.scope_identity("Mol", "13025")
assert regional == {
"scope_display_name": "Kempen (28 gemeenten)",
"member_nis_codes": ["13001", "13008", "13025"],
"municipality": None,
"nis_code": None,
}
assert municipal["municipality"] == "Mol"
assert municipal["nis_code"] == "13025"
def test_regional_forest_partition_rasters_merge_without_resolution_loss(tmp_path: Path) -> None:
module = load_script("provision_official_landuse_timeseries.py")
left = tmp_path / "left.tif"
right = tmp_path / "right.tif"
profile = {
"driver": "GTiff",
"height": 2,
"width": 2,
"count": 1,
"dtype": "uint8",
"crs": "EPSG:31370",
"transform": from_origin(100000, 200000, 10, 10),
"nodata": 0,
}
with rasterio.open(left, "w", **profile) as target:
target.write(np.full((1, 2, 2), 12, dtype="uint8"))
with rasterio.open(
right,
"w",
**{**profile, "transform": from_origin(100020, 200000, 10, 10)},
) as target:
target.write(np.full((1, 2, 2), 17, dtype="uint8"))
destination = tmp_path / "regional.tif"
result = module.merge_partition_rasters([left, right], destination)
assert result["width"] == 4
assert result["height"] == 2
assert result["resolution_metres"] == 10.0
with rasterio.open(destination) as merged:
assert merged.read(1).tolist() == [[12, 12, 17, 17], [12, 12, 17, 17]]
def test_regional_timeseries_operator_is_packaged_and_release_checked() -> None:
dockerfile = (ROOT / "deploy/unraid/Dockerfile.all-in-one").read_text(encoding="utf-8")
readiness = (ROOT / "scripts/run_readiness_check.sh").read_text(encoding="utf-8")
assert "COPY scripts/provision_regional_timeseries.py" in dockerfile
assert "py_compile scripts/provision_regional_timeseries.py" in readiness
def test_end_user_dataset_sources_are_human_readable() -> None:
display = (ROOT / "frontend/src/lib/datasetDisplay.ts").read_text(encoding="utf-8")
workspace = read_feature("map_workspace")
catalog = read_feature("datasets")
status = (ROOT / "frontend/src/components/WorkbenchStatusStrip.tsx").read_text(encoding="utf-8")
detection = read_feature("detection")
exports = read_feature("exports")
assert "department_omgeving_land_use: 'Departement Omgeving'" in display
assert "statbel: 'Statbel'" in display
assert "getDatasetSourceDisplayName(activeThemeDataset)" in workspace
assert "getDatasetSourceDisplayName(resultDataset)" in workspace
assert "Zoek optioneel een gemeente" in workspace
assert "latestDatasetBySeries" in catalog
assert "Historische meetmomenten" in catalog
assert "getDatasetSourceDisplayName(dataset)" in catalog
assert "statusLabel(item.state)" in status
assert "Technische modelevaluatie" in detection
assert "Nog geen recente downloads" in exports
def test_full_area_fast_path_requires_matching_area_and_clipped_operator_provenance() -> None:
project_id = uuid4()
area_id = uuid4()
trusted = Dataset(
id=uuid4(),
project_id=project_id,
area_id=area_id,
name="Regional forest",
dataset_type="vector",
source="operator_official_import",
provenance_metadata={"operator_tool": "provision_official_landuse_timeseries.py"},
)
explicit = Dataset(
id=uuid4(),
project_id=project_id,
area_id=area_id,
name="Clipped vector",
dataset_type="vector",
source="manual",
source_metadata={"geometry_clipped_to_area": True},
)
regional_historical = Dataset(
id=uuid4(),
project_id=project_id,
area_id=area_id,
name="Regional historical buildings",
dataset_type="vector",
source="operator_official_import",
provenance_metadata={"operator_tool": "provision_regional_historical_landuse.py"},
)
untrusted = Dataset(
id=uuid4(),
project_id=project_id,
area_id=area_id,
name="Assigned only",
dataset_type="vector",
source="manual",
)
assert VectorFeatureService.can_use_full_area_fast_path(trusted, area_id) is True
assert VectorFeatureService.can_use_full_area_fast_path(explicit, area_id) is True
assert VectorFeatureService.can_use_full_area_fast_path(regional_historical, area_id) is True
assert VectorFeatureService.can_use_full_area_fast_path(untrusted, area_id) is False
assert VectorFeatureService.can_use_full_area_fast_path(trusted, uuid4()) is False
assert VectorFeatureService.can_use_full_area_fast_path(trusted, None) is False
def test_full_area_summary_uses_exact_stored_values_without_partial_intersection() -> None:
class ScalarQuery:
def __init__(self, value: float) -> None:
self.value = value
def filter(self, *_args):
return self
def scalar(self):
return self.value
class ScalarSession:
def __init__(self, value: float) -> None:
self.value = value
self.query_count = 0
def query(self, *_args):
self.query_count += 1
return ScalarQuery(self.value)
dataset = Dataset(
id=uuid4(),
project_id=uuid4(),
name="Population",
dataset_type="vector",
source="operator_official_import",
source_metadata={
"selection_aggregation": {
"method": "area_weighted_sum",
"property": "population_total",
"label": "Inwoners",
"unit": "inwoners",
"warning_only_when_estimate": True,
"warning": "Partial-sector estimate",
}
},
)
session = ScalarSession(506_473.0)
result = VectorFeatureService.summarize_features_by_bbox(
session,
dataset=dataset,
bbox={"min_x": 4.5, "min_y": 51.0, "max_x": 5.3, "max_y": 51.6, "crs": "EPSG:4326"},
total_feature_count=733,
selection_geometry=SimpleNamespace(),
full_dataset_area=True,
)
assert result["metric_value"] == 506_473.0
assert result["is_estimate"] is False
assert result["warning"] is None
assert session.query_count == 1