Files
geointel/backend/tests/test_sprint205_agricultural_parcel_history.py
T
JensandClaude Opus 5 6572e4ad5f scope frontend contracts to the feature, not to one file
93 test files read a single frontend source and asserted identifiers in it. The
MapWorkspace split showed what that costs: 24 tests went red for a move that
changed no behaviour at all. A contract belongs to the feature — a container,
its hooks, its domain layer — not to whichever file currently holds it.

232 read sites now resolve through read_feature(). The distinction that makes
this safe is direction: a *positive* contract ("this is wired") may widen,
because the identifier must still exist somewhere in the feature; a *negative*
one ("this component performs no transport") is a statement about one file, and
widening it would quietly weaken the check. The 73 single-file reads that
remain are exactly those, and a guard now enforces the rule for new tests.

Verified rather than assumed: of the 732 migrated positive assertions, 644 still
match exactly one module — as specific as before — and the other 86 already
spanned a container and its hook by nature. Two apparent misses are an artefact
of the checking regex reading an escaped newline literally.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-22 22:05:43 +02:00

295 lines
11 KiB
Python

from __future__ import annotations
import importlib.util
import json
import sys
import zipfile
from pathlib import Path
from uuid import uuid4
import pytest
from shapely.geometry import box, shape
from shapely.ops import transform as transform_geometry
from app.models import Dataset
from app.schemas.operations import VectorSelectionSummary
from app.services.vector_feature_service import VectorFeatureService
from tests.frontend_contract import read_map_workspace, read_feature
ROOT = Path(__file__).resolve().parents[2]
BBOX = {"min_x": 5.0, "min_y": 51.1, "max_x": 5.3, "max_y": 51.4, "crs": "EPSG:4326"}
def load_operator():
script_path = ROOT / "scripts" / "provision_agricultural_parcel_history.py"
spec = importlib.util.spec_from_file_location("agricultural_parcel_history_operator", script_path)
assert spec is not None
assert spec.loader is not None
module = importlib.util.module_from_spec(spec)
sys.modules[spec.name] = module
spec.loader.exec_module(module)
return module
class FakeRow:
def __init__(self, geometry, values: dict): # noqa: ANN001
self.geometry = geometry
self.values = values
def __getitem__(self, key): # noqa: ANN001
return self.values[key]
class FakeFrame:
def __init__(self, rows: list[FakeRow], columns: list[str]):
self.rows = rows
self.columns = columns
def iterrows(self):
return iter(enumerate(self.rows))
class ScalarQuery:
def __init__(self, value: float):
self.value = value
def filter(self, *args): # noqa: ANN002, ARG002
return self
def scalar(self):
return self.value
class SequenceScalarSession:
def __init__(self, values: list[float]):
self.values = iter(values)
def query(self, *args): # noqa: ANN002, ARG002
return ScalarQuery(next(self.values))
class ApiResponse:
ok = True
status_code = 200
text = ""
def __init__(self, data: dict):
self.data = data
def json(self):
return {"data": self.data}
class PaginatedApiSession:
def __init__(self):
self.offsets: list[int] = []
def get(self, url, *, params, timeout): # noqa: ANN001, ARG002
self.offsets.append(params["offset"])
if params["offset"] == 0:
return ApiResponse({"items": [{"id": index} for index in range(200)], "total": 201})
return ApiResponse({"items": [{"id": 200}], "total": 201})
def test_only_definitive_2008_through_2025_archives_are_allowed() -> None:
module = load_operator()
assert module.SUPPORTED_YEARS == tuple(range(2008, 2026))
assert 2026 not in module.ARCHIVE_URLS
assert module.ARCHIVE_URLS[2025].endswith("agpa_2025_2026-05-13_public.zip")
assert all(url.startswith("https://www.landbouwvlaanderen.be/bestanden/gis/agpa_") for url in module.ARCHIVE_URLS.values())
with pytest.raises(ValueError, match="Supported definitive years"):
module.parse_years("2025,2026")
def test_canonical_api_collection_reader_respects_200_item_limit_and_paginates() -> None:
module = load_operator()
session = PaginatedApiSession()
items = module.api_items(session, "http://geointel/api/v1/projects/project-id/datasets", 30)
assert len(items) == 201
assert session.offsets == [0, 200]
def test_archive_requires_exactly_one_safe_geopackage(tmp_path: Path) -> None:
module = load_operator()
valid = tmp_path / "valid.zip"
with zipfile.ZipFile(valid, "w") as archive:
archive.writestr("agpa_2025.gpkg", b"source")
archive.writestr("metadata.pdf", b"metadata")
assert module.archive_geopackage_member(valid) == "agpa_2025.gpkg"
unsafe = tmp_path / "unsafe.zip"
with zipfile.ZipFile(unsafe, "w") as archive:
archive.writestr("nested/agpa_2025.gpkg", b"source")
with pytest.raises(RuntimeError, match="unsafe"):
module.archive_geopackage_member(unsafe)
ambiguous = tmp_path / "ambiguous.zip"
with zipfile.ZipFile(ambiguous, "w") as archive:
archive.writestr("one.gpkg", b"one")
archive.writestr("two.gpkg", b"two")
with pytest.raises(RuntimeError, match="exactly one"):
module.archive_geopackage_member(ambiguous)
def test_crop_code_list_preserves_year_specific_titles_and_reports_conflicts() -> None:
module = load_operator()
result = module.build_crop_code_list(
[
{"maincrop_code": "201", "maincrop_title": "Mais", "maincropgroup_title": "Mais"},
{"maincrop_code": "201", "maincrop_title": "Korrelmais", "maincropgroup_title": "Mais"},
{"maincrop_code": "901", "maincrop_title": "Grasland", "maincropgroup_title": "Grasland"},
],
year=2025,
)
assert result["year"] == 2025
assert len(result["crop_entries"]) == 3
assert result["code_title_conflicts"] == {"201": ["Korrelmais", "Mais"]}
assert "maincropgroup_title" in result["historical_comparison_rule"]
assert module.normalized_group_title("Maïs") == "maize"
assert module.normalized_group_title("Granen, zaden en peulvruchten") == "grains_seeds_legumes"
assert module.normalized_group_title("Groenten, kruiden en sierplanten") == "horticulture"
def test_persisted_first_import_group_keys_remain_query_compatible() -> None:
assert VectorFeatureService._expanded_selection_filter_values(
"main_crop_group_key",
["grains_seeds_legumes", "horticulture"],
) == [
"grains_seeds_legumes",
"granen,_zaden_en_peulvruchten",
"horticulture",
"groenten,_kruiden_en_sierplanten",
]
def test_features_are_exactly_clipped_in_lambert72_and_keep_source_fields() -> None:
module = load_operator()
boundary_wgs84 = box(5.10, 51.20, 5.11, 51.21)
boundary_lambert72 = transform_geometry(module.TO_LAMBERT72.transform, boundary_wgs84)
source_geometry = transform_geometry(module.TO_LAMBERT72.transform, box(5.095, 51.195, 5.105, 51.205))
values = {
"agpakey": "2025-42",
"parcelnumber": "42",
"area_ha": 1.25,
"maincrop_code": "201",
"maincrop_title": "Korrelmais",
"maincropgroup_title": "Maïs",
"geometry": source_geometry,
}
frame = FakeFrame([FakeRow(source_geometry, values)], list(values))
features, summary = module.normalize_frame(
frame,
year=2025,
boundary_lambert72=boundary_lambert72,
max_features=10,
)
assert summary["feature_count"] == 1
assert summary["clipped_feature_count"] == 1
feature = features[0]
assert feature["id"] == "alz:2025:2025-42"
assert shape(feature["geometry"]).difference(boundary_wgs84.buffer(1e-7)).area < 1e-12
properties = feature["properties"]
assert properties["maincrop_title"] == "Korrelmais"
assert properties["main_crop_group_key"] == "maize"
assert properties["geometry_was_clipped"] is True
assert properties["historical_parcel_identity_stable"] is False
assert properties["clipped_area_ha"] < properties["source_geometry_area_ha"]
def test_duplicate_annual_source_identity_fails_closed() -> None:
module = load_operator()
boundary = box(100_000, 200_000, 101_000, 201_000)
values = {"agpakey": "same", "maincropgroup_title": "Grasland", "geometry": boundary}
frame = FakeFrame([FakeRow(boundary, values), FakeRow(boundary, values)], list(values))
with pytest.raises(RuntimeError, match="duplicate agpakey"):
module.normalize_frame(frame, year=2025, boundary_lambert72=boundary, max_features=10)
def test_agriculture_summary_returns_grouped_hectares_without_parcel_lineage_claim() -> None:
module = load_operator()
dataset = Dataset(
id=uuid4(),
project_id=uuid4(),
name="agricultural_use_parcels_2025.geojson",
dataset_type="vector",
dataset_role="reference",
source_name=module.SOURCE_NAME,
reference_layer_name="agriculture",
source_metadata={
"theme": "agriculture",
"semantic_metrics": False,
"selection_aggregation": {
"metric_key": "declared_agricultural_use_area",
"method": "intersection_area",
"label": "Aangegeven gebruiksoppervlakte",
"unit": "ha",
"geometry_dimension": 2,
},
"selection_metrics": module.selection_metrics(),
},
)
square_metres = [150_000.0, 40_000.0, 30_000.0, 20_000.0, 10_000.0, 5_000.0, 4_000.0, 3_000.0, 2_000.0, 1_000.0, 500.0, 250.0]
result = VectorFeatureService.summarize_features_by_bbox(
SequenceScalarSession(square_metres),
dataset=dataset,
bbox=BBOX,
total_feature_count=321,
full_dataset_area=True,
)
assert result["primary_metric_key"] == "declared_agricultural_use_area"
assert result["metric_value"] == 15.0
metrics = {item["metric_key"]: item for item in result["metrics"]}
assert metrics["grassland_area"]["metric_value"] == 4.0
assert metrics["maize_area"]["metric_value"] == 3.0
assert metrics["agricultural_water_area"]["metric_value"] == 0.025
assert metrics["feature_count"]["metric_value"] == 321
assert "perceelidentiteiten" in metrics["grassland_area"]["warning"]
VectorSelectionSummary(**result)
def test_operator_uses_canonical_upload_and_is_packaged_for_runtime() -> None:
operator = (ROOT / "scripts" / "provision_agricultural_parcel_history.py").read_text(encoding="utf-8")
service = (ROOT / "backend" / "app" / "services" / "vector_feature_service.py").read_text(encoding="utf-8")
dockerfile = (ROOT / "deploy" / "unraid" / "Dockerfile.all-in-one").read_text(encoding="utf-8")
readiness = (ROOT / "scripts" / "run_readiness_check.sh").read_text(encoding="utf-8")
map_workspace = read_map_workspace()
source_catalog = read_feature("datasets")
dataset_display = (ROOT / "frontend" / "src" / "lib" / "datasetDisplay.ts").read_text(encoding="utf-8")
assert "/datasets/upload" in operator
assert "VectorFeature" not in operator
assert "INSERT INTO vector_features" not in operator
assert "geo.api.vlaanderen.be/Landbgebrperc" not in operator
assert '"provision_agricultural_parcel_history.py"' in service
assert "COPY scripts/provision_agricultural_parcel_history.py" in dockerfile
assert "py_compile scripts/provision_agricultural_parcel_history.py" in readiness
assert "id: 'agriculture'" in map_workspace
assert "Landbouwgebruikspercelen" in source_catalog
assert "agriculture: 'Landbouwgebruikspercelen'" in dataset_display
assert "agentschap_landbouw_zeevisserij_agricultural_parcels: 'Agentschap Landbouw en Zeevisserij'" in dataset_display
assert "const source = first ? getDatasetDisplayName(first) : 'Tijdreeks'" in map_workspace
def test_upload_contract_is_annual_definitive_and_scope_specific(tmp_path: Path) -> None:
module = load_operator()
scope = module.GEOGRAPHIC_SCOPES["mol"]
assert module.series_key(scope) == "alz:agricultural-use-parcels:mol"
metrics = module.selection_metrics()
assert {item["metric_key"] for item in metrics} >= {"grassland_area", "maize_area", "agricultural_water_area"}
assert all(item["method"] == "intersection_area" for item in metrics)
assert all(item["filter_property"] == "main_crop_group_key" for item in metrics)
paths = module.artifact_paths(tmp_path, scope.key, 2025)
assert paths["archive"].name == "agpa_2025_2026-05-13_public.zip"
assert paths["artifact"].name == "agricultural_use_parcels_2025_mol.geojson"