report what an area selection actually measured
Four ways a selection produced a confident number about a different area than the operator drew: Flood hazard divided the inundated cells by every cell in the drawn rectangle, including cells the VMM raster does not model at all. A selection reaching past the modelled extent therefore reported a diluted risk share, turning missing data into an implied absence of risk. Terrain, bathymetry and thematic raster already divided by valid cells; flood hazard was the outlier. It now reports the three populations separately, states model coverage next to the drawn area, and returns a null fraction rather than a zero when nothing was modelled. geometry_mask selects a cell when its centre falls inside the geometry, so a rectangle smaller than one cell — or one landing between four centres — selected nothing and the analysis returned zeros indistinguishable on screen from "we looked and there is nothing here". On a 100 m population raster a 40 m rectangle over a city block reported no inhabitants. Selection now falls back to the touched cells and says that it did, since the answer then covers more ground than was requested. rasterio.mask applies the same centre rule when cropping, so that call is widened too; the cells that count are still decided by the centre rule wherever it selects anything. The object count treated any feature touching the selection as whole, while intersection_area clipped it — two headline numbers on one panel describing different populations. The count stays whole-feature, which is what "objecten" means to an operator, but now reports how many the edge cuts and is marked an estimate when it does. The area_weighted_sum branch reuses that same count instead of issuing its own near-identical query. Partitioned selection de-duplicated the count on source_feature_id but returned the raw rows, so a building on a municipal boundary was counted once and drawn twice. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,103 @@
|
||||
"""Flood risk must be a share of what was modelled, not of what was drawn.
|
||||
|
||||
The share and fraction divided the inundated cells by every cell whose centre
|
||||
fell inside the selection, including cells where the VMM raster holds nodata
|
||||
because the area lies outside the modelled extent. An operator drawing a
|
||||
rectangle that reaches past the model coverage read "3% at risk" where the
|
||||
honest answer is "of the 40% we have a model for, 7.5% is at risk, and for the
|
||||
rest there is no model at all".
|
||||
|
||||
Terrain, bathymetry and thematic raster analysis already divide by valid cells
|
||||
and report a coverage ratio; this brings flood hazard in line.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
np = pytest.importorskip("numpy")
|
||||
|
||||
from app.services.flood_hazard_analysis_service import FloodHazardCellStatistics
|
||||
|
||||
|
||||
NODATA = -9999.0
|
||||
|
||||
|
||||
def _stats(values, selected) -> FloodHazardCellStatistics:
|
||||
return FloodHazardCellStatistics.from_cells(
|
||||
np.asarray(values, dtype="float64"),
|
||||
np.asarray(selected, dtype=bool),
|
||||
nodata=NODATA,
|
||||
)
|
||||
|
||||
|
||||
def test_share_ignores_cells_the_model_does_not_cover() -> None:
|
||||
# Ten selected cells: four modelled (one of them wet), six nodata.
|
||||
values = [1.5, 0.0, 0.0, 0.0] + [NODATA] * 6
|
||||
selected = [True] * 10
|
||||
|
||||
stats = _stats(values, selected)
|
||||
|
||||
assert stats.selected_cell_count == 10
|
||||
assert stats.valid_cell_count == 4
|
||||
assert stats.no_data_cell_count == 6
|
||||
assert stats.inundated_cell_count == 1
|
||||
# 1 of 4 modelled cells, not 1 of 10 drawn cells.
|
||||
assert stats.inundated_fraction == pytest.approx(0.25)
|
||||
assert stats.data_coverage_ratio == pytest.approx(0.4)
|
||||
|
||||
|
||||
def test_cells_outside_the_drawn_selection_are_not_counted() -> None:
|
||||
values = [1.5, 1.5, 0.0, 0.0]
|
||||
selected = [True, False, True, False]
|
||||
|
||||
stats = _stats(values, selected)
|
||||
|
||||
assert stats.selected_cell_count == 2
|
||||
assert stats.valid_cell_count == 2
|
||||
assert stats.inundated_cell_count == 1
|
||||
assert stats.inundated_fraction == pytest.approx(0.5)
|
||||
|
||||
|
||||
def test_a_selection_without_any_model_data_reports_zero_coverage() -> None:
|
||||
stats = _stats([NODATA] * 4, [True] * 4)
|
||||
|
||||
assert stats.valid_cell_count == 0
|
||||
assert stats.no_data_cell_count == 4
|
||||
assert stats.data_coverage_ratio == 0.0
|
||||
# No model, so no risk figure may be invented.
|
||||
assert stats.inundated_fraction is None
|
||||
|
||||
|
||||
def test_nan_is_treated_as_missing_model_data() -> None:
|
||||
stats = _stats([float("nan"), 2.0], [True, True])
|
||||
|
||||
assert stats.valid_cell_count == 1
|
||||
assert stats.no_data_cell_count == 1
|
||||
assert stats.inundated_cell_count == 1
|
||||
|
||||
|
||||
def test_negative_depths_are_data_but_not_inundation() -> None:
|
||||
"""A modelled zero or negative depth means dry, not unknown."""
|
||||
|
||||
stats = _stats([0.0, 0.0, 3.0], [True, True, True])
|
||||
|
||||
assert stats.valid_cell_count == 3
|
||||
assert stats.inundated_cell_count == 1
|
||||
assert stats.inundated_fraction == pytest.approx(1 / 3)
|
||||
|
||||
|
||||
def test_depth_statistics_use_only_inundated_cells() -> None:
|
||||
stats = _stats([0.0, 2.0, 4.0, NODATA], [True] * 4)
|
||||
|
||||
assert stats.depth_values.tolist() == [2.0, 4.0]
|
||||
assert stats.depth_values.mean() == pytest.approx(3.0)
|
||||
|
||||
|
||||
def test_areas_are_derived_from_the_matching_cell_populations() -> None:
|
||||
stats = _stats([1.0, 1.0, 0.0, NODATA], [True] * 4)
|
||||
|
||||
# 100 m2 cells: 2 inundated, 3 modelled, 4 drawn.
|
||||
assert stats.inundated_area_ha(100.0) == pytest.approx(2 * 100.0 / 10_000.0)
|
||||
assert stats.analysed_area_ha(100.0) == pytest.approx(3 * 100.0 / 10_000.0)
|
||||
assert stats.selected_area_ha(100.0) == pytest.approx(4 * 100.0 / 10_000.0)
|
||||
@@ -0,0 +1,61 @@
|
||||
"""A rectangle across a municipal boundary must not return the same object twice.
|
||||
|
||||
Partitioned selection de-duplicated ``total_feature_count`` on
|
||||
``source_feature_id`` but returned the raw rows. A feature present in two
|
||||
municipal partitions was therefore drawn twice on the map and counted once in
|
||||
the headline, so the number on the panel disagreed with the geometry beside it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from uuid import uuid4
|
||||
|
||||
from app.services.vector_feature_service import VectorFeatureService
|
||||
|
||||
|
||||
class _Row:
|
||||
def __init__(self, source_feature_id, row_id=None, dataset_id=None):
|
||||
self.source_feature_id = source_feature_id
|
||||
self.id = row_id or uuid4()
|
||||
self.dataset_id = dataset_id or uuid4()
|
||||
|
||||
|
||||
def _ids(rows):
|
||||
return [row.source_feature_id or str(row.id) for row in rows]
|
||||
|
||||
|
||||
def test_a_feature_in_two_partitions_is_returned_once() -> None:
|
||||
shared = "grb-building-42"
|
||||
rows = [_Row(shared), _Row("grb-building-7"), _Row(shared)]
|
||||
|
||||
kept = VectorFeatureService.deduplicate_rows(rows)
|
||||
|
||||
assert _ids(kept) == [shared, "grb-building-7"]
|
||||
|
||||
|
||||
def test_the_first_occurrence_wins_so_the_result_is_stable() -> None:
|
||||
first = _Row("dup")
|
||||
second = _Row("dup")
|
||||
|
||||
assert VectorFeatureService.deduplicate_rows([first, second])[0] is first
|
||||
assert VectorFeatureService.deduplicate_rows([second, first])[0] is second
|
||||
|
||||
|
||||
def test_rows_without_a_source_id_fall_back_to_their_own_identity() -> None:
|
||||
"""Two distinct rows with no source id are two distinct features."""
|
||||
|
||||
rows = [_Row(None), _Row(None)]
|
||||
|
||||
assert len(VectorFeatureService.deduplicate_rows(rows)) == 2
|
||||
|
||||
|
||||
def test_an_empty_source_id_is_not_treated_as_a_shared_identity() -> None:
|
||||
rows = [_Row(""), _Row("")]
|
||||
|
||||
assert len(VectorFeatureService.deduplicate_rows(rows)) == 2
|
||||
|
||||
|
||||
def test_deduplication_leaves_a_clean_population_untouched() -> None:
|
||||
rows = [_Row("a"), _Row("b"), _Row("c")]
|
||||
|
||||
assert VectorFeatureService.deduplicate_rows(rows) == rows
|
||||
@@ -0,0 +1,84 @@
|
||||
"""A selection smaller than a raster cell must not silently read as zero.
|
||||
|
||||
``geometry_mask`` selects a cell when its *centre* falls inside the geometry.
|
||||
A rectangle smaller than one cell, or one that lands between four centres,
|
||||
therefore selects nothing at all — and the analysis returned zeros, which on
|
||||
screen is indistinguishable from "we looked and there is nothing here". On a
|
||||
100 m population raster a 40 m rectangle over a city block reported no
|
||||
inhabitants.
|
||||
|
||||
Selection now falls back to every touched cell and says that it did, so the
|
||||
value is readable as "at least one whole cell", not as an empty area.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
np = pytest.importorskip("numpy")
|
||||
rasterio = pytest.importorskip("rasterio")
|
||||
|
||||
from rasterio.transform import from_origin
|
||||
from shapely.geometry import box
|
||||
|
||||
from app.services.raster_cell_selection import select_cells
|
||||
|
||||
|
||||
# 100 m cells, origin at the top-left corner of a 3x3 grid.
|
||||
TRANSFORM = from_origin(200_000, 210_000, 100.0, 100.0)
|
||||
SHAPE = (3, 3)
|
||||
|
||||
|
||||
def test_a_normal_selection_uses_cell_centres() -> None:
|
||||
selection = select_cells(box(200_000, 209_700, 200_300, 210_000), out_shape=SHAPE, transform=TRANSFORM)
|
||||
|
||||
assert selection.mask.sum() == 9
|
||||
assert selection.mode == "cell_centre"
|
||||
assert selection.expanded_to_touched_cells is False
|
||||
assert selection.warning is None
|
||||
|
||||
|
||||
def test_a_rectangle_smaller_than_one_cell_still_returns_that_cell() -> None:
|
||||
selection = select_cells(box(200_010, 209_960, 200_050, 209_990), out_shape=SHAPE, transform=TRANSFORM)
|
||||
|
||||
assert selection.mask.sum() == 1
|
||||
assert selection.mode == "all_touched"
|
||||
assert selection.expanded_to_touched_cells is True
|
||||
assert "cel" in selection.warning
|
||||
|
||||
|
||||
def test_a_rectangle_between_four_cell_centres_returns_all_four() -> None:
|
||||
selection = select_cells(box(200_080, 209_880, 200_120, 209_920), out_shape=SHAPE, transform=TRANSFORM)
|
||||
|
||||
assert selection.mask.sum() == 4
|
||||
assert selection.expanded_to_touched_cells is True
|
||||
|
||||
|
||||
def test_a_selection_entirely_off_the_raster_selects_nothing() -> None:
|
||||
"""Falling back must not invent coverage where the geometry does not reach."""
|
||||
|
||||
selection = select_cells(box(300_000, 300_000, 300_100, 300_100), out_shape=SHAPE, transform=TRANSFORM)
|
||||
|
||||
assert selection.mask.sum() == 0
|
||||
assert selection.expanded_to_touched_cells is False
|
||||
assert selection.mode == "cell_centre"
|
||||
|
||||
|
||||
def test_the_warning_states_how_much_larger_the_analysed_area_is() -> None:
|
||||
selection = select_cells(
|
||||
box(200_010, 209_960, 200_050, 209_990),
|
||||
out_shape=SHAPE,
|
||||
transform=TRANSFORM,
|
||||
cell_area_m2=100.0 * 100.0,
|
||||
)
|
||||
|
||||
# One 100x100 m cell was analysed for a 40x30 m request.
|
||||
assert "1 rastercel" in selection.warning
|
||||
assert "1.0 ha" in selection.warning
|
||||
|
||||
|
||||
def test_the_mask_shape_always_matches_the_raster_window() -> None:
|
||||
selection = select_cells(box(200_010, 209_960, 200_050, 209_990), out_shape=SHAPE, transform=TRANSFORM)
|
||||
|
||||
assert selection.mask.shape == SHAPE
|
||||
assert selection.mask.dtype == np.bool_
|
||||
@@ -0,0 +1,117 @@
|
||||
"""A selection finer than the source raster must answer, not return zero.
|
||||
|
||||
End-to-end counterpart to ``test_raster_cell_selection``: the analysis reads a
|
||||
real GeoTIFF, so it proves the fallback survives the clip/mask path the service
|
||||
actually uses rather than only the helper in isolation.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from uuid import uuid4
|
||||
|
||||
import pytest
|
||||
|
||||
np = pytest.importorskip("numpy")
|
||||
rasterio = pytest.importorskip("rasterio")
|
||||
|
||||
from pyproj import Transformer
|
||||
from rasterio.transform import from_origin
|
||||
|
||||
from app.core.config import Settings
|
||||
from app.models import Dataset
|
||||
from app.schemas.flood_hazard import FloodHazardSelectionRequest
|
||||
from app.services.flood_hazard_acquisition_service import FloodHazardAcquisitionService
|
||||
from app.services.flood_hazard_analysis_service import FloodHazardAnalysisService
|
||||
|
||||
|
||||
TO_4326 = Transformer.from_crs("EPSG:31370", "EPSG:4326", always_xy=True)
|
||||
PRODUCT_KEY = "fluviaal_current_t100"
|
||||
|
||||
|
||||
class FakeSession:
|
||||
def __init__(self, objects):
|
||||
self.objects = objects
|
||||
|
||||
def get(self, model, item_id):
|
||||
return self.objects.get((model, item_id))
|
||||
|
||||
|
||||
def _write_raster(path: Path, *, resolution: float, depth: float) -> None:
|
||||
values = np.full((4, 4), depth, dtype="float32")
|
||||
with rasterio.open(
|
||||
path,
|
||||
"w",
|
||||
driver="GTiff",
|
||||
width=4,
|
||||
height=4,
|
||||
count=1,
|
||||
dtype="float32",
|
||||
crs="EPSG:31370",
|
||||
transform=from_origin(200_000, 210_000, resolution, resolution),
|
||||
nodata=-9999.0,
|
||||
) as output:
|
||||
output.write(values, 1)
|
||||
|
||||
|
||||
def _dataset(project_id, dataset_id, path: Path) -> Dataset:
|
||||
return Dataset(
|
||||
id=dataset_id,
|
||||
project_id=project_id,
|
||||
name="vmm-flood.tif",
|
||||
dataset_type="raster",
|
||||
source="vmm",
|
||||
source_name=FloodHazardAcquisitionService.PROVIDER,
|
||||
status="ready",
|
||||
storage_path=str(path),
|
||||
source_metadata={"product_key": PRODUCT_KEY, "normalized_value_unit": "m"},
|
||||
)
|
||||
|
||||
|
||||
def _bbox_for(min_x: float, min_y: float, max_x: float, max_y: float) -> dict:
|
||||
left, bottom = TO_4326.transform(min_x, min_y)
|
||||
right, top = TO_4326.transform(max_x, max_y)
|
||||
return {"min_x": left, "min_y": bottom, "max_x": right, "max_y": top, "crs": "EPSG:4326"}
|
||||
|
||||
|
||||
def _analyze(tmp_path: Path, bbox: dict, *, resolution: float = 100.0) -> dict:
|
||||
project_id = uuid4()
|
||||
dataset_id = uuid4()
|
||||
path = tmp_path / "flood.tif"
|
||||
_write_raster(path, resolution=resolution, depth=2.0)
|
||||
dataset = _dataset(project_id, dataset_id, path)
|
||||
db = FakeSession({(Dataset, dataset_id): dataset})
|
||||
|
||||
return FloodHazardAnalysisService.analyze(
|
||||
db,
|
||||
project_id,
|
||||
dataset_id,
|
||||
FloodHazardSelectionRequest(bbox=bbox),
|
||||
settings=Settings(_env_file=None),
|
||||
)
|
||||
|
||||
|
||||
def test_a_selection_smaller_than_one_cell_reports_the_cell_it_touches(tmp_path: Path) -> None:
|
||||
# A 40 x 30 m rectangle wholly inside one 100 m cell: no cell centre falls
|
||||
# inside it, so the centre rule alone would report an empty selection.
|
||||
result = _analyze(tmp_path, _bbox_for(200_010, 209_960, 200_050, 209_990))
|
||||
|
||||
assert result["inundated_cell_count"] == 1
|
||||
assert result["inundated_fraction"] == pytest.approx(1.0)
|
||||
assert "kleiner dan één rastercel" in result["coverage_warning"]
|
||||
|
||||
|
||||
def test_a_normal_selection_is_unaffected(tmp_path: Path) -> None:
|
||||
result = _analyze(tmp_path, _bbox_for(200_000, 209_700, 200_300, 210_000))
|
||||
|
||||
assert result["inundated_cell_count"] >= 9
|
||||
assert result["coverage_warning"] is None
|
||||
|
||||
|
||||
def test_the_reported_area_matches_the_cells_that_were_analysed(tmp_path: Path) -> None:
|
||||
result = _analyze(tmp_path, _bbox_for(200_010, 209_960, 200_050, 209_990))
|
||||
metrics = {item["metric_key"]: item["metric_value"] for item in result["summary"]["metrics"]}
|
||||
|
||||
# One 100 x 100 m cell, not the 0.12 ha that was drawn.
|
||||
assert metrics["modelled_inundated_area_ha"] == pytest.approx(1.0)
|
||||
assert metrics["selection_area_ha"] == pytest.approx(1.0)
|
||||
@@ -292,13 +292,13 @@ def test_population_area_weighting_is_exact_for_full_features_and_estimated_for_
|
||||
bbox = {"min_x": 5.0, "min_y": 51.1, "max_x": 5.2, "max_y": 51.3, "crs": "EPSG:4326"}
|
||||
|
||||
full = VectorFeatureService.summarize_features_by_bbox(
|
||||
SequenceScalarSession([38_675.0, 0]),
|
||||
SequenceScalarSession([49, 38_675.0]),
|
||||
dataset=dataset,
|
||||
bbox=bbox,
|
||||
total_feature_count=49,
|
||||
)
|
||||
partial = VectorFeatureService.summarize_features_by_bbox(
|
||||
SequenceScalarSession([1_250.5, 2]),
|
||||
SequenceScalarSession([1, 1_250.5]),
|
||||
dataset=dataset,
|
||||
bbox=bbox,
|
||||
total_feature_count=3,
|
||||
|
||||
@@ -24,8 +24,15 @@ class ScalarQuery:
|
||||
|
||||
|
||||
class SequenceScalarSession:
|
||||
def __init__(self, values: list[float]):
|
||||
self.values = iter(values)
|
||||
"""Answers scalar queries in the order the summary issues them.
|
||||
|
||||
The first query is the fully-covered feature count that produces the
|
||||
selection-edge disclosure; ``covered_count`` defaults to the full
|
||||
population, i.e. a selection that cuts nothing.
|
||||
"""
|
||||
|
||||
def __init__(self, values: list[float], covered_count: float | None = None):
|
||||
self.values = iter(([covered_count] if covered_count is not None else []) + values)
|
||||
|
||||
def query(self, *args): # noqa: ANN002, ARG002
|
||||
return ScalarQuery(next(self.values))
|
||||
@@ -53,7 +60,7 @@ def themed_dataset(theme: str, *, method: str = "feature_count") -> Dataset:
|
||||
|
||||
def test_building_selection_promotes_footprint_area_and_retains_object_count() -> None:
|
||||
result = VectorFeatureService.summarize_features_by_bbox(
|
||||
SequenceScalarSession([125_000.0]),
|
||||
SequenceScalarSession([125_000.0], covered_count=40),
|
||||
dataset=themed_dataset("buildings"),
|
||||
bbox=BBOX,
|
||||
total_feature_count=40,
|
||||
@@ -73,7 +80,7 @@ def test_building_selection_promotes_footprint_area_and_retains_object_count() -
|
||||
|
||||
def test_water_selection_reports_surface_length_and_honest_volume_limitation() -> None:
|
||||
result = VectorFeatureService.summarize_features_by_bbox(
|
||||
SequenceScalarSession([52_500.0, 12_750.0]),
|
||||
SequenceScalarSession([52_500.0, 12_750.0], covered_count=23),
|
||||
dataset=themed_dataset("water"),
|
||||
bbox=BBOX,
|
||||
total_feature_count=23,
|
||||
@@ -95,7 +102,7 @@ def test_population_keeps_configured_metric_and_adds_sector_count() -> None:
|
||||
{"metric_key": "population", "property": "population_total", "label": "Inwoners", "unit": "inwoners"}
|
||||
)
|
||||
result = VectorFeatureService.summarize_features_by_bbox(
|
||||
SequenceScalarSession([86_458.0]),
|
||||
SequenceScalarSession([86_458.0], covered_count=733),
|
||||
dataset=dataset,
|
||||
bbox=BBOX,
|
||||
total_feature_count=733,
|
||||
@@ -132,7 +139,7 @@ def test_station_measurement_uses_numeric_mean_without_area_extrapolation() -> N
|
||||
)
|
||||
|
||||
result = VectorFeatureService.summarize_features_by_bbox(
|
||||
SequenceScalarSession([30.455]),
|
||||
SequenceScalarSession([30.455], covered_count=1),
|
||||
dataset=dataset,
|
||||
bbox=BBOX,
|
||||
total_feature_count=1,
|
||||
@@ -152,7 +159,7 @@ def test_regional_historical_polygons_do_not_emit_irrelevant_line_metrics() -> N
|
||||
dataset.provenance_metadata = {"operator_tool": "provision_regional_historical_landuse.py"}
|
||||
|
||||
result = VectorFeatureService.summarize_features_by_bbox(
|
||||
SequenceScalarSession([52_500.0]),
|
||||
SequenceScalarSession([52_500.0], covered_count=23),
|
||||
dataset=dataset,
|
||||
bbox=BBOX,
|
||||
total_feature_count=23,
|
||||
|
||||
@@ -287,8 +287,20 @@ def test_flood_hazard_analysis_reports_scenario_metrics_without_claiming_waterbo
|
||||
metrics = {item["metric_key"]: item for item in result["summary"]["metrics"]}
|
||||
|
||||
assert result["inundated_cell_count"] == 200
|
||||
assert result["inundated_fraction"] == pytest.approx(0.5)
|
||||
# The fixture models the left half and marks the right half nodata. All of
|
||||
# the modelled half is wet, and the model covers half the selection. The
|
||||
# earlier 0.5 conflated "not modelled" with "modelled dry" and reported
|
||||
# half the risk that the model actually describes.
|
||||
assert result["valid_cell_count"] == 200
|
||||
assert result["no_data_cell_count"] == 200
|
||||
assert result["inundated_fraction"] == pytest.approx(1.0)
|
||||
assert result["data_coverage_ratio"] == pytest.approx(0.5)
|
||||
assert "50.0%" in result["coverage_warning"]
|
||||
assert metrics["modelled_inundated_share_pct"]["metric_value"] == pytest.approx(100.0)
|
||||
assert metrics["model_coverage_pct"]["metric_value"] == pytest.approx(50.0)
|
||||
assert metrics["modelled_inundated_area_ha"]["metric_value"] == pytest.approx(0.5)
|
||||
assert metrics["modelled_area_ha"]["metric_value"] == pytest.approx(0.5)
|
||||
assert metrics["selection_area_ha"]["metric_value"] == pytest.approx(1.0)
|
||||
assert metrics["modelled_depth_mean_m"]["metric_value"] == pytest.approx(1.0)
|
||||
assert metrics["modelled_max_depth_area_integral_m3"]["metric_value"] == pytest.approx(5000.0)
|
||||
assert "concurrent_flood_volume_m3" in result["unsupported_metrics"]
|
||||
|
||||
@@ -0,0 +1,161 @@
|
||||
"""The object count and the area metric must describe the same selection.
|
||||
|
||||
``intersection_area`` clips a feature to the drawn rectangle, but the object
|
||||
count treated any feature that merely touches the rectangle as wholly inside.
|
||||
For a rectangle across a built-up area that overstates the count at every
|
||||
edge, and the two headline numbers on the same panel then describe different
|
||||
populations: "1.000 gebouwen" next to the clipped area of rather fewer.
|
||||
|
||||
The count now reports how many features lie entirely inside and how many are
|
||||
cut by the selection edge, and is marked as an estimate when any are.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from app.services.vector_feature_service import VectorFeatureService
|
||||
|
||||
|
||||
def test_a_count_without_partial_features_is_exact() -> None:
|
||||
disclosure = VectorFeatureService.count_disclosure(
|
||||
total_feature_count=120,
|
||||
fully_covered_feature_count=120,
|
||||
)
|
||||
|
||||
assert disclosure["partially_covered_feature_count"] == 0
|
||||
assert disclosure["is_estimate"] is False
|
||||
assert disclosure["warning"] is None
|
||||
|
||||
|
||||
def test_features_cut_by_the_selection_edge_are_reported() -> None:
|
||||
disclosure = VectorFeatureService.count_disclosure(
|
||||
total_feature_count=120,
|
||||
fully_covered_feature_count=98,
|
||||
)
|
||||
|
||||
assert disclosure["partially_covered_feature_count"] == 22
|
||||
assert disclosure["is_estimate"] is True
|
||||
assert "22" in disclosure["warning"]
|
||||
assert "rand" in disclosure["warning"]
|
||||
|
||||
|
||||
def test_a_selection_of_only_partial_features_is_still_coherent() -> None:
|
||||
disclosure = VectorFeatureService.count_disclosure(
|
||||
total_feature_count=3,
|
||||
fully_covered_feature_count=0,
|
||||
)
|
||||
|
||||
assert disclosure["partially_covered_feature_count"] == 3
|
||||
assert disclosure["is_estimate"] is True
|
||||
|
||||
|
||||
def test_an_empty_selection_makes_no_claim() -> None:
|
||||
disclosure = VectorFeatureService.count_disclosure(
|
||||
total_feature_count=0,
|
||||
fully_covered_feature_count=0,
|
||||
)
|
||||
|
||||
assert disclosure["partially_covered_feature_count"] == 0
|
||||
assert disclosure["is_estimate"] is False
|
||||
assert disclosure["warning"] is None
|
||||
|
||||
|
||||
def test_a_preclipped_full_area_selection_has_no_edge_effect() -> None:
|
||||
"""Selecting the whole work area cuts nothing; the count is exact."""
|
||||
|
||||
disclosure = VectorFeatureService.count_disclosure(
|
||||
total_feature_count=500,
|
||||
fully_covered_feature_count=None,
|
||||
)
|
||||
|
||||
assert disclosure["partially_covered_feature_count"] is None
|
||||
assert disclosure["is_estimate"] is False
|
||||
assert disclosure["warning"] is None
|
||||
|
||||
|
||||
def test_an_inconsistent_covered_count_never_produces_a_negative() -> None:
|
||||
disclosure = VectorFeatureService.count_disclosure(
|
||||
total_feature_count=10,
|
||||
fully_covered_feature_count=14,
|
||||
)
|
||||
|
||||
assert disclosure["partially_covered_feature_count"] == 0
|
||||
assert disclosure["is_estimate"] is False
|
||||
|
||||
|
||||
class _ScalarQuery:
|
||||
def __init__(self, value):
|
||||
self.value = value
|
||||
|
||||
def filter(self, *args): # noqa: ANN002, ARG002
|
||||
return self
|
||||
|
||||
def scalar(self):
|
||||
return self.value
|
||||
|
||||
|
||||
class _SequenceSession:
|
||||
"""Answers the summary's scalar queries in order: covered count, then metrics."""
|
||||
|
||||
def __init__(self, values):
|
||||
self.values = iter(values)
|
||||
|
||||
def query(self, *args): # noqa: ANN002, ARG002
|
||||
return _ScalarQuery(next(self.values))
|
||||
|
||||
|
||||
def _buildings_dataset():
|
||||
from uuid import uuid4
|
||||
|
||||
from app.models import Dataset
|
||||
|
||||
return Dataset(
|
||||
id=uuid4(),
|
||||
project_id=uuid4(),
|
||||
name="grb-buildings.geojson",
|
||||
dataset_type="vector",
|
||||
dataset_role="reference",
|
||||
source_name="grb",
|
||||
reference_layer_name="buildings",
|
||||
source_metadata={"theme": "buildings"},
|
||||
)
|
||||
|
||||
|
||||
BBOX = {"min_x": 5.0, "min_y": 51.1, "max_x": 5.2, "max_y": 51.3, "crs": "EPSG:4326"}
|
||||
|
||||
|
||||
def test_summary_reports_the_edge_cut_next_to_the_object_count() -> None:
|
||||
summary = VectorFeatureService.summarize_features_by_bbox(
|
||||
_SequenceSession([88, 125_000.0]),
|
||||
dataset=_buildings_dataset(),
|
||||
bbox=BBOX,
|
||||
total_feature_count=100,
|
||||
)
|
||||
|
||||
assert summary["feature_count"] == 100
|
||||
assert summary["fully_covered_feature_count"] == 88
|
||||
assert summary["partially_covered_feature_count"] == 12
|
||||
assert "12 van de 100" in summary["selection_edge_warning"]
|
||||
|
||||
count_metric = next(
|
||||
item for item in summary["metrics"] if item["aggregation_method"] == "feature_count"
|
||||
)
|
||||
assert count_metric["is_estimate"] is True
|
||||
assert "doorgesneden" in count_metric["warning"]
|
||||
|
||||
# The clipped area metric is exact and must not inherit the count's caveat.
|
||||
area_metric = next(
|
||||
item for item in summary["metrics"] if item["aggregation_method"] == "intersection_area"
|
||||
)
|
||||
assert area_metric["is_estimate"] is False
|
||||
|
||||
|
||||
def test_summary_stays_exact_when_the_selection_cuts_nothing() -> None:
|
||||
summary = VectorFeatureService.summarize_features_by_bbox(
|
||||
_SequenceSession([100, 125_000.0]),
|
||||
dataset=_buildings_dataset(),
|
||||
bbox=BBOX,
|
||||
total_feature_count=100,
|
||||
)
|
||||
|
||||
assert summary["partially_covered_feature_count"] == 0
|
||||
assert summary["selection_edge_warning"] is None
|
||||
Reference in New Issue
Block a user