report what an area selection actually measured
Four ways a selection produced a confident number about a different area than the operator drew: Flood hazard divided the inundated cells by every cell in the drawn rectangle, including cells the VMM raster does not model at all. A selection reaching past the modelled extent therefore reported a diluted risk share, turning missing data into an implied absence of risk. Terrain, bathymetry and thematic raster already divided by valid cells; flood hazard was the outlier. It now reports the three populations separately, states model coverage next to the drawn area, and returns a null fraction rather than a zero when nothing was modelled. geometry_mask selects a cell when its centre falls inside the geometry, so a rectangle smaller than one cell — or one landing between four centres — selected nothing and the analysis returned zeros indistinguishable on screen from "we looked and there is nothing here". On a 100 m population raster a 40 m rectangle over a city block reported no inhabitants. Selection now falls back to the touched cells and says that it did, since the answer then covers more ground than was requested. rasterio.mask applies the same centre rule when cropping, so that call is widened too; the cells that count are still decided by the centre rule wherever it selects anything. The object count treated any feature touching the selection as whole, while intersection_area clipped it — two headline numbers on one panel describing different populations. The count stays whole-feature, which is what "objecten" means to an operator, but now reports how many the edge cuts and is marked an estimate when it does. The area_weighted_sum branch reuses that same count instead of issuing its own near-identical query. Partitioned selection de-duplicated the count on source_feature_id but returned the raw rows, so a building on a municipal boundary was counted once and drawn twice. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -160,6 +160,10 @@ class BathymetryRasterSelectionResponse(BaseModel):
|
||||
selected_cell_count: int = Field(ge=1)
|
||||
valid_cell_count: int = Field(ge=1)
|
||||
coverage_ratio: float = Field(ge=0, le=1)
|
||||
# Set when the drawn selection is smaller than one source cell and the
|
||||
# analysis was widened to the cells it touches, so the value covers more
|
||||
# ground than was requested.
|
||||
cell_selection_warning: str | None = None
|
||||
resolution_m: float = Field(gt=0)
|
||||
vertical_reference: str
|
||||
survey_period: str
|
||||
|
||||
@@ -90,6 +90,10 @@ class TerrainSelectionResponse(BaseModel):
|
||||
sample_count: int
|
||||
slope_sample_count: int
|
||||
coverage_ratio: float
|
||||
# Set when the drawn selection is smaller than one source cell and the
|
||||
# analysis was widened to the cells it touches, so the value covers more
|
||||
# ground than was requested.
|
||||
cell_selection_warning: str | None = None
|
||||
resolution_m: float
|
||||
vertical_reference: str
|
||||
summary: TerrainSelectionSummary
|
||||
|
||||
@@ -93,9 +93,17 @@ class FloodHazardSelectionResponse(BaseModel):
|
||||
return_period_years: int
|
||||
selection_bbox: VectorSelectionBBox
|
||||
selection_area_id: UUID | None = None
|
||||
# Three populations kept apart: cells drawn, cells the model covers, and
|
||||
# cells with a positive modelled depth. ``inundated_fraction`` is a share
|
||||
# of the modelled cells, and is null when nothing was modelled — absence
|
||||
# of a model is not evidence of zero risk.
|
||||
selected_cell_count: int
|
||||
valid_cell_count: int = 0
|
||||
no_data_cell_count: int = 0
|
||||
data_coverage_ratio: float = 1.0
|
||||
inundated_cell_count: int
|
||||
inundated_fraction: float
|
||||
inundated_fraction: float | None = None
|
||||
coverage_warning: str | None = None
|
||||
resolution_m: float
|
||||
summary: FloodHazardSelectionSummary
|
||||
unsupported_metrics: list[str]
|
||||
|
||||
@@ -236,7 +236,13 @@ class VectorSelectionSummary(BaseModel):
|
||||
metric_unit: str
|
||||
aggregation_method: str
|
||||
primary_metric_key: str | None = None
|
||||
# ``feature_count`` counts whole features that touch the selection, while
|
||||
# area and length metrics clip to it. These fields say how far the two
|
||||
# populations diverge, so the numbers on one panel can be read together.
|
||||
feature_count: int
|
||||
fully_covered_feature_count: int | None = None
|
||||
partially_covered_feature_count: int | None = None
|
||||
selection_edge_warning: str | None = None
|
||||
is_estimate: bool = False
|
||||
warning: str | None = None
|
||||
metrics: list[VectorSelectionMetric] = Field(default_factory=list)
|
||||
|
||||
@@ -113,6 +113,10 @@ class ThematicRasterSelectionResponse(BaseModel):
|
||||
selected_cell_count: int
|
||||
valid_cell_count: int
|
||||
coverage_ratio: float
|
||||
# Set when the drawn selection is smaller than one source cell and the
|
||||
# analysis was widened to the cells it touches, so the value covers more
|
||||
# ground than was requested.
|
||||
cell_selection_warning: str | None = None
|
||||
resolution_m: float
|
||||
observation_year: int
|
||||
summary: ThematicRasterSelectionSummary
|
||||
|
||||
@@ -13,6 +13,7 @@ from shapely.ops import transform as shapely_transform
|
||||
|
||||
from app.core.config import Settings, get_settings
|
||||
from app.core.errors import AppError
|
||||
from app.services.raster_cell_selection import select_cells
|
||||
from app.models import Area, Dataset
|
||||
from app.schemas.bathymetry import (
|
||||
BathymetryRasterMetric,
|
||||
@@ -157,21 +158,27 @@ class BathymetryRasterAnalysisService:
|
||||
},
|
||||
status_code=422,
|
||||
)
|
||||
# ``all_touched`` keeps the values of cells the selection only
|
||||
# clips, so a selection finer than one cell still has data to
|
||||
# read. Which of those cells actually count is decided by
|
||||
# ``select_cells`` below, so the normal result is unchanged.
|
||||
clipped, clipped_transform = mask(
|
||||
source,
|
||||
[mapping(analysis_geometry)],
|
||||
crop=True,
|
||||
filled=False,
|
||||
indexes=[1],
|
||||
all_touched=True,
|
||||
)
|
||||
band = np.ma.asarray(clipped[0], dtype="float64")
|
||||
raw = band.filled(np.nan)
|
||||
selected_cells = geometry_mask(
|
||||
[mapping(analysis_geometry)],
|
||||
cell_selection = select_cells(
|
||||
analysis_geometry,
|
||||
out_shape=band.shape,
|
||||
transform=clipped_transform,
|
||||
invert=True,
|
||||
cell_area_m2=abs(float(source.res[0])) * abs(float(source.res[1])),
|
||||
)
|
||||
selected_cells = cell_selection.mask
|
||||
valid_cells = selected_cells & ~np.ma.getmaskarray(band) & np.isfinite(raw)
|
||||
if source.nodata is not None:
|
||||
valid_cells &= ~np.isclose(raw, float(source.nodata))
|
||||
@@ -268,6 +275,7 @@ class BathymetryRasterAnalysisService:
|
||||
selected_cell_count=selected_cell_count,
|
||||
valid_cell_count=valid_cell_count,
|
||||
coverage_ratio=round(coverage_ratio, 6),
|
||||
cell_selection_warning=cell_selection.warning,
|
||||
resolution_m=round(max(resolution_x, resolution_y), 4),
|
||||
vertical_reference=vertical_unit,
|
||||
survey_period=str(source_metadata.get("survey_period") or "2019-2022"),
|
||||
|
||||
@@ -2,8 +2,10 @@ from __future__ import annotations
|
||||
|
||||
import io
|
||||
import math
|
||||
from dataclasses import dataclass
|
||||
from datetime import UTC, datetime
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
from uuid import UUID
|
||||
|
||||
from geoalchemy2.shape import to_shape
|
||||
@@ -13,6 +15,7 @@ from shapely.ops import transform as shapely_transform
|
||||
|
||||
from app.core.config import Settings, get_settings
|
||||
from app.core.errors import AppError
|
||||
from app.services.raster_cell_selection import select_cells
|
||||
from app.models import Area, Dataset
|
||||
from app.schemas.flood_hazard import (
|
||||
FloodHazardMetric,
|
||||
@@ -25,6 +28,83 @@ from app.services.flood_hazard_acquisition_service import FloodHazardAcquisition
|
||||
from app.services.raster_partition_analysis_service import RasterPartitionAnalysisService
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class FloodHazardCellStatistics:
|
||||
"""Cell populations behind one flood-hazard selection.
|
||||
|
||||
Three populations, deliberately kept apart:
|
||||
|
||||
``selected``
|
||||
every cell whose centre falls inside the drawn selection;
|
||||
``valid``
|
||||
the subset the VMM raster actually models — finite, not nodata;
|
||||
``inundated``
|
||||
the subset of valid cells with a positive modelled depth.
|
||||
|
||||
Risk is a share of what was modelled. Dividing by the selected cells
|
||||
instead silently reports "no data" as "no risk", which for a selection
|
||||
reaching past the modelled extent understates the hazard by whatever
|
||||
fraction of the rectangle the model never covered.
|
||||
"""
|
||||
|
||||
selected_cell_count: int
|
||||
valid_cell_count: int
|
||||
inundated_cell_count: int
|
||||
depth_values: Any
|
||||
|
||||
@property
|
||||
def no_data_cell_count(self) -> int:
|
||||
return max(0, self.selected_cell_count - self.valid_cell_count)
|
||||
|
||||
@property
|
||||
def data_coverage_ratio(self) -> float:
|
||||
if self.selected_cell_count <= 0:
|
||||
return 0.0
|
||||
return self.valid_cell_count / self.selected_cell_count
|
||||
|
||||
@property
|
||||
def inundated_fraction(self) -> float | None:
|
||||
"""``None`` when nothing was modelled: absence of data is not a zero."""
|
||||
|
||||
if self.valid_cell_count <= 0:
|
||||
return None
|
||||
return self.inundated_cell_count / self.valid_cell_count
|
||||
|
||||
def inundated_area_ha(self, cell_area_m2: float) -> float:
|
||||
return self.inundated_cell_count * cell_area_m2 / 10_000.0
|
||||
|
||||
def analysed_area_ha(self, cell_area_m2: float) -> float:
|
||||
"""Area the model actually covers inside the selection."""
|
||||
|
||||
return self.valid_cell_count * cell_area_m2 / 10_000.0
|
||||
|
||||
def selected_area_ha(self, cell_area_m2: float) -> float:
|
||||
"""Area of the selection as rasterised, model coverage aside."""
|
||||
|
||||
return self.selected_cell_count * cell_area_m2 / 10_000.0
|
||||
|
||||
@classmethod
|
||||
def from_cells(cls, values: Any, selected: Any, *, nodata: float | None) -> "FloodHazardCellStatistics":
|
||||
import numpy as np
|
||||
|
||||
raw = np.asarray(values, dtype="float64")
|
||||
selected_mask = np.asarray(selected, dtype=bool)
|
||||
|
||||
has_data = selected_mask & np.isfinite(raw)
|
||||
if nodata is not None:
|
||||
has_data &= ~np.isclose(raw, float(nodata))
|
||||
# A modelled zero or negative depth is data: it says "dry here", which
|
||||
# is a different statement from "not modelled here".
|
||||
inundated = has_data & (raw > 0.0)
|
||||
|
||||
return cls(
|
||||
selected_cell_count=int(selected_mask.sum()),
|
||||
valid_cell_count=int(has_data.sum()),
|
||||
inundated_cell_count=int(inundated.sum()),
|
||||
depth_values=raw[inundated],
|
||||
)
|
||||
|
||||
|
||||
class FloodHazardAnalysisService:
|
||||
UNSUPPORTED_METRICS = [
|
||||
"bathymetry_depth_m",
|
||||
@@ -36,6 +116,77 @@ class FloodHazardAnalysisService:
|
||||
"gemodelleerde maxima op en is geen gelijktijdig opgeslagen watervolume, actuele waterstand of bathymetrie."
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _coverage_metrics(stats: "FloodHazardCellStatistics", cell_area_m2: float, metric) -> list[FloodHazardMetric]:
|
||||
"""Headline metrics, each stating which population it is a share of.
|
||||
|
||||
The analysed area is reported next to the drawn area so an operator can
|
||||
see immediately how much of the rectangle the flood model covers. A
|
||||
selection with no model data reports 0% coverage rather than 0% risk.
|
||||
"""
|
||||
|
||||
metrics = [
|
||||
metric(
|
||||
"modelled_inundated_area_ha",
|
||||
"Gemodelleerd overstroomd oppervlak",
|
||||
stats.inundated_area_ha(cell_area_m2),
|
||||
"ha",
|
||||
"positive_depth_cells_times_cell_area",
|
||||
),
|
||||
metric(
|
||||
"modelled_inundated_share_pct",
|
||||
"Aandeel gemodelleerd gebied met diepte",
|
||||
0.0 if stats.inundated_fraction is None else stats.inundated_fraction * 100.0,
|
||||
"%",
|
||||
"positive_depth_cells_divided_by_modelled_cells",
|
||||
),
|
||||
metric(
|
||||
"modelled_area_ha",
|
||||
"Oppervlak met overstromingsmodel",
|
||||
stats.analysed_area_ha(cell_area_m2),
|
||||
"ha",
|
||||
"modelled_cells_times_cell_area",
|
||||
),
|
||||
metric(
|
||||
"selection_area_ha",
|
||||
"Oppervlak van de selectie",
|
||||
stats.selected_area_ha(cell_area_m2),
|
||||
"ha",
|
||||
"selected_cells_times_cell_area",
|
||||
),
|
||||
metric(
|
||||
"model_coverage_pct",
|
||||
"Deel van de selectie met een model",
|
||||
stats.data_coverage_ratio * 100.0,
|
||||
"%",
|
||||
"modelled_cells_divided_by_selected_cells",
|
||||
),
|
||||
]
|
||||
return metrics
|
||||
|
||||
@staticmethod
|
||||
def _combined_warning(stats: "FloodHazardCellStatistics", cell_selection_warning: str | None) -> str | None:
|
||||
parts = [
|
||||
part
|
||||
for part in (cell_selection_warning, FloodHazardAnalysisService._coverage_warning(stats))
|
||||
if part
|
||||
]
|
||||
return " ".join(parts) if parts else None
|
||||
|
||||
@staticmethod
|
||||
def _coverage_warning(stats: "FloodHazardCellStatistics") -> str | None:
|
||||
if stats.valid_cell_count <= 0:
|
||||
return (
|
||||
"Voor deze selectie bestaat geen VMM-overstromingsmodel. Er is dus geen overstromingsrisico "
|
||||
"gemeten; dit is geen bevestiging dat het risico nul is."
|
||||
)
|
||||
if stats.data_coverage_ratio < 0.999:
|
||||
return (
|
||||
f"Het VMM-model dekt {stats.data_coverage_ratio * 100:.1f}% van deze selectie. Percentages gelden "
|
||||
"voor het gemodelleerde deel, niet voor de volledige selectie."
|
||||
)
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def _load_dataset(db, project_id: UUID, dataset_id: UUID) -> Dataset:
|
||||
dataset = db.get(Dataset, dataset_id)
|
||||
@@ -110,17 +261,32 @@ class FloodHazardAnalysisService:
|
||||
details={"pixel_count": expected_cells, "max_pixels": resolved_settings.flood_hazard_max_pixels},
|
||||
status_code=422,
|
||||
)
|
||||
clipped, clipped_transform = mask(source, [mapping(analysis_geometry)], crop=True, filled=False, indexes=[1])
|
||||
# ``all_touched`` keeps the values of cells the selection only
|
||||
# clips, so a selection finer than one cell still has data to
|
||||
# read. Which of those cells actually count is decided by
|
||||
# ``select_cells`` below, so the normal result is unchanged.
|
||||
clipped, clipped_transform = mask(
|
||||
source,
|
||||
[mapping(analysis_geometry)],
|
||||
crop=True,
|
||||
filled=False,
|
||||
indexes=[1],
|
||||
all_touched=True,
|
||||
)
|
||||
depth = np.ma.asarray(clipped[0], dtype="float64")
|
||||
raw = depth.filled(np.nan)
|
||||
selected_cells = geometry_mask([mapping(analysis_geometry)], out_shape=depth.shape, transform=clipped_transform, invert=True)
|
||||
nodata = source.nodata
|
||||
valid = selected_cells & ~np.ma.getmaskarray(depth) & np.isfinite(raw) & (raw > 0.0)
|
||||
if nodata is not None:
|
||||
valid &= raw != float(nodata)
|
||||
values = raw[valid]
|
||||
selected_cell_count = int(selected_cells.sum())
|
||||
inundated_cell_count = int(values.size)
|
||||
cell_selection = select_cells(
|
||||
analysis_geometry,
|
||||
out_shape=depth.shape,
|
||||
transform=clipped_transform,
|
||||
cell_area_m2=abs(float(source.res[0])) * abs(float(source.res[1])),
|
||||
)
|
||||
selected_cells = cell_selection.mask
|
||||
# A masked cell carries no model value, so fold the mask into
|
||||
# the raw array before the populations are separated.
|
||||
raw = np.where(np.ma.getmaskarray(depth), np.nan, raw)
|
||||
stats = FloodHazardCellStatistics.from_cells(raw, selected_cells, nodata=source.nodata)
|
||||
values = stats.depth_values
|
||||
resolution_x = abs(float(source.res[0]))
|
||||
resolution_y = abs(float(source.res[1]))
|
||||
cell_area_m2 = resolution_x * resolution_y
|
||||
@@ -143,18 +309,8 @@ class FloodHazardAnalysisService:
|
||||
aggregation_method=method,
|
||||
)
|
||||
|
||||
inundated_area_ha = inundated_cell_count * cell_area_m2 / 10_000.0
|
||||
metrics = [
|
||||
metric("modelled_inundated_area_ha", "Gemodelleerd overstroomd oppervlak", inundated_area_ha, "ha", "positive_depth_cells_times_cell_area"),
|
||||
metric(
|
||||
"modelled_inundated_share_pct",
|
||||
"Aandeel selectie met gemodelleerde diepte",
|
||||
inundated_cell_count / max(1, selected_cell_count) * 100.0,
|
||||
"%",
|
||||
"positive_depth_cells_divided_by_selected_cells",
|
||||
),
|
||||
]
|
||||
if inundated_cell_count:
|
||||
metrics = FloodHazardAnalysisService._coverage_metrics(stats, cell_area_m2, metric)
|
||||
if stats.inundated_cell_count:
|
||||
metrics.extend(
|
||||
[
|
||||
metric("modelled_depth_mean_m", "Gemiddelde gemodelleerde maximumdiepte", values.mean(), "m", "mean_positive_depth_cells"),
|
||||
@@ -181,9 +337,14 @@ class FloodHazardAnalysisService:
|
||||
return_period_years=product.return_period_years,
|
||||
selection_bbox=payload.bbox,
|
||||
selection_area_id=payload.area_id,
|
||||
selected_cell_count=selected_cell_count,
|
||||
inundated_cell_count=inundated_cell_count,
|
||||
inundated_fraction=round(inundated_cell_count / max(1, selected_cell_count), 6),
|
||||
selected_cell_count=stats.selected_cell_count,
|
||||
valid_cell_count=stats.valid_cell_count,
|
||||
no_data_cell_count=stats.no_data_cell_count,
|
||||
data_coverage_ratio=round(stats.data_coverage_ratio, 6),
|
||||
inundated_cell_count=stats.inundated_cell_count,
|
||||
inundated_fraction=(
|
||||
None if stats.inundated_fraction is None else round(stats.inundated_fraction, 6)
|
||||
),
|
||||
resolution_m=round(max(resolution_x, resolution_y), 4),
|
||||
summary=FloodHazardSelectionSummary(
|
||||
metric_label=primary.metric_label,
|
||||
@@ -193,6 +354,7 @@ class FloodHazardAnalysisService:
|
||||
primary_metric_key=primary.metric_key,
|
||||
metrics=metrics,
|
||||
),
|
||||
coverage_warning=FloodHazardAnalysisService._combined_warning(stats, cell_selection.warning),
|
||||
unsupported_metrics=FloodHazardAnalysisService.UNSUPPORTED_METRICS,
|
||||
limitation_message=FloodHazardAnalysisService.LIMITATION,
|
||||
generated_at=datetime.now(UTC).isoformat(),
|
||||
@@ -236,16 +398,12 @@ class FloodHazardAnalysisService:
|
||||
status_code=503,
|
||||
) from exc
|
||||
|
||||
raw = partition.values
|
||||
valid = (
|
||||
partition.selected_cells
|
||||
& np.isfinite(raw)
|
||||
& (raw != FloodHazardAcquisitionService.NODATA)
|
||||
& (raw > 0.0)
|
||||
stats = FloodHazardCellStatistics.from_cells(
|
||||
partition.values,
|
||||
partition.selected_cells,
|
||||
nodata=FloodHazardAcquisitionService.NODATA,
|
||||
)
|
||||
values = raw[valid]
|
||||
selected_cell_count = int(partition.selected_cells.sum())
|
||||
inundated_cell_count = int(values.size)
|
||||
values = stats.depth_values
|
||||
cell_area_m2 = partition.resolution_x * partition.resolution_y
|
||||
|
||||
def metric(key: str, label: str, value: float, unit: str, method: str) -> FloodHazardMetric:
|
||||
@@ -257,24 +415,8 @@ class FloodHazardAnalysisService:
|
||||
aggregation_method=method,
|
||||
)
|
||||
|
||||
inundated_area_ha = inundated_cell_count * cell_area_m2 / 10_000.0
|
||||
metrics = [
|
||||
metric(
|
||||
"modelled_inundated_area_ha",
|
||||
"Gemodelleerd overstroomd oppervlak",
|
||||
inundated_area_ha,
|
||||
"ha",
|
||||
"positive_depth_cells_times_cell_area",
|
||||
),
|
||||
metric(
|
||||
"modelled_inundated_share_pct",
|
||||
"Aandeel selectie met gemodelleerde diepte",
|
||||
inundated_cell_count / max(1, selected_cell_count) * 100.0,
|
||||
"%",
|
||||
"positive_depth_cells_divided_by_selected_cells",
|
||||
),
|
||||
]
|
||||
if inundated_cell_count:
|
||||
metrics = FloodHazardAnalysisService._coverage_metrics(stats, cell_area_m2, metric)
|
||||
if stats.inundated_cell_count:
|
||||
metrics.extend(
|
||||
[
|
||||
metric("modelled_depth_mean_m", "Gemiddelde gemodelleerde maximumdiepte", values.mean(), "m", "mean_positive_depth_cells"),
|
||||
@@ -302,9 +444,14 @@ class FloodHazardAnalysisService:
|
||||
return_period_years=product.return_period_years,
|
||||
selection_bbox=payload.bbox,
|
||||
selection_area_id=payload.area_id,
|
||||
selected_cell_count=selected_cell_count,
|
||||
inundated_cell_count=inundated_cell_count,
|
||||
inundated_fraction=round(inundated_cell_count / max(1, selected_cell_count), 6),
|
||||
selected_cell_count=stats.selected_cell_count,
|
||||
valid_cell_count=stats.valid_cell_count,
|
||||
no_data_cell_count=stats.no_data_cell_count,
|
||||
data_coverage_ratio=round(stats.data_coverage_ratio, 6),
|
||||
inundated_cell_count=stats.inundated_cell_count,
|
||||
inundated_fraction=(
|
||||
None if stats.inundated_fraction is None else round(stats.inundated_fraction, 6)
|
||||
),
|
||||
resolution_m=round(max(partition.resolution_x, partition.resolution_y), 4),
|
||||
summary=FloodHazardSelectionSummary(
|
||||
metric_label=primary.metric_label,
|
||||
@@ -314,6 +461,7 @@ class FloodHazardAnalysisService:
|
||||
primary_metric_key=primary.metric_key,
|
||||
metrics=metrics,
|
||||
),
|
||||
coverage_warning=FloodHazardAnalysisService._combined_warning(stats, partition.cell_selection_warning),
|
||||
unsupported_metrics=FloodHazardAnalysisService.UNSUPPORTED_METRICS,
|
||||
limitation_message=(
|
||||
f"{FloodHazardAnalysisService.LIMITATION} De selectie werd exact berekend over "
|
||||
|
||||
@@ -0,0 +1,85 @@
|
||||
"""Choosing which raster cells a drawn selection covers.
|
||||
|
||||
``rasterio.features.geometry_mask`` selects a cell when the cell *centre* falls
|
||||
inside the geometry. That is the right rule for a selection spanning many
|
||||
cells, and the wrong one for a small selection: a rectangle smaller than a cell,
|
||||
or one landing between four centres, selects nothing. The analysis then reports
|
||||
zeros, which on screen reads as "we measured this area and found nothing"
|
||||
rather than "this selection is finer than the source raster".
|
||||
|
||||
Falling back to every touched cell keeps a small selection answerable, at the
|
||||
cost of analysing more ground than was drawn. That trade is only honest if it
|
||||
is stated, so the fallback is reported alongside the result.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RasterCellSelection:
|
||||
mask: Any
|
||||
mode: str
|
||||
expanded_to_touched_cells: bool
|
||||
warning: str | None
|
||||
|
||||
|
||||
def select_cells(
|
||||
geometry,
|
||||
*,
|
||||
out_shape: tuple[int, int],
|
||||
transform,
|
||||
cell_area_m2: float | None = None,
|
||||
) -> RasterCellSelection:
|
||||
"""Mask the cells a selection covers, widening only when it covers none.
|
||||
|
||||
``geometry`` must already be in the raster's own CRS.
|
||||
"""
|
||||
|
||||
from rasterio.features import geometry_mask
|
||||
from shapely.geometry import mapping
|
||||
|
||||
shapes = [mapping(geometry)]
|
||||
mask = geometry_mask(shapes, out_shape=out_shape, transform=transform, invert=True)
|
||||
if mask.any():
|
||||
return RasterCellSelection(
|
||||
mask=mask,
|
||||
mode="cell_centre",
|
||||
expanded_to_touched_cells=False,
|
||||
warning=None,
|
||||
)
|
||||
|
||||
touched = geometry_mask(
|
||||
shapes,
|
||||
out_shape=out_shape,
|
||||
transform=transform,
|
||||
invert=True,
|
||||
all_touched=True,
|
||||
)
|
||||
if not touched.any():
|
||||
# The selection does not reach the raster at all. Widening the rule
|
||||
# must never manufacture coverage that is genuinely absent.
|
||||
return RasterCellSelection(
|
||||
mask=mask,
|
||||
mode="cell_centre",
|
||||
expanded_to_touched_cells=False,
|
||||
warning=None,
|
||||
)
|
||||
|
||||
cell_count = int(touched.sum())
|
||||
if cell_area_m2:
|
||||
analysed_ha = cell_count * float(cell_area_m2) / 10_000.0
|
||||
detail = f"{cell_count} rastercel{'' if cell_count == 1 else 'len'} ({analysed_ha:.1f} ha)"
|
||||
else:
|
||||
detail = f"{cell_count} rastercel{'' if cell_count == 1 else 'len'}"
|
||||
return RasterCellSelection(
|
||||
mask=touched,
|
||||
mode="all_touched",
|
||||
expanded_to_touched_cells=True,
|
||||
warning=(
|
||||
f"De selectie is kleiner dan één rastercel van deze bron. Het resultaat geldt voor {detail} "
|
||||
"die de selectie raken, dus voor een groter gebied dan getekend."
|
||||
),
|
||||
)
|
||||
@@ -12,6 +12,7 @@ from shapely.geometry import mapping
|
||||
from shapely.ops import transform as shapely_transform
|
||||
|
||||
from app.core.errors import AppError
|
||||
from app.services.raster_cell_selection import select_cells
|
||||
from app.models import Dataset
|
||||
|
||||
|
||||
@@ -22,6 +23,9 @@ class RasterPartitionSelection:
|
||||
selected_cells: Any
|
||||
resolution_x: float
|
||||
resolution_y: float
|
||||
# Set when the selection is finer than one source cell and the analysis was
|
||||
# widened to the touched cells, so it covers more ground than was drawn.
|
||||
cell_selection_warning: str | None = None
|
||||
|
||||
|
||||
class RasterPartitionAnalysisService:
|
||||
@@ -185,18 +189,20 @@ class RasterPartitionAnalysisService:
|
||||
dtype="float32",
|
||||
)
|
||||
values = np.asarray(mosaic[0], dtype="float64")
|
||||
selected_cells = geometry_mask(
|
||||
[mapping(selection_metric)],
|
||||
cell_selection = select_cells(
|
||||
selection_metric,
|
||||
out_shape=values.shape,
|
||||
transform=transform,
|
||||
invert=True,
|
||||
cell_area_m2=target_resolution * target_resolution,
|
||||
)
|
||||
selected_cells = cell_selection.mask
|
||||
return RasterPartitionSelection(
|
||||
datasets=datasets,
|
||||
values=values,
|
||||
selected_cells=selected_cells,
|
||||
resolution_x=target_resolution,
|
||||
resolution_y=target_resolution,
|
||||
cell_selection_warning=cell_selection.warning,
|
||||
)
|
||||
except AppError:
|
||||
raise
|
||||
|
||||
@@ -13,6 +13,7 @@ from shapely.ops import transform as shapely_transform
|
||||
|
||||
from app.core.config import Settings, get_settings
|
||||
from app.core.errors import AppError
|
||||
from app.services.raster_cell_selection import select_cells
|
||||
from app.models import Area, Dataset
|
||||
from app.schemas.dhmv import (
|
||||
TerrainMetric,
|
||||
@@ -171,12 +172,17 @@ class TerrainAnalysisService:
|
||||
},
|
||||
status_code=422,
|
||||
)
|
||||
# ``all_touched`` keeps the values of cells the selection only
|
||||
# clips, so a selection finer than one cell still has data to
|
||||
# read. Which of those cells actually count is decided by
|
||||
# ``select_cells`` below, so the normal result is unchanged.
|
||||
clipped, clipped_transform = mask(
|
||||
source,
|
||||
[mapping(analysis_geometry)],
|
||||
crop=True,
|
||||
filled=False,
|
||||
indexes=[1],
|
||||
all_touched=True,
|
||||
)
|
||||
elevation = np.ma.asarray(clipped[0], dtype="float64")
|
||||
raw = elevation.filled(np.nan)
|
||||
@@ -184,12 +190,13 @@ class TerrainAnalysisService:
|
||||
invalid = ~np.isfinite(raw)
|
||||
if nodata is not None:
|
||||
invalid |= raw == float(nodata)
|
||||
selected_cells = geometry_mask(
|
||||
[mapping(analysis_geometry)],
|
||||
cell_selection = select_cells(
|
||||
analysis_geometry,
|
||||
out_shape=elevation.shape,
|
||||
transform=clipped_transform,
|
||||
invert=True,
|
||||
cell_area_m2=abs(float(source.res[0])) * abs(float(source.res[1])),
|
||||
)
|
||||
selected_cells = cell_selection.mask
|
||||
valid_mask = selected_cells & ~np.ma.getmaskarray(elevation) & ~invalid
|
||||
values = raw[valid_mask]
|
||||
if values.size == 0:
|
||||
@@ -319,6 +326,7 @@ class TerrainAnalysisService:
|
||||
sample_count=int(values.size),
|
||||
slope_sample_count=int(slope_values.size),
|
||||
coverage_ratio=round(float(values.size / max(1, selected_cell_count)), 6),
|
||||
cell_selection_warning=cell_selection.warning,
|
||||
resolution_m=round(max(resolution_x, resolution_y), 4),
|
||||
vertical_reference=str(
|
||||
source_metadata.get("vertical_reference")
|
||||
@@ -516,6 +524,7 @@ class TerrainAnalysisService:
|
||||
sample_count=int(values.size),
|
||||
slope_sample_count=int(slope_values.size),
|
||||
coverage_ratio=round(float(values.size / max(1, selected_cell_count)), 6),
|
||||
cell_selection_warning=partition.cell_selection_warning,
|
||||
resolution_m=round(max(partition.resolution_x, partition.resolution_y), 4),
|
||||
vertical_reference=DhmvAcquisitionService.VERTICAL_REFERENCE,
|
||||
summary=TerrainSelectionSummary(
|
||||
|
||||
@@ -13,6 +13,7 @@ from shapely.ops import transform as shapely_transform
|
||||
|
||||
from app.core.config import Settings, get_settings
|
||||
from app.core.errors import AppError
|
||||
from app.services.raster_cell_selection import select_cells
|
||||
from app.models import Area, Dataset
|
||||
from app.schemas.thematic_raster import (
|
||||
ThematicRasterMetric,
|
||||
@@ -118,10 +119,27 @@ class ThematicRasterAnalysisService:
|
||||
details={"pixel_count": expected_cells, "max_pixels": resolved_settings.thematic_raster_max_pixels},
|
||||
status_code=422,
|
||||
)
|
||||
clipped, clipped_transform = mask(source, [mapping(analysis_geometry)], crop=True, filled=False, indexes=[1])
|
||||
# ``all_touched`` keeps the values of cells the selection only
|
||||
# clips, so a selection finer than one cell still has data to
|
||||
# read. Which of those cells actually count is decided by
|
||||
# ``select_cells`` below, so the normal result is unchanged.
|
||||
clipped, clipped_transform = mask(
|
||||
source,
|
||||
[mapping(analysis_geometry)],
|
||||
crop=True,
|
||||
filled=False,
|
||||
indexes=[1],
|
||||
all_touched=True,
|
||||
)
|
||||
band = np.ma.asarray(clipped[0], dtype="float64")
|
||||
raw = band.filled(np.nan)
|
||||
selected = geometry_mask([mapping(analysis_geometry)], out_shape=band.shape, transform=clipped_transform, invert=True)
|
||||
cell_selection = select_cells(
|
||||
analysis_geometry,
|
||||
out_shape=band.shape,
|
||||
transform=clipped_transform,
|
||||
cell_area_m2=abs(float(source.res[0])) * abs(float(source.res[1])),
|
||||
)
|
||||
selected = cell_selection.mask
|
||||
valid = selected & ~np.ma.getmaskarray(band) & np.isfinite(raw)
|
||||
if source.nodata is not None:
|
||||
valid &= ~np.isclose(raw, float(source.nodata))
|
||||
@@ -195,6 +213,7 @@ class ThematicRasterAnalysisService:
|
||||
selected_cell_count=selected_cell_count,
|
||||
valid_cell_count=valid_cell_count,
|
||||
coverage_ratio=round(valid_cell_count / max(1, selected_cell_count), 6),
|
||||
cell_selection_warning=cell_selection.warning,
|
||||
resolution_m=round(max(resolution_x, resolution_y), 4),
|
||||
observation_year=product.observation_year,
|
||||
summary=ThematicRasterSelectionSummary(
|
||||
|
||||
@@ -220,6 +220,73 @@ class VectorFeatureService:
|
||||
source_metadata = dataset.source_metadata if isinstance(dataset.source_metadata, dict) else {}
|
||||
return isinstance(source_metadata.get("selection_aggregation"), dict) or VectorFeatureService._dataset_theme(dataset) is not None
|
||||
|
||||
@staticmethod
|
||||
def deduplicate_rows(rows: list[Any]) -> list[Any]:
|
||||
"""Collapse rows that describe one source feature across partitions.
|
||||
|
||||
Municipal partitions of one product overlap at their shared boundary,
|
||||
so a rectangle drawn across it returns the same building from both.
|
||||
An empty or missing ``source_feature_id`` is not a shared identity —
|
||||
two rows without one are two features, not a duplicate pair.
|
||||
"""
|
||||
|
||||
seen: set[str] = set()
|
||||
kept: list[Any] = []
|
||||
for row in rows:
|
||||
source_feature_id = getattr(row, "source_feature_id", None)
|
||||
identity = str(source_feature_id).strip() if source_feature_id is not None else ""
|
||||
if not identity:
|
||||
kept.append(row)
|
||||
continue
|
||||
if identity in seen:
|
||||
continue
|
||||
seen.add(identity)
|
||||
kept.append(row)
|
||||
return kept
|
||||
|
||||
@staticmethod
|
||||
def count_disclosure(
|
||||
*,
|
||||
total_feature_count: int,
|
||||
fully_covered_feature_count: int | None,
|
||||
) -> dict[str, Any]:
|
||||
"""Describe how much of the counted population the selection cuts.
|
||||
|
||||
A feature that merely touches the drawn rectangle is counted whole,
|
||||
while ``intersection_area`` clips it. Reporting both numbers without
|
||||
saying so puts two figures for different populations side by side. The
|
||||
count stays whole-feature — that is what an operator expects from
|
||||
"objecten" — but says how many of them the edge cuts, and is flagged as
|
||||
an estimate when it does.
|
||||
|
||||
``fully_covered_feature_count`` is ``None`` when the selection covers a
|
||||
pre-clipped whole work area, where no edge effect exists.
|
||||
"""
|
||||
|
||||
if fully_covered_feature_count is None:
|
||||
return {
|
||||
"partially_covered_feature_count": None,
|
||||
"is_estimate": False,
|
||||
"warning": None,
|
||||
}
|
||||
|
||||
partial = max(0, int(total_feature_count) - int(fully_covered_feature_count))
|
||||
if partial <= 0:
|
||||
return {
|
||||
"partially_covered_feature_count": 0,
|
||||
"is_estimate": False,
|
||||
"warning": None,
|
||||
}
|
||||
return {
|
||||
"partially_covered_feature_count": partial,
|
||||
"is_estimate": True,
|
||||
"warning": (
|
||||
f"{partial} van de {int(total_feature_count)} objecten liggen deels buiten de selectie en zijn "
|
||||
"aan de rand doorgesneden. Ze tellen volledig mee in het aantal; oppervlakte- en lengtematen "
|
||||
"gebruiken alleen het deel binnen de selectie."
|
||||
),
|
||||
}
|
||||
|
||||
@staticmethod
|
||||
def constrain_bbox_to_area(
|
||||
bbox: dict[str, Any],
|
||||
@@ -696,6 +763,11 @@ class VectorFeatureService:
|
||||
.limit(safe_limit + 1)
|
||||
.all()
|
||||
)
|
||||
if deduplicate_source_features:
|
||||
# ``total_feature_count`` is already distinct; without this the map
|
||||
# would draw a boundary feature once per partition and the returned
|
||||
# count would exceed the headline number beside it.
|
||||
rows = VectorFeatureService.deduplicate_rows(rows)
|
||||
truncated = total_feature_count > safe_limit
|
||||
selected_rows = rows[:safe_limit]
|
||||
features = [VectorFeatureService._row_to_geojson_feature(row) for row in selected_rows]
|
||||
@@ -767,6 +839,27 @@ class VectorFeatureService:
|
||||
if feature_count is None:
|
||||
feature_count = int(db.query(func.count(VectorFeature.id)).filter(*selection_filter).scalar() or 0)
|
||||
|
||||
# How many of the counted features the selection edge cuts. Skipped for
|
||||
# a pre-clipped whole-area selection, which has no edge to cut against.
|
||||
fully_covered_feature_count: int | None = None
|
||||
if not full_dataset_area and feature_count:
|
||||
try:
|
||||
fully_covered_feature_count = int(
|
||||
db.query(func.count(VectorFeature.id))
|
||||
.filter(*selection_filter)
|
||||
.filter(func.ST_CoveredBy(VectorFeature.geometry, selection_shape))
|
||||
.scalar()
|
||||
or 0
|
||||
)
|
||||
except Exception:
|
||||
# Lightweight unit-test sessions do not implement every spatial
|
||||
# predicate; the count then simply carries no edge disclosure.
|
||||
fully_covered_feature_count = None
|
||||
count_disclosure = VectorFeatureService.count_disclosure(
|
||||
total_feature_count=feature_count,
|
||||
fully_covered_feature_count=fully_covered_feature_count,
|
||||
)
|
||||
|
||||
source_metadata = dataset.source_metadata if isinstance(dataset.source_metadata, dict) else {}
|
||||
config = source_metadata.get("selection_aggregation")
|
||||
if not isinstance(config, dict):
|
||||
@@ -834,9 +927,17 @@ class VectorFeatureService:
|
||||
selection_shape=selection_shape,
|
||||
feature_count=feature_count,
|
||||
full_dataset_area=selection_is_preclipped,
|
||||
partially_covered_feature_count=count_disclosure["partially_covered_feature_count"],
|
||||
)
|
||||
for metric_config in metric_configs
|
||||
]
|
||||
# The whole-feature count carries the edge disclosure; a metric that
|
||||
# already clips to the selection (area, length) does not need it.
|
||||
for computed_metric in metrics:
|
||||
if computed_metric["aggregation_method"] == "feature_count" and count_disclosure["is_estimate"]:
|
||||
computed_metric["is_estimate"] = True
|
||||
computed_metric["warning"] = computed_metric.get("warning") or count_disclosure["warning"]
|
||||
|
||||
primary_metric = metrics[0]
|
||||
return {
|
||||
"metric_label": primary_metric["metric_label"],
|
||||
@@ -845,6 +946,9 @@ class VectorFeatureService:
|
||||
"aggregation_method": primary_metric["aggregation_method"],
|
||||
"primary_metric_key": primary_metric["metric_key"],
|
||||
"feature_count": feature_count,
|
||||
"fully_covered_feature_count": fully_covered_feature_count,
|
||||
"partially_covered_feature_count": count_disclosure["partially_covered_feature_count"],
|
||||
"selection_edge_warning": count_disclosure["warning"],
|
||||
"is_estimate": primary_metric["is_estimate"],
|
||||
"warning": primary_metric.get("warning"),
|
||||
"metrics": metrics,
|
||||
@@ -860,6 +964,7 @@ class VectorFeatureService:
|
||||
selection_shape: Any,
|
||||
feature_count: int,
|
||||
full_dataset_area: bool,
|
||||
partially_covered_feature_count: int | None = None,
|
||||
) -> dict[str, Any]:
|
||||
method = str(config.get("method") or "feature_count")
|
||||
unit = str(config.get("unit") or "objecten")
|
||||
@@ -950,12 +1055,16 @@ class VectorFeatureService:
|
||||
)
|
||||
metric_value = float(aggregate_value or 0.0)
|
||||
if method == "area_weighted_sum" and not full_dataset_area:
|
||||
partial_feature_count = (
|
||||
db.query(func.count(VectorFeature.id))
|
||||
.filter(*metric_filter)
|
||||
.filter(~covered_by_selection)
|
||||
.scalar()
|
||||
)
|
||||
# The selection-edge count was already established for the
|
||||
# feature count; a second query would ask the same question.
|
||||
partial_feature_count = partially_covered_feature_count
|
||||
if partial_feature_count is None:
|
||||
partial_feature_count = (
|
||||
db.query(func.count(VectorFeature.id))
|
||||
.filter(*metric_filter)
|
||||
.filter(~covered_by_selection)
|
||||
.scalar()
|
||||
)
|
||||
is_estimate = bool(config.get("is_estimate", False)) or bool(partial_feature_count)
|
||||
if not is_estimate and config.get("warning_only_when_estimate", True):
|
||||
warning = None
|
||||
|
||||
Reference in New Issue
Block a user