distinguish a redrawn footprint from a demolition, and derive estimate

disclosure from data

Change detection had only added/removed/unchanged, so a building extended by
an annexe dropped below the IoU threshold and was reported twice: once as
removed and once as added. That hides exactly the category a change-detection
product exists to show and inflates both counts. A "modified" class now covers
the band between the modified floor and the unchanged threshold.

Matching also ran as a full cross product with no spatial index, unlike the QA
matcher beside it: two municipal building layers meant hundreds of millions of
geometry intersections. It uses an STRtree and considers larger footprints
first, so a big footprint is not left over after a small neighbour claimed its
counterpart.

The assistant guaranteed honesty about estimated values by rewriting the
model's sentences with regular expressions, which only fires when it
recognises the phrasing the model happened to produce. estimate_disclosures
derives the same statement from the metric metadata, so it holds regardless of
how the answer was worded. The prose substitution stays as a second layer.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Jens
2026-08-22 14:33:37 +02:00
co-authored by Claude Opus 5
parent dd87a62e8f
commit 23d6e0372b
6 changed files with 373 additions and 59 deletions
+4
View File
@@ -20,8 +20,12 @@ class ChangeDetectionSummary(BaseModel):
target_feature_count: int target_feature_count: int
added_count: int added_count: int
removed_count: int removed_count: int
# A footprint that was redrawn rather than demolished and rebuilt. Without
# this class it appeared as one removal plus one addition.
modified_count: int = 0
unchanged_count: int unchanged_count: int
iou_threshold: float iou_threshold: float
modified_iou_threshold: float | None = None
warnings: list[str] = Field(default_factory=list) warnings: list[str] = Field(default_factory=list)
generated_at: datetime generated_at: datetime
geojson: dict geojson: dict
+16
View File
@@ -66,12 +66,28 @@ class AssistantTemporalSeries(BaseModel):
observation_count: int observation_count: int
class AssistantEstimateDisclosure(BaseModel):
"""A value in the answer that the source itself calls an estimate.
Derived from metric metadata rather than from the generated sentences, so
the disclosure is present whatever wording the model chose.
"""
theme: str
label: str
unit: str
source: str
dataset_id: UUID
reason: str
class AssistantQueryResponse(BaseModel): class AssistantQueryResponse(BaseModel):
answer: str answer: str
model: str model: str
scope_label: str scope_label: str
context_metrics: list[AssistantContextMetric] context_metrics: list[AssistantContextMetric]
temporal_series: list[AssistantTemporalSeries] temporal_series: list[AssistantTemporalSeries]
estimate_disclosures: list[AssistantEstimateDisclosure] = Field(default_factory=list)
source_dataset_ids: list[UUID] source_dataset_ids: list[UUID]
warnings: list[str] warnings: list[str]
generated_at: datetime generated_at: datetime
+111 -59
View File
@@ -7,6 +7,7 @@ from uuid import UUID
from geoalchemy2.shape import to_shape from geoalchemy2.shape import to_shape
from shapely.geometry import mapping from shapely.geometry import mapping
from shapely.geometry.base import BaseGeometry from shapely.geometry.base import BaseGeometry
from shapely.strtree import STRtree
from shapely.validation import make_valid from shapely.validation import make_valid
from sqlalchemy.orm import Session from sqlalchemy.orm import Session
@@ -28,11 +29,18 @@ class ChangeDetectionService:
target_dataset_id: UUID, target_dataset_id: UUID,
iou_threshold: float = 0.8, iou_threshold: float = 0.8,
include_unchanged: bool = True, include_unchanged: bool = True,
modified_threshold: float = 0.3,
) -> ChangeDetectionSummary: ) -> ChangeDetectionSummary:
if source_dataset_id == target_dataset_id: if source_dataset_id == target_dataset_id:
raise AppError(code="INVALID_PARAMETERS", message="Source and target datasets must differ", status_code=400) raise AppError(code="INVALID_PARAMETERS", message="Source and target datasets must differ", status_code=400)
if iou_threshold < 0 or iou_threshold > 1: if iou_threshold < 0 or iou_threshold > 1:
raise AppError(code="INVALID_PARAMETERS", message="iou_threshold must be between 0 and 1", status_code=400) raise AppError(code="INVALID_PARAMETERS", message="iou_threshold must be between 0 and 1", status_code=400)
if modified_threshold < 0 or modified_threshold > iou_threshold:
raise AppError(
code="INVALID_PARAMETERS",
message="modified_threshold must be between 0 and iou_threshold",
status_code=400,
)
source_dataset = ChangeDetectionService._get_project_vector_dataset(db, source_dataset_id, project_id, "Source") source_dataset = ChangeDetectionService._get_project_vector_dataset(db, source_dataset_id, project_id, "Source")
target_dataset = ChangeDetectionService._get_project_vector_dataset(db, target_dataset_id, project_id, "Target") target_dataset = ChangeDetectionService._get_project_vector_dataset(db, target_dataset_id, project_id, "Target")
@@ -45,80 +53,124 @@ class ChangeDetectionService:
if not target_features: if not target_features:
raise AppError(code="EMPTY_VECTOR_DATASET", message="Target dataset has no comparable vector features", status_code=422) raise AppError(code="EMPTY_VECTOR_DATASET", message="Target dataset has no comparable vector features", status_code=422)
matched_target_indices: set[int] = set() classified = ChangeDetectionService._classify_features(
unchanged: list[dict[str, Any]] = [] source_features,
removed: list[dict[str, Any]] = [] target_features,
iou_threshold=iou_threshold,
modified_threshold=modified_threshold,
)
for source_feature in source_features: buckets: dict[str, list[dict[str, Any]]] = {"added": [], "removed": [], "modified": [], "unchanged": []}
best_iou = 0.0 for item in classified:
best_index: int | None = None buckets[item["change_type"]].append(
for target_index, target_feature in enumerate(target_features): ChangeDetectionService._feature(
if target_index in matched_target_indices: geometry=item["geometry"],
continue change_type=item["change_type"],
candidate_iou = ChangeDetectionService._iou(source_feature["geometry"], target_feature["geometry"]) source_dataset_id=source_dataset_id,
if candidate_iou > best_iou: target_dataset_id=target_dataset_id,
best_iou = candidate_iou source_feature_id=item["source_feature_id"],
best_index = target_index target_feature_id=item["target_feature_id"],
iou=item["iou"],
if best_index is not None and best_iou >= iou_threshold: properties=item["properties"],
matched_target_indices.add(best_index)
if include_unchanged:
unchanged.append(
ChangeDetectionService._feature(
geometry=source_feature["geometry"],
change_type="unchanged",
source_dataset_id=source_dataset_id,
target_dataset_id=target_dataset_id,
source_feature_id=source_feature["feature_id"],
target_feature_id=target_features[best_index]["feature_id"],
iou=best_iou,
properties=source_feature["properties"],
)
)
else:
removed.append(
ChangeDetectionService._feature(
geometry=source_feature["geometry"],
change_type="removed",
source_dataset_id=source_dataset_id,
target_dataset_id=target_dataset_id,
source_feature_id=source_feature["feature_id"],
target_feature_id=None,
iou=best_iou if best_iou > 0 else None,
properties=source_feature["properties"],
)
) )
added = [
ChangeDetectionService._feature(
geometry=target_feature["geometry"],
change_type="added",
source_dataset_id=source_dataset_id,
target_dataset_id=target_dataset_id,
source_feature_id=None,
target_feature_id=target_feature["feature_id"],
iou=None,
properties=target_feature["properties"],
) )
for target_index, target_feature in enumerate(target_features)
if target_index not in matched_target_indices
]
geojson_features = added + removed + unchanged unchanged_count = len(buckets["unchanged"])
if not include_unchanged:
buckets["unchanged"] = []
geojson_features = buckets["added"] + buckets["removed"] + buckets["modified"] + buckets["unchanged"]
return ChangeDetectionSummary( return ChangeDetectionSummary(
source_dataset_id=source_dataset_id, source_dataset_id=source_dataset_id,
target_dataset_id=target_dataset_id, target_dataset_id=target_dataset_id,
source_feature_count=len(source_features), source_feature_count=len(source_features),
target_feature_count=len(target_features), target_feature_count=len(target_features),
added_count=len(added), added_count=len(buckets["added"]),
removed_count=len(removed), removed_count=len(buckets["removed"]),
unchanged_count=len(unchanged) if include_unchanged else len(matched_target_indices), modified_count=len(buckets["modified"]),
unchanged_count=unchanged_count,
iou_threshold=iou_threshold, iou_threshold=iou_threshold,
modified_iou_threshold=modified_threshold,
warnings=source_warnings + target_warnings, warnings=source_warnings + target_warnings,
generated_at=datetime.now(timezone.utc), generated_at=datetime.now(timezone.utc),
geojson={"type": "FeatureCollection", "features": geojson_features}, geojson={"type": "FeatureCollection", "features": geojson_features},
) )
@staticmethod
def _classify_features(
source_features: list[dict[str, Any]],
target_features: list[dict[str, Any]],
*,
iou_threshold: float,
modified_threshold: float,
) -> list[dict[str, Any]]:
"""Pair source with target footprints and label how each one changed.
Matching is indexed rather than a full cross product: comparing two
municipal building layers is otherwise hundreds of millions of geometry
intersections. Sources are considered largest first so a big footprint
is not left over after a small neighbour claimed its counterpart.
"""
target_geometries = [feature["geometry"] for feature in target_features]
tree = STRtree(target_geometries) if target_geometries else None
claimed: set[int] = set()
classified: list[dict[str, Any]] = []
order = sorted(
range(len(source_features)),
key=lambda index: (-source_features[index]["geometry"].area, str(source_features[index]["feature_id"])),
)
for source_index in order:
source_feature = source_features[source_index]
geometry = source_feature["geometry"]
best_iou = 0.0
best_index: int | None = None
candidates = [] if tree is None else sorted(int(value) for value in tree.query(geometry))
for target_index in candidates:
if target_index in claimed:
continue
candidate_iou = ChangeDetectionService._iou(geometry, target_geometries[target_index])
if candidate_iou > best_iou:
best_iou = candidate_iou
best_index = target_index
if best_index is not None and best_iou >= iou_threshold:
claimed.add(best_index)
change_type = "unchanged"
elif best_index is not None and best_iou >= modified_threshold:
# The same object, redrawn: an annexe, a demolition of one wing,
# or a resurvey. Reporting it as removed + added would hide it.
claimed.add(best_index)
change_type = "modified"
else:
change_type = "removed"
classified.append(
{
"change_type": change_type,
"geometry": geometry if change_type != "modified" else target_geometries[best_index],
"source_feature_id": source_feature["feature_id"],
"target_feature_id": target_features[best_index]["feature_id"] if change_type != "removed" else None,
"iou": best_iou if best_iou > 0 else None,
"properties": source_feature["properties"],
}
)
classified.extend(
{
"change_type": "added",
"geometry": target_feature["geometry"],
"source_feature_id": None,
"target_feature_id": target_feature["feature_id"],
"iou": None,
"properties": target_feature["properties"],
}
for target_index, target_feature in enumerate(target_features)
if target_index not in claimed
)
return classified
@staticmethod @staticmethod
def _get_project_vector_dataset(db: Session, dataset_id: UUID, project_id: UUID, label: str) -> Dataset: def _get_project_vector_dataset(db: Session, dataset_id: UUID, project_id: UUID, label: str) -> Dataset:
dataset = db.get(Dataset, dataset_id) dataset = db.get(Dataset, dataset_id)
@@ -16,6 +16,7 @@ from app.core.errors import AppError
from app.models import Area, Dataset, Project from app.models import Area, Dataset, Project
from app.schemas.assistant import ( from app.schemas.assistant import (
AssistantContextMetric, AssistantContextMetric,
AssistantEstimateDisclosure,
AssistantModelRead, AssistantModelRead,
AssistantQueryRequest, AssistantQueryRequest,
AssistantQueryResponse, AssistantQueryResponse,
@@ -112,6 +113,44 @@ class GeoAssistantService:
} }
return themes or None return themes or None
@classmethod
def estimate_disclosures(
cls,
metrics: list[AssistantContextMetric],
) -> list[AssistantEstimateDisclosure]:
"""List every estimated value behind the answer, straight from metadata.
``ensure_estimate_disclosure`` can only add a caveat when it recognises
the phrasing the model produced, which makes the guarantee dependent on
generated text. This derives the same statement from the source
metadata, so it holds regardless of how the answer was written.
"""
seen: set[tuple[str, UUID]] = set()
disclosures: list[AssistantEstimateDisclosure] = []
for metric in sorted(metrics, key=lambda item: (item.theme, item.label)):
if not metric.is_estimate:
continue
key = (metric.theme, metric.dataset_id)
if key in seen:
continue
seen.add(key)
topic = cls.ESTIMATE_TOPIC_LABELS.get(metric.theme, metric.label)
disclosures.append(
AssistantEstimateDisclosure(
theme=metric.theme,
label=metric.label,
unit=metric.unit,
source=metric.source,
dataset_id=metric.dataset_id,
reason=(
f"De bronmetadata van {metric.source} markeert {topic} als schatting, "
"geen exacte telling."
),
)
)
return disclosures
@classmethod @classmethod
def ensure_estimate_disclosure( def ensure_estimate_disclosure(
cls, cls,
@@ -702,6 +741,7 @@ class GeoAssistantService:
scope_label=scope_label, scope_label=scope_label,
context_metrics=metrics, context_metrics=metrics,
temporal_series=series, temporal_series=series,
estimate_disclosures=self.estimate_disclosures(metrics),
source_dataset_ids=dataset_ids, source_dataset_ids=dataset_ids,
warnings=warnings, warnings=warnings,
generated_at=datetime.now(timezone.utc), generated_at=datetime.now(timezone.utc),
@@ -0,0 +1,88 @@
"""Estimate disclosure must come from the data, not from patching prose.
``ensure_estimate_disclosure`` rewrites the model's sentences with regular
expressions to insert the word "schatting". That only fires when the generated
text happens to contain one of the phrasings it knows, so whether a number is
labelled an estimate depends on how the language model worded it. The
disclosure is derived from the metric metadata instead, so the honesty of the
answer no longer depends on string matching.
"""
from __future__ import annotations
from datetime import datetime, timezone
from uuid import uuid4
from app.schemas.assistant import AssistantContextMetric
from app.services.geo_assistant_service import GeoAssistantService
def _metric(theme: str, label: str, *, is_estimate: bool) -> AssistantContextMetric:
return AssistantContextMetric(
theme=theme,
label=label,
value=36783.0,
unit="inwoners",
source="Statbel",
dataset_id=uuid4(),
observed_at=datetime(2024, 1, 1, tzinfo=timezone.utc),
is_estimate=is_estimate,
)
def test_every_estimated_metric_produces_a_disclosure() -> None:
metrics = [
_metric("population", "Inwoners", is_estimate=True),
_metric("buildings", "Gebouwen", is_estimate=False),
]
disclosures = GeoAssistantService.estimate_disclosures(metrics)
assert len(disclosures) == 1
assert disclosures[0].theme == "population"
assert disclosures[0].label == "Inwoners"
assert disclosures[0].source == "Statbel"
assert disclosures[0].dataset_id == metrics[0].dataset_id
assert "schatting" in disclosures[0].reason.casefold()
def test_disclosure_does_not_depend_on_the_generated_wording() -> None:
"""The regex path only fires on phrasings it recognises; this does not."""
metrics = [_metric("population", "Inwoners", is_estimate=True)]
patched = GeoAssistantService.ensure_estimate_disclosure(
"Er wonen daar 36.783 mensen.", metrics
)
disclosures = GeoAssistantService.estimate_disclosures(metrics)
# The prose was left untouched because no known phrase matched...
assert "Datakwaliteit" not in patched
# ...but the structured disclosure is present regardless.
assert len(disclosures) == 1
def test_no_estimates_means_no_disclosures() -> None:
metrics = [_metric("buildings", "Gebouwen", is_estimate=False)]
assert GeoAssistantService.estimate_disclosures(metrics) == []
def test_disclosures_are_deduplicated_per_theme_and_dataset() -> None:
shared = _metric("population", "Inwoners", is_estimate=True)
duplicate = AssistantContextMetric(**{**shared.model_dump(), "label": "Inwoners (2024)"})
disclosures = GeoAssistantService.estimate_disclosures([shared, duplicate])
assert len(disclosures) == 1
def test_disclosures_are_ordered_deterministically() -> None:
metrics = [
_metric("space_occupation", "Ruimtebeslag", is_estimate=True),
_metric("population", "Inwoners", is_estimate=True),
]
themes = [item.theme for item in GeoAssistantService.estimate_disclosures(metrics)]
assert themes == ["population", "space_occupation"]
@@ -0,0 +1,114 @@
"""An extended building is a change, not a deletion plus a new building.
With only added/removed/unchanged, a footprint that grew by an annexe drops
below the IoU threshold and is reported twice: once as removed and once as
added. That hides exactly the category a change-detection product exists to
show, and inflates both counts.
"""
from __future__ import annotations
import pytest
from shapely.geometry import box
from app.services.change_detection_service import ChangeDetectionService
def _feature(feature_id: str, geometry):
return {"feature_id": feature_id, "properties": {}, "geometry": geometry}
def _classify(source, target, *, iou_threshold=0.8, modified_threshold=0.3):
return ChangeDetectionService._classify_features(
source,
target,
iou_threshold=iou_threshold,
modified_threshold=modified_threshold,
)
def test_an_extended_footprint_is_reported_as_modified() -> None:
source = [_feature("b1", box(0, 0, 10, 10))]
target = [_feature("b1-new", box(0, 0, 10, 14))] # IoU 100/140 = 0.71
result = _classify(source, target)
assert [item["change_type"] for item in result] == ["modified"]
assert result[0]["source_feature_id"] == "b1"
assert result[0]["target_feature_id"] == "b1-new"
assert result[0]["iou"] == pytest.approx(100 / 140)
def test_a_nearly_identical_footprint_is_unchanged() -> None:
source = [_feature("b1", box(0, 0, 10, 10))]
target = [_feature("b1", box(0, 0, 10, 10.2))]
result = _classify(source, target)
assert [item["change_type"] for item in result] == ["unchanged"]
def test_a_genuinely_new_building_stays_added() -> None:
source = [_feature("b1", box(0, 0, 10, 10))]
target = [_feature("b1", box(0, 0, 10, 10)), _feature("b2", box(50, 50, 60, 60))]
result = _classify(source, target)
assert sorted(item["change_type"] for item in result) == ["added", "unchanged"]
def test_a_demolished_building_stays_removed() -> None:
source = [_feature("b1", box(0, 0, 10, 10)), _feature("b2", box(50, 50, 60, 60))]
target = [_feature("b1", box(0, 0, 10, 10))]
result = _classify(source, target)
assert sorted(item["change_type"] for item in result) == ["removed", "unchanged"]
def test_barely_overlapping_footprints_are_not_called_modified() -> None:
"""Below the modified floor the two are separate objects, not one changed."""
source = [_feature("b1", box(0, 0, 10, 10))]
target = [_feature("b2", box(9, 9, 19, 19))] # IoU ~0.005
result = _classify(source, target)
assert sorted(item["change_type"] for item in result) == ["added", "removed"]
def test_each_target_is_claimed_at_most_once() -> None:
source = [_feature("a", box(0, 0, 10, 10)), _feature("b", box(0, 0, 10, 12))]
target = [_feature("t", box(0, 0, 10, 10))]
result = _classify(source, target)
claimed = [item for item in result if item["target_feature_id"] == "t"]
assert len(claimed) == 1
def test_classification_is_independent_of_input_order() -> None:
source = [_feature("a", box(0, 0, 10, 10)), _feature("b", box(30, 30, 40, 40))]
target = [_feature("a2", box(0, 0, 10, 14)), _feature("c", box(70, 70, 80, 80))]
forward = _classify(source, target)
reverse = _classify(list(reversed(source)), list(reversed(target)))
def signature(items):
return sorted(
(item["change_type"], item["source_feature_id"], item["target_feature_id"]) for item in items
)
assert signature(forward) == signature(reverse)
def test_large_populations_do_not_use_a_full_cross_product() -> None:
"""A spatial index keeps a city-sized comparison tractable."""
source = [_feature(f"s{i}", box(i * 10, 0, i * 10 + 8, 8)) for i in range(400)]
target = [_feature(f"t{i}", box(i * 10, 0, i * 10 + 8, 8)) for i in range(400)]
result = _classify(source, target)
assert all(item["change_type"] == "unchanged" for item in result)
assert len(result) == 400