distinguish a redrawn footprint from a demolition, and derive estimate

disclosure from data

Change detection had only added/removed/unchanged, so a building extended by
an annexe dropped below the IoU threshold and was reported twice: once as
removed and once as added. That hides exactly the category a change-detection
product exists to show and inflates both counts. A "modified" class now covers
the band between the modified floor and the unchanged threshold.

Matching also ran as a full cross product with no spatial index, unlike the QA
matcher beside it: two municipal building layers meant hundreds of millions of
geometry intersections. It uses an STRtree and considers larger footprints
first, so a big footprint is not left over after a small neighbour claimed its
counterpart.

The assistant guaranteed honesty about estimated values by rewriting the
model's sentences with regular expressions, which only fires when it
recognises the phrasing the model happened to produce. estimate_disclosures
derives the same statement from the metric metadata, so it holds regardless of
how the answer was worded. The prose substitution stays as a second layer.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Jens
2026-08-22 14:33:37 +02:00
co-authored by Claude Opus 5
parent dd87a62e8f
commit 23d6e0372b
6 changed files with 373 additions and 59 deletions
+4
View File
@@ -20,8 +20,12 @@ class ChangeDetectionSummary(BaseModel):
target_feature_count: int
added_count: int
removed_count: int
# A footprint that was redrawn rather than demolished and rebuilt. Without
# this class it appeared as one removal plus one addition.
modified_count: int = 0
unchanged_count: int
iou_threshold: float
modified_iou_threshold: float | None = None
warnings: list[str] = Field(default_factory=list)
generated_at: datetime
geojson: dict
+16
View File
@@ -66,12 +66,28 @@ class AssistantTemporalSeries(BaseModel):
observation_count: int
class AssistantEstimateDisclosure(BaseModel):
"""A value in the answer that the source itself calls an estimate.
Derived from metric metadata rather than from the generated sentences, so
the disclosure is present whatever wording the model chose.
"""
theme: str
label: str
unit: str
source: str
dataset_id: UUID
reason: str
class AssistantQueryResponse(BaseModel):
answer: str
model: str
scope_label: str
context_metrics: list[AssistantContextMetric]
temporal_series: list[AssistantTemporalSeries]
estimate_disclosures: list[AssistantEstimateDisclosure] = Field(default_factory=list)
source_dataset_ids: list[UUID]
warnings: list[str]
generated_at: datetime
+111 -59
View File
@@ -7,6 +7,7 @@ from uuid import UUID
from geoalchemy2.shape import to_shape
from shapely.geometry import mapping
from shapely.geometry.base import BaseGeometry
from shapely.strtree import STRtree
from shapely.validation import make_valid
from sqlalchemy.orm import Session
@@ -28,11 +29,18 @@ class ChangeDetectionService:
target_dataset_id: UUID,
iou_threshold: float = 0.8,
include_unchanged: bool = True,
modified_threshold: float = 0.3,
) -> ChangeDetectionSummary:
if source_dataset_id == target_dataset_id:
raise AppError(code="INVALID_PARAMETERS", message="Source and target datasets must differ", status_code=400)
if iou_threshold < 0 or iou_threshold > 1:
raise AppError(code="INVALID_PARAMETERS", message="iou_threshold must be between 0 and 1", status_code=400)
if modified_threshold < 0 or modified_threshold > iou_threshold:
raise AppError(
code="INVALID_PARAMETERS",
message="modified_threshold must be between 0 and iou_threshold",
status_code=400,
)
source_dataset = ChangeDetectionService._get_project_vector_dataset(db, source_dataset_id, project_id, "Source")
target_dataset = ChangeDetectionService._get_project_vector_dataset(db, target_dataset_id, project_id, "Target")
@@ -45,80 +53,124 @@ class ChangeDetectionService:
if not target_features:
raise AppError(code="EMPTY_VECTOR_DATASET", message="Target dataset has no comparable vector features", status_code=422)
matched_target_indices: set[int] = set()
unchanged: list[dict[str, Any]] = []
removed: list[dict[str, Any]] = []
classified = ChangeDetectionService._classify_features(
source_features,
target_features,
iou_threshold=iou_threshold,
modified_threshold=modified_threshold,
)
for source_feature in source_features:
best_iou = 0.0
best_index: int | None = None
for target_index, target_feature in enumerate(target_features):
if target_index in matched_target_indices:
continue
candidate_iou = ChangeDetectionService._iou(source_feature["geometry"], target_feature["geometry"])
if candidate_iou > best_iou:
best_iou = candidate_iou
best_index = target_index
if best_index is not None and best_iou >= iou_threshold:
matched_target_indices.add(best_index)
if include_unchanged:
unchanged.append(
ChangeDetectionService._feature(
geometry=source_feature["geometry"],
change_type="unchanged",
source_dataset_id=source_dataset_id,
target_dataset_id=target_dataset_id,
source_feature_id=source_feature["feature_id"],
target_feature_id=target_features[best_index]["feature_id"],
iou=best_iou,
properties=source_feature["properties"],
)
)
else:
removed.append(
ChangeDetectionService._feature(
geometry=source_feature["geometry"],
change_type="removed",
source_dataset_id=source_dataset_id,
target_dataset_id=target_dataset_id,
source_feature_id=source_feature["feature_id"],
target_feature_id=None,
iou=best_iou if best_iou > 0 else None,
properties=source_feature["properties"],
)
buckets: dict[str, list[dict[str, Any]]] = {"added": [], "removed": [], "modified": [], "unchanged": []}
for item in classified:
buckets[item["change_type"]].append(
ChangeDetectionService._feature(
geometry=item["geometry"],
change_type=item["change_type"],
source_dataset_id=source_dataset_id,
target_dataset_id=target_dataset_id,
source_feature_id=item["source_feature_id"],
target_feature_id=item["target_feature_id"],
iou=item["iou"],
properties=item["properties"],
)
added = [
ChangeDetectionService._feature(
geometry=target_feature["geometry"],
change_type="added",
source_dataset_id=source_dataset_id,
target_dataset_id=target_dataset_id,
source_feature_id=None,
target_feature_id=target_feature["feature_id"],
iou=None,
properties=target_feature["properties"],
)
for target_index, target_feature in enumerate(target_features)
if target_index not in matched_target_indices
]
geojson_features = added + removed + unchanged
unchanged_count = len(buckets["unchanged"])
if not include_unchanged:
buckets["unchanged"] = []
geojson_features = buckets["added"] + buckets["removed"] + buckets["modified"] + buckets["unchanged"]
return ChangeDetectionSummary(
source_dataset_id=source_dataset_id,
target_dataset_id=target_dataset_id,
source_feature_count=len(source_features),
target_feature_count=len(target_features),
added_count=len(added),
removed_count=len(removed),
unchanged_count=len(unchanged) if include_unchanged else len(matched_target_indices),
added_count=len(buckets["added"]),
removed_count=len(buckets["removed"]),
modified_count=len(buckets["modified"]),
unchanged_count=unchanged_count,
iou_threshold=iou_threshold,
modified_iou_threshold=modified_threshold,
warnings=source_warnings + target_warnings,
generated_at=datetime.now(timezone.utc),
geojson={"type": "FeatureCollection", "features": geojson_features},
)
@staticmethod
def _classify_features(
source_features: list[dict[str, Any]],
target_features: list[dict[str, Any]],
*,
iou_threshold: float,
modified_threshold: float,
) -> list[dict[str, Any]]:
"""Pair source with target footprints and label how each one changed.
Matching is indexed rather than a full cross product: comparing two
municipal building layers is otherwise hundreds of millions of geometry
intersections. Sources are considered largest first so a big footprint
is not left over after a small neighbour claimed its counterpart.
"""
target_geometries = [feature["geometry"] for feature in target_features]
tree = STRtree(target_geometries) if target_geometries else None
claimed: set[int] = set()
classified: list[dict[str, Any]] = []
order = sorted(
range(len(source_features)),
key=lambda index: (-source_features[index]["geometry"].area, str(source_features[index]["feature_id"])),
)
for source_index in order:
source_feature = source_features[source_index]
geometry = source_feature["geometry"]
best_iou = 0.0
best_index: int | None = None
candidates = [] if tree is None else sorted(int(value) for value in tree.query(geometry))
for target_index in candidates:
if target_index in claimed:
continue
candidate_iou = ChangeDetectionService._iou(geometry, target_geometries[target_index])
if candidate_iou > best_iou:
best_iou = candidate_iou
best_index = target_index
if best_index is not None and best_iou >= iou_threshold:
claimed.add(best_index)
change_type = "unchanged"
elif best_index is not None and best_iou >= modified_threshold:
# The same object, redrawn: an annexe, a demolition of one wing,
# or a resurvey. Reporting it as removed + added would hide it.
claimed.add(best_index)
change_type = "modified"
else:
change_type = "removed"
classified.append(
{
"change_type": change_type,
"geometry": geometry if change_type != "modified" else target_geometries[best_index],
"source_feature_id": source_feature["feature_id"],
"target_feature_id": target_features[best_index]["feature_id"] if change_type != "removed" else None,
"iou": best_iou if best_iou > 0 else None,
"properties": source_feature["properties"],
}
)
classified.extend(
{
"change_type": "added",
"geometry": target_feature["geometry"],
"source_feature_id": None,
"target_feature_id": target_feature["feature_id"],
"iou": None,
"properties": target_feature["properties"],
}
for target_index, target_feature in enumerate(target_features)
if target_index not in claimed
)
return classified
@staticmethod
def _get_project_vector_dataset(db: Session, dataset_id: UUID, project_id: UUID, label: str) -> Dataset:
dataset = db.get(Dataset, dataset_id)
@@ -16,6 +16,7 @@ from app.core.errors import AppError
from app.models import Area, Dataset, Project
from app.schemas.assistant import (
AssistantContextMetric,
AssistantEstimateDisclosure,
AssistantModelRead,
AssistantQueryRequest,
AssistantQueryResponse,
@@ -112,6 +113,44 @@ class GeoAssistantService:
}
return themes or None
@classmethod
def estimate_disclosures(
cls,
metrics: list[AssistantContextMetric],
) -> list[AssistantEstimateDisclosure]:
"""List every estimated value behind the answer, straight from metadata.
``ensure_estimate_disclosure`` can only add a caveat when it recognises
the phrasing the model produced, which makes the guarantee dependent on
generated text. This derives the same statement from the source
metadata, so it holds regardless of how the answer was written.
"""
seen: set[tuple[str, UUID]] = set()
disclosures: list[AssistantEstimateDisclosure] = []
for metric in sorted(metrics, key=lambda item: (item.theme, item.label)):
if not metric.is_estimate:
continue
key = (metric.theme, metric.dataset_id)
if key in seen:
continue
seen.add(key)
topic = cls.ESTIMATE_TOPIC_LABELS.get(metric.theme, metric.label)
disclosures.append(
AssistantEstimateDisclosure(
theme=metric.theme,
label=metric.label,
unit=metric.unit,
source=metric.source,
dataset_id=metric.dataset_id,
reason=(
f"De bronmetadata van {metric.source} markeert {topic} als schatting, "
"geen exacte telling."
),
)
)
return disclosures
@classmethod
def ensure_estimate_disclosure(
cls,
@@ -702,6 +741,7 @@ class GeoAssistantService:
scope_label=scope_label,
context_metrics=metrics,
temporal_series=series,
estimate_disclosures=self.estimate_disclosures(metrics),
source_dataset_ids=dataset_ids,
warnings=warnings,
generated_at=datetime.now(timezone.utc),