Files
geointel/backend/app/services/statbel_catalog_probe.py
T
Codex da2371d662
GeoIntel CI / docs-smoke (push) Canceled after 0s
GeoIntel CI / contract-smoke (push) Canceled after 0s
Add governed Statbel edition probe
2026-07-16 22:14:48 +02:00

257 lines
9.6 KiB
Python

from __future__ import annotations
from dataclasses import dataclass
from datetime import datetime, timezone
import re
from urllib.parse import urlsplit
from rdflib import Graph, Literal, URIRef
from rdflib.namespace import RDF
_DCAT = "http://www.w3.org/ns/dcat#"
_DCT = "http://purl.org/dc/terms/"
_EXPECTED_TITLE = "Bevolking per statistische sector"
_CATALOG_HOST = "doc.statbel.be"
_CATALOG_PATH = "/publications/DCAT/DCAT_opendata_datasets.ttl"
_STATBEL_HOST = "statbel.fgov.be"
_LANDING_PATH = re.compile(r"^/nl/open-data/bevolking-statistische-sector(?:-\d+)?$")
_DISTRIBUTION_PATH = re.compile(
r"^/sites/default/files/files/opendata/bevolking/sectoren/"
r"OPENDATA_SECTOREN_(20\d{2})(?:_(NEW|OLD))?\.(zip|xlsx)$",
re.IGNORECASE,
)
_DISTRIBUTION_FRAGMENT = re.compile(r"^distribution\d+$")
_ALTERNATIVE_YEAR = re.compile(r"\[Periode:\s*(20\d{2})\]", re.IGNORECASE)
_CC_BY_4 = "https://creativecommons.org/licenses/by/4.0/"
DCT_TITLE = URIRef(f"{_DCT}title")
DCT_ALTERNATIVE = URIRef(f"{_DCT}alternative")
DCT_IDENTIFIER = URIRef(f"{_DCT}identifier")
DCT_LICENSE = URIRef(f"{_DCT}license")
DCT_MODIFIED = URIRef(f"{_DCT}modified")
DCT_TEMPORAL = URIRef(f"{_DCT}temporal")
DCAT_CATALOG = URIRef(f"{_DCAT}Catalog")
DCAT_DATASET = URIRef(f"{_DCAT}Dataset")
DCAT_DISTRIBUTION = URIRef(f"{_DCAT}distribution")
DCAT_LANDING_PAGE = URIRef(f"{_DCAT}landingPage")
DCAT_START_DATE = URIRef(f"{_DCAT}startDate")
class StatbelCatalogError(RuntimeError):
def __init__(self, code: str, message: str) -> None:
super().__init__(message)
self.code = code
self.message = message
@dataclass(frozen=True)
class StatbelPopulationRelease:
identifier: str
version: str
landing_page: str
catalog_modified_at: datetime | None
distribution_count: int
current_distribution_variant: str
legacy_distribution_available: bool
def validate_statbel_catalog_url(url: str) -> str:
parsed = urlsplit(url)
if (
parsed.scheme != "https"
or parsed.hostname != _CATALOG_HOST
or parsed.port not in {None, 443}
or parsed.username
or parsed.password
or parsed.path != _CATALOG_PATH
or parsed.query
or parsed.fragment
):
raise StatbelCatalogError(
"CATALOG_STATBEL_URL_REJECTED",
"De ingestelde Statbel DCAT-catalogus valt buiten de toegestane officiële URL.",
)
return url
def _dutch_literal(values: list[object], expected: str | None = None) -> Literal | None:
for value in values:
if not isinstance(value, Literal) or value.language != "nl":
continue
if expected is None or str(value).strip() == expected:
return value
return None
def _dataset_year(graph: Graph, subject: object) -> int | None:
years: set[int] = set()
for alternative in graph.objects(subject, DCT_ALTERNATIVE):
if isinstance(alternative, Literal) and alternative.language == "nl":
match = _ALTERNATIVE_YEAR.search(str(alternative))
if match:
years.add(int(match.group(1)))
for period in graph.objects(subject, DCT_TEMPORAL):
for value in graph.objects(period, DCAT_START_DATE):
match = re.match(r"^(20\d{2})-\d{2}-\d{2}$", str(value))
if match:
years.add(int(match.group(1)))
if len(years) > 1:
raise StatbelCatalogError(
"CATALOG_STATBEL_PERIOD_AMBIGUOUS",
"De officiële Statbel dataset bevat tegenstrijdige referentiejaren.",
)
return next(iter(years), None)
def _validate_landing_page(url: str) -> str:
parsed = urlsplit(url)
if (
parsed.scheme != "https"
or parsed.hostname != _STATBEL_HOST
or parsed.port not in {None, 443}
or parsed.username
or parsed.password
or not _LANDING_PATH.fullmatch(parsed.path)
or parsed.query
or parsed.fragment
):
raise StatbelCatalogError(
"CATALOG_STATBEL_LANDING_PAGE_REJECTED",
"De Statbel DCAT-dataset verwijst niet naar de toegestane Nederlandstalige landingspagina.",
)
return url
def _validate_distribution(url: str, expected_year: int) -> tuple[str | None, str]:
parsed = urlsplit(url)
match = _DISTRIBUTION_PATH.fullmatch(parsed.path)
if (
parsed.scheme != "https"
or parsed.hostname != _STATBEL_HOST
or parsed.port not in {None, 443}
or parsed.username
or parsed.password
or parsed.query
or not match
or (parsed.fragment and not _DISTRIBUTION_FRAGMENT.fullmatch(parsed.fragment))
):
raise StatbelCatalogError(
"CATALOG_STATBEL_DISTRIBUTION_REJECTED",
"De Statbel DCAT-dataset bevat een distributie buiten de toegestane officiële URL-structuur.",
)
if int(match.group(1)) != expected_year:
raise StatbelCatalogError(
"CATALOG_STATBEL_DISTRIBUTION_YEAR_MISMATCH",
"De Statbel distributie hoort niet bij het gepubliceerde referentiejaar.",
)
return match.group(2).lower() if match.group(2) else None, match.group(3).lower()
def _catalog_modified_at(graph: Graph) -> datetime | None:
dates: list[datetime] = []
for catalog in graph.subjects(RDF.type, DCAT_CATALOG):
for value in graph.objects(catalog, DCT_MODIFIED):
try:
dates.append(datetime.fromisoformat(str(value)).replace(tzinfo=timezone.utc))
except ValueError:
continue
return max(dates) if dates else None
def parse_statbel_population_catalog(content: bytes) -> StatbelPopulationRelease:
try:
text = content.decode("utf-8")
except UnicodeDecodeError as exc:
raise StatbelCatalogError(
"CATALOG_STATBEL_INVALID_TURTLE",
"De officiële Statbel DCAT-catalogus is niet geldige UTF-8 Turtle.",
) from exc
graph = Graph()
try:
graph.parse(data=text, format="turtle")
except Exception as exc:
raise StatbelCatalogError(
"CATALOG_STATBEL_INVALID_TURTLE",
"De officiële Statbel DCAT-catalogus kon niet als RDF/Turtle worden gelezen.",
) from exc
candidates: list[tuple[int, object]] = []
for subject in graph.subjects(RDF.type, DCAT_DATASET):
title = _dutch_literal(list(graph.objects(subject, DCT_TITLE)), _EXPECTED_TITLE)
if title is None:
continue
year = _dataset_year(graph, subject)
if year is not None:
candidates.append((year, subject))
if not candidates:
raise StatbelCatalogError(
"CATALOG_STATBEL_POPULATION_MISSING",
"De officiële Statbel DCAT-catalogus bevat geen herkenbare bevolking-per-sectorpublicatie.",
)
latest_year = max(year for year, _subject in candidates)
latest = [subject for year, subject in candidates if year == latest_year]
if len(latest) != 1:
raise StatbelCatalogError(
"CATALOG_STATBEL_POPULATION_AMBIGUOUS",
"De officiële Statbel DCAT-catalogus bevat meerdere bevolking-per-sectorpublicaties voor hetzelfde nieuwste jaar.",
)
subject = latest[0]
identifiers = [str(value).strip() for value in graph.objects(subject, DCT_IDENTIFIER) if str(value).strip()]
if len(set(identifiers)) != 1:
raise StatbelCatalogError(
"CATALOG_STATBEL_IDENTIFIER_MISSING",
"De nieuwste Statbel bevolking-per-sectorpublicatie heeft geen eenduidige datasetidentiteit.",
)
landing_pages = [
_validate_landing_page(str(value))
for value in graph.objects(subject, DCAT_LANDING_PAGE)
if urlsplit(str(value)).path.startswith("/nl/")
]
if len(set(landing_pages)) != 1:
raise StatbelCatalogError(
"CATALOG_STATBEL_LANDING_PAGE_MISSING",
"De nieuwste Statbel bevolking-per-sectorpublicatie heeft geen eenduidige Nederlandstalige landingspagina.",
)
licenses = {str(value) for value in graph.objects(subject, DCT_LICENSE)}
if _CC_BY_4 not in licenses:
raise StatbelCatalogError(
"CATALOG_STATBEL_LICENSE_MISSING",
"De nieuwste Statbel bevolking-per-sectorpublicatie bevestigt de vereiste CC BY 4.0-licentie niet.",
)
distributions = [str(value) for value in graph.objects(subject, DCAT_DISTRIBUTION)]
if not distributions:
raise StatbelCatalogError(
"CATALOG_STATBEL_DISTRIBUTION_MISSING",
"De nieuwste Statbel bevolking-per-sectorpublicatie bevat geen distributies.",
)
variants: list[tuple[str | None, str]] = [
_validate_distribution(value, latest_year) for value in distributions
]
zip_variants = {variant for variant, file_type in variants if file_type == "zip"}
if latest_year == 2025:
current_variant = "new" if "new" in zip_variants else ""
elif "new" in zip_variants:
current_variant = "new"
elif None in zip_variants:
current_variant = "standard"
else:
current_variant = ""
if not current_variant:
raise StatbelCatalogError(
"CATALOG_STATBEL_CURRENT_DISTRIBUTION_MISSING",
"De nieuwste Statbel bevolking-per-sectorpublicatie bevat geen herkenbare actuele TXT/ZIP-distributie.",
)
return StatbelPopulationRelease(
identifier=identifiers[0],
version=str(latest_year),
landing_page=landing_pages[0],
catalog_modified_at=_catalog_modified_at(graph),
distribution_count=len(distributions),
current_distribution_variant=current_variant,
legacy_distribution_available="old" in zip_variants,
)