441 lines
20 KiB
Python
441 lines
20 KiB
Python
"""Fail-closed provenance gates at data-consumption boundaries.
|
|
|
|
Import validation protects newly staged assets, but a persisted record can
|
|
subsequently become quarantined or have its provenance marked incomplete. The
|
|
callers of this service therefore re-check the durable Dataset state directly
|
|
before production inference, QA, derived processing, export, or authoritative
|
|
coverage reporting.
|
|
|
|
The only legacy relaxation is deliberately narrow: an *explicitly tagged*
|
|
fixture with no Phase-2 state can be used for fixture QA. A caller-provided
|
|
``fixture_mode`` flag alone never creates that trust claim. Fixture data can
|
|
never become a production inference, derived-processing, export or
|
|
authoritative-coverage input, and it never relaxes a recorded failed,
|
|
incomplete, or quarantined state.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from collections.abc import Mapping
|
|
from dataclasses import dataclass
|
|
import re
|
|
from typing import Any, Literal
|
|
|
|
from sqlalchemy import inspect as sa_inspect
|
|
|
|
from app.core.errors import AppError
|
|
from app.models import Dataset
|
|
|
|
|
|
DatasetConsumptionPurpose = Literal[
|
|
"production_inference",
|
|
"quality_assessment",
|
|
"reference_validation",
|
|
"derived_processing",
|
|
"authoritative_coverage",
|
|
"export",
|
|
]
|
|
|
|
_FIXTURE_SOURCE_KEYS = {"fixture", "test", "test_fixture", "test-fixture", "unit-test-fixture"}
|
|
_UNTRUSTED_SOURCE_KEYS = {"manual", "fixture", "experimental", "legacy_unknown"}
|
|
_CHECKSUM_SHA256 = re.compile(r"^[0-9a-f]{64}$", re.IGNORECASE)
|
|
_CONSUMABLE_SNAPSHOT_FRESHNESS = {"current", "not_applicable"}
|
|
_VALID_PURPOSES = {
|
|
"production_inference",
|
|
"quality_assessment",
|
|
"reference_validation",
|
|
"derived_processing",
|
|
"authoritative_coverage",
|
|
"export",
|
|
}
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class DatasetConsumptionDecision:
|
|
"""Auditable decision returned by the consumption gate."""
|
|
|
|
eligible: bool
|
|
purpose: DatasetConsumptionPurpose
|
|
fixture_legacy_exception: bool
|
|
reasons: tuple[str, ...]
|
|
evidence: dict[str, Any]
|
|
|
|
|
|
class DatasetConsumptionGate:
|
|
"""Evaluate durable provenance before a Dataset is consumed downstream."""
|
|
|
|
@staticmethod
|
|
def _value(dataset: Any, field: str, default: Any = None) -> Any:
|
|
if isinstance(dataset, Mapping):
|
|
return dataset.get(field, default)
|
|
return getattr(dataset, field, default)
|
|
|
|
@staticmethod
|
|
def _mapping(value: Any) -> dict[str, Any]:
|
|
return dict(value) if isinstance(value, Mapping) else {}
|
|
|
|
@classmethod
|
|
def _normalise(cls, value: Any) -> str:
|
|
return str(value or "").strip().lower()
|
|
|
|
@classmethod
|
|
def is_explicit_fixture(cls, dataset: Any) -> bool:
|
|
source = cls._normalise(cls._value(dataset, "source"))
|
|
source_name = cls._normalise(cls._value(dataset, "source_name"))
|
|
metadata = cls._mapping(cls._value(dataset, "metadata_json"))
|
|
source_metadata = cls._mapping(cls._value(dataset, "source_metadata"))
|
|
provenance = cls._mapping(cls._value(dataset, "provenance_metadata"))
|
|
return bool(
|
|
source in _FIXTURE_SOURCE_KEYS
|
|
or source_name in _FIXTURE_SOURCE_KEYS
|
|
or metadata.get("fixture") is True
|
|
or metadata.get("fixture_mode") is True
|
|
or source_metadata.get("fixture") is True
|
|
or source_metadata.get("fixture_mode") is True
|
|
or provenance.get("fixture") is True
|
|
or provenance.get("fixture_mode") is True
|
|
)
|
|
|
|
@classmethod
|
|
def _phase2_state_is_absent(cls, dataset: Any) -> bool:
|
|
fields = (
|
|
"data_contract_key",
|
|
"data_contract_version",
|
|
"validation_status",
|
|
"provenance_status",
|
|
"lineage_status",
|
|
"quarantine_status",
|
|
"source_registry_id",
|
|
"source_snapshot_id",
|
|
)
|
|
return all(cls._value(dataset, field) in {None, ""} for field in fields)
|
|
|
|
@staticmethod
|
|
def _is_transient_orm_dataset(dataset: Any) -> bool:
|
|
"""Recognize only unpersisted ORM fixtures, never database rows."""
|
|
|
|
if not isinstance(dataset, Dataset):
|
|
return False
|
|
try:
|
|
return bool(sa_inspect(dataset).transient)
|
|
except Exception: # pragma: no cover - defensive for unusual test doubles
|
|
return False
|
|
|
|
@classmethod
|
|
def _evidence(
|
|
cls,
|
|
dataset: Any,
|
|
purpose: DatasetConsumptionPurpose,
|
|
reference_task: str | None = None,
|
|
) -> dict[str, Any]:
|
|
source_registry = cls._value(dataset, "source_registry")
|
|
source_snapshot = cls._value(dataset, "source_snapshot")
|
|
source_policy = cls._mapping(cls._value(source_registry, "usage_policy_json"))
|
|
validation_authority = cls._mapping(source_policy.get("validation_authority"))
|
|
reference_approvals = cls._mapping(source_policy.get("reference_validation_approvals"))
|
|
return {
|
|
"dataset_id": str(cls._value(dataset, "id") or ""),
|
|
"purpose": purpose,
|
|
"dataset_status": cls._normalise(cls._value(dataset, "status")),
|
|
"source": cls._normalise(cls._value(dataset, "source")),
|
|
"source_name": cls._normalise(cls._value(dataset, "source_name")),
|
|
"data_contract_key": cls._value(dataset, "data_contract_key"),
|
|
"data_contract_version": cls._value(dataset, "data_contract_version"),
|
|
"validation_status": cls._normalise(cls._value(dataset, "validation_status")),
|
|
"provenance_status": cls._normalise(cls._value(dataset, "provenance_status")),
|
|
"lineage_status": cls._normalise(cls._value(dataset, "lineage_status")),
|
|
"quarantine_status": cls._normalise(cls._value(dataset, "quarantine_status")),
|
|
"checksum_sha256": cls._value(dataset, "checksum_sha256"),
|
|
"source_registry_id": str(cls._value(dataset, "source_registry_id") or ""),
|
|
"source_snapshot_id": str(cls._value(dataset, "source_snapshot_id") or ""),
|
|
"source_classification": cls._normalise(cls._value(source_registry, "classification")),
|
|
"source_key": cls._normalise(cls._value(source_registry, "source_key")),
|
|
"source_ground_truth_allowed": source_policy.get("ground_truth_allowed") is True,
|
|
"source_validation_authority": validation_authority,
|
|
"source_reference_validation_approvals": reference_approvals,
|
|
"source_authority_scope": cls._mapping(cls._value(source_registry, "authority_scope_json")),
|
|
"reference_task": cls._normalise(reference_task),
|
|
"snapshot_source_registry_id": str(cls._value(source_snapshot, "source_registry_id") or ""),
|
|
"snapshot_ingest_status": cls._normalise(cls._value(source_snapshot, "ingest_status")),
|
|
"snapshot_freshness_status": cls._normalise(cls._value(source_snapshot, "freshness_status")),
|
|
"snapshot_checksum_sha256": cls._value(source_snapshot, "checksum_sha256"),
|
|
}
|
|
|
|
@classmethod
|
|
def evaluate(
|
|
cls,
|
|
dataset: Any,
|
|
*,
|
|
purpose: DatasetConsumptionPurpose,
|
|
fixture_mode: bool = False,
|
|
reference_task: str | None = None,
|
|
) -> DatasetConsumptionDecision:
|
|
"""Return a stable decision without mutating the Dataset.
|
|
|
|
``fixture_mode`` can be supplied only by an explicitly fixture-only
|
|
caller. It is evidence for a QA fixture path, never a relaxation for
|
|
a production boundary or an explicit unsafe state.
|
|
"""
|
|
|
|
if purpose not in _VALID_PURPOSES:
|
|
raise ValueError(f"Unsupported dataset-consumption purpose: {purpose}")
|
|
|
|
evidence = cls._evidence(dataset, purpose, reference_task)
|
|
reasons: list[str] = []
|
|
explicit_fixture = cls.is_explicit_fixture(dataset)
|
|
phase2_absent = cls._phase2_state_is_absent(dataset)
|
|
|
|
# These are irrevocable safety states. They are checked before a
|
|
# fixture exception, so fixture rows cannot hide a bad recorded state.
|
|
if evidence["dataset_status"] in {"failed", "quarantined"}:
|
|
reasons.append("dataset_status_unsafe")
|
|
if evidence["quarantine_status"] == "quarantined":
|
|
reasons.append("dataset_quarantined")
|
|
if evidence["validation_status"] == "failed":
|
|
reasons.append("validation_failed")
|
|
if evidence["provenance_status"] == "incomplete":
|
|
reasons.append("provenance_incomplete")
|
|
if evidence["lineage_status"] == "incomplete":
|
|
reasons.append("lineage_incomplete")
|
|
if evidence["snapshot_ingest_status"] in {"failed", "quarantined"}:
|
|
reasons.append("source_snapshot_unsafe")
|
|
# A source family can be authoritative while an individual snapshot
|
|
# remains too old or insufficiently described to trust. Historical
|
|
# data that is intentionally valid needs an explicit
|
|
# ``not_applicable`` contract policy; an omitted, due or stale status
|
|
# cannot silently enter a production boundary.
|
|
if evidence["source_snapshot_id"] and evidence["snapshot_freshness_status"] not in _CONSUMABLE_SNAPSHOT_FRESHNESS:
|
|
reasons.append("source_snapshot_freshness_not_eligible")
|
|
|
|
if reasons:
|
|
return DatasetConsumptionDecision(
|
|
eligible=False,
|
|
purpose=purpose,
|
|
fixture_legacy_exception=False,
|
|
reasons=tuple(sorted(set(reasons))),
|
|
evidence=evidence,
|
|
)
|
|
|
|
# Fixtures are evidence for tests and QA only. They cannot become
|
|
# production inference, derived processing or export inputs merely by
|
|
# presenting a fixture flag at a public service boundary.
|
|
if phase2_absent and explicit_fixture:
|
|
if purpose == "quality_assessment":
|
|
return DatasetConsumptionDecision(
|
|
eligible=True,
|
|
purpose=purpose,
|
|
fixture_legacy_exception=True,
|
|
reasons=(),
|
|
evidence=evidence,
|
|
)
|
|
if purpose == "authoritative_coverage":
|
|
reasons.append("fixture_not_authoritative_coverage")
|
|
else:
|
|
reasons.append("fixture_qa_only")
|
|
elif phase2_absent and fixture_mode:
|
|
# `fixture_mode` is a caller flag, not a source trust claim. A
|
|
# manual/unknown production dataset must never self-designate as a
|
|
# fixture merely by supplying this parameter.
|
|
reasons.append("fixture_source_required")
|
|
|
|
# Existing service tests construct transient SQLAlchemy Dataset objects
|
|
# directly rather than retrieving a persisted row. A production
|
|
# `db.get()` result is persistent and never enters this branch. This
|
|
# compatibility path is intentionally unavailable to coverage, where
|
|
# a fixture must never appear authoritative.
|
|
if (
|
|
phase2_absent
|
|
and not reasons
|
|
and cls._is_transient_orm_dataset(dataset)
|
|
and purpose == "quality_assessment"
|
|
):
|
|
return DatasetConsumptionDecision(
|
|
eligible=True,
|
|
purpose=purpose,
|
|
fixture_legacy_exception=True,
|
|
reasons=(),
|
|
evidence=evidence,
|
|
)
|
|
|
|
# QA unit tests intentionally use projection objects
|
|
# instead of persisted ORM Datasets. Those projections cannot enter an
|
|
# application API boundary; keep the exception isolated to read-only
|
|
# candidate verification. Inference, reference validation, derived
|
|
# processing, export and coverage never accept a projection.
|
|
if phase2_absent and not reasons and not isinstance(dataset, Dataset) and purpose == "quality_assessment":
|
|
return DatasetConsumptionDecision(
|
|
eligible=True,
|
|
purpose=purpose,
|
|
fixture_legacy_exception=True,
|
|
reasons=(),
|
|
evidence=evidence,
|
|
)
|
|
|
|
if phase2_absent:
|
|
reasons.append("phase2_provenance_missing")
|
|
if evidence["dataset_status"] != "ready":
|
|
reasons.append("dataset_not_ready")
|
|
if evidence["validation_status"] != "passed":
|
|
reasons.append("validation_not_passed")
|
|
if evidence["provenance_status"] != "complete":
|
|
reasons.append("provenance_not_complete")
|
|
if evidence["lineage_status"] not in {"complete", "not_applicable"}:
|
|
reasons.append("lineage_not_complete")
|
|
if evidence["quarantine_status"] != "not_quarantined":
|
|
reasons.append("quarantine_status_not_clear")
|
|
if not evidence["data_contract_key"] or not evidence["data_contract_version"]:
|
|
reasons.append("data_contract_not_versioned")
|
|
if not _CHECKSUM_SHA256.fullmatch(str(evidence["checksum_sha256"] or "")):
|
|
reasons.append("dataset_checksum_invalid")
|
|
if not evidence["source_registry_id"]:
|
|
reasons.append("source_registry_missing")
|
|
if not evidence["source_snapshot_id"]:
|
|
reasons.append("source_snapshot_missing")
|
|
if evidence["source_registry_id"] and not evidence["source_classification"]:
|
|
reasons.append("source_registry_unresolved")
|
|
if evidence["source_snapshot_id"] and evidence["snapshot_ingest_status"] != "ingested":
|
|
reasons.append("source_snapshot_not_ingested")
|
|
if evidence["source_snapshot_id"] and not _CHECKSUM_SHA256.fullmatch(
|
|
str(evidence["snapshot_checksum_sha256"] or "")
|
|
):
|
|
reasons.append("source_snapshot_checksum_invalid")
|
|
if (
|
|
evidence["source_registry_id"]
|
|
and evidence["source_snapshot_id"]
|
|
and evidence["snapshot_source_registry_id"]
|
|
and evidence["source_registry_id"] != evidence["snapshot_source_registry_id"]
|
|
):
|
|
reasons.append("source_snapshot_registry_mismatch")
|
|
if (
|
|
_CHECKSUM_SHA256.fullmatch(str(evidence["checksum_sha256"] or ""))
|
|
and _CHECKSUM_SHA256.fullmatch(str(evidence["snapshot_checksum_sha256"] or ""))
|
|
and str(evidence["checksum_sha256"]).lower() != str(evidence["snapshot_checksum_sha256"]).lower()
|
|
):
|
|
reasons.append("source_snapshot_checksum_mismatch")
|
|
|
|
if evidence["source_key"] and evidence["source_name"] and evidence["source_key"] != evidence["source_name"]:
|
|
reasons.append("source_registry_identity_mismatch")
|
|
|
|
# Contextual, corroborative, authoritative and properly derived
|
|
# sources can serve their declared non-ground-truth roles once the
|
|
# full governed contract passes. Experimental/manual sources cannot
|
|
# cross a production boundary; only an explicitly marked fixture may
|
|
# participate in candidate QA.
|
|
experimental_source = (
|
|
evidence["source_classification"] == "experimental"
|
|
or evidence["source_key"] in _UNTRUSTED_SOURCE_KEYS
|
|
)
|
|
if experimental_source:
|
|
if purpose == "quality_assessment" and explicit_fixture:
|
|
pass
|
|
elif purpose == "quality_assessment":
|
|
reasons.append("experimental_source_requires_fixture_qa")
|
|
else:
|
|
reasons.append("experimental_source_not_allowed_for_purpose")
|
|
|
|
if purpose == "authoritative_coverage":
|
|
if evidence["source_classification"] != "authoritative":
|
|
reasons.append("coverage_source_not_authoritative")
|
|
if evidence["source_key"] and evidence["source_name"] and evidence["source_key"] != evidence["source_name"]:
|
|
reasons.append("coverage_source_identity_mismatch")
|
|
|
|
if purpose == "reference_validation":
|
|
if cls._normalise(cls._value(dataset, "dataset_role")) != "reference":
|
|
reasons.append("reference_dataset_role_required")
|
|
if evidence["source_classification"] != "authoritative":
|
|
reasons.append("reference_source_not_authoritative")
|
|
if evidence["source_ground_truth_allowed"] is not True:
|
|
reasons.append("reference_source_not_ground_truth_allowed")
|
|
if not evidence["reference_task"]:
|
|
reasons.append("reference_validation_task_required")
|
|
elif not cls._reference_task_is_approved(evidence):
|
|
reasons.append("reference_task_authority_not_approved")
|
|
|
|
return DatasetConsumptionDecision(
|
|
eligible=not reasons,
|
|
purpose=purpose,
|
|
fixture_legacy_exception=False,
|
|
reasons=tuple(sorted(set(reasons))),
|
|
evidence=evidence,
|
|
)
|
|
|
|
@classmethod
|
|
def assert_eligible(
|
|
cls,
|
|
dataset: Any,
|
|
*,
|
|
purpose: DatasetConsumptionPurpose,
|
|
fixture_mode: bool = False,
|
|
reference_task: str | None = None,
|
|
) -> DatasetConsumptionDecision:
|
|
decision = cls.evaluate(
|
|
dataset,
|
|
purpose=purpose,
|
|
fixture_mode=fixture_mode,
|
|
reference_task=reference_task,
|
|
)
|
|
if decision.eligible:
|
|
return decision
|
|
code = "DATASET_QUARANTINED" if any(
|
|
reason in {"dataset_status_unsafe", "dataset_quarantined", "validation_failed", "source_snapshot_unsafe"}
|
|
for reason in decision.reasons
|
|
) else "DATASET_PROVENANCE_INCOMPLETE"
|
|
raise AppError(
|
|
code=code,
|
|
message="Dataset cannot be consumed until its provenance and validation gates are satisfied.",
|
|
status_code=409,
|
|
details={
|
|
"dataset_id": decision.evidence["dataset_id"],
|
|
"purpose": purpose,
|
|
"reasons": list(decision.reasons),
|
|
"fixture_legacy_exception": decision.fixture_legacy_exception,
|
|
},
|
|
)
|
|
|
|
@classmethod
|
|
def eligible_for_authoritative_coverage(cls, dataset: Any) -> bool:
|
|
"""Return false instead of raising so coverage can report a gap safely."""
|
|
|
|
return cls.evaluate(dataset, purpose="authoritative_coverage").eligible
|
|
|
|
@staticmethod
|
|
def _reference_task_is_approved(evidence: Mapping[str, Any]) -> bool:
|
|
"""Require task-specific primary authority or an explicit zone approval.
|
|
|
|
``classification=authoritative`` is deliberately not a blanket
|
|
permission to serve as building truth. A source may be authoritative
|
|
for an address lifecycle or elevation product while remaining only
|
|
corroborative for footprint QA. A regional product that is marked
|
|
``*_pending_contract`` similarly remains blocked until an operator
|
|
records a narrow product-and-zone approval in its server-owned policy.
|
|
"""
|
|
|
|
task = str(evidence.get("reference_task") or "").strip().lower()
|
|
authority = DatasetConsumptionGate._normalise(
|
|
DatasetConsumptionGate._mapping(evidence.get("source_validation_authority")).get(task)
|
|
)
|
|
if authority == "primary":
|
|
return True
|
|
if authority not in {"approved", "approved_product_zone"}:
|
|
return False
|
|
|
|
approvals = DatasetConsumptionGate._mapping(evidence.get("source_reference_validation_approvals"))
|
|
approval = DatasetConsumptionGate._mapping(approvals.get(task))
|
|
if approval.get("approved") is not True:
|
|
return False
|
|
|
|
source_key = str(evidence.get("source_key") or "").strip().lower()
|
|
source_scope = DatasetConsumptionGate._mapping(evidence.get("source_authority_scope"))
|
|
source_zone = str(source_scope.get("zone") or source_scope.get("scope") or "").strip()
|
|
approved_keys = approval.get("source_keys")
|
|
approved_zones = approval.get("zones")
|
|
if not isinstance(approved_keys, list) or source_key not in {
|
|
str(value).strip().lower() for value in approved_keys
|
|
}:
|
|
return False
|
|
if not isinstance(approved_zones, list) or source_zone not in {
|
|
str(value).strip() for value in approved_zones
|
|
}:
|
|
return False
|
|
return True
|