Initial deploy setup
deploy / deploy (push) Canceled after 0s

This commit is contained in:
Jens
2026-07-21 14:00:00 +02:00
commit b8091e59bd
285 changed files with 27854 additions and 0 deletions
+349
View File
@@ -0,0 +1,349 @@
from __future__ import annotations
from collections.abc import Iterable
from dataclasses import dataclass
from datetime import datetime, timedelta
from django.utils import timezone
from apps.sources.models import Source, SourcePolicyReview, SourceRun
from .policy import create_policy_review
@dataclass(frozen=True)
class SourceHealth:
source_id: int
source_name: str
source_type: str
source_status: str
monitored_runs: int
success_ratio: float
avg_latency_ms: float | None
error_counts: dict[str, int]
http_status_counts: dict[str, int]
extracted_count: int
updated_count: int
duplicate_count: int
last_parser: str | None
last_parser_warnings: int
last_health_action: str | None
last_health_reason: str | None
SOURCE_HEALTH_RUN_WINDOW = 30
SOURCE_HEALTH_MIN_RUNS_FOR_TEMPORARY_QUARANTINE = 6
SOURCE_HEALTH_TEMPORARY_ERROR_RATIO = 0.7
SOURCE_HEALTH_PARSER_DRIFT_SUCCESS_STREAK = 3
SOURCE_HEALTH_PARSER_MIN_WARNINGS = 2
SOURCE_HEALTH_PARSER_DRIFT_EXTRACTED_MAX = 0
SOURCE_HEALTH_RECOVERY_COOLDOWN_HOURS = 12
TEMPORARY_ERROR_CATEGORIES = {
"timeout",
"rate_limited",
"FetchTimeoutError",
"FetchError",
"unexpected",
"NetworkError",
}
POLICY_ERROR_CATEGORIES = {"policy"}
def _normalize_metrics(raw: object) -> dict[str, object]:
if isinstance(raw, dict):
return raw
return {}
def _warnings_from_run(run: SourceRun) -> list[str]:
metrics = _normalize_metrics(run.metrics)
warnings = metrics.get("warnings", [])
if not isinstance(warnings, list):
return []
return [str(item) for item in warnings if isinstance(item, str)]
def _parser_from_run(run: SourceRun) -> str | None:
metrics = _normalize_metrics(run.metrics)
parser = metrics.get("parser")
if isinstance(parser, str) and parser:
return parser
return None
def _int(value: object) -> int:
try:
return int(value)
except (TypeError, ValueError):
return 0
def _float(value: object) -> float:
try:
return float(value)
except (TypeError, ValueError):
return 0.0
def _to_iso(dt: datetime | None) -> str | None:
if dt is None:
return None
return dt.isoformat()
def _from_iso(value: str | None) -> datetime | None:
if not value:
return None
try:
dt = datetime.fromisoformat(value)
except ValueError:
return None
if timezone.is_naive(dt):
return timezone.make_aware(dt)
return dt
def _health_metadata(source: Source) -> dict[str, object]:
metadata = source.metadata
if not isinstance(metadata, dict):
return {}
health = metadata.get("source_health")
return health if isinstance(health, dict) else {}
def _set_health_metadata(source: Source, health_data: dict[str, object]) -> None:
metadata = source.metadata
if not isinstance(metadata, dict):
metadata = {}
metadata["source_health"] = health_data
source.metadata = metadata
source.save(update_fields=["metadata", "updated_at"])
def _run_counts(runs: Iterable[SourceRun]) -> tuple[dict[str, int], dict[str, int]]:
error_counts: dict[str, int] = {}
http_status_counts: dict[str, int] = {}
for run in runs:
if run.status != SourceRun.Status.SUCCESS:
category = run.error_category or "unknown"
error_counts[category] = error_counts.get(category, 0) + 1
if run.http_status:
status = str(run.http_status)
http_status_counts[status] = http_status_counts.get(status, 0) + 1
return error_counts, http_status_counts
def _last_parser_output(runs: list[SourceRun]) -> tuple[str | None, int]:
for run in runs:
if run.status != SourceRun.Status.SUCCESS:
continue
parser = _parser_from_run(run)
if parser:
warnings = _warnings_from_run(run)
return parser, len(warnings)
return None, 0
def _latency_ms(runs: list[SourceRun]) -> float | None:
latencies = []
for run in runs:
if run.finished_at is None or run.started_at is None:
continue
latencies.append(max(0.0, (run.finished_at - run.started_at).total_seconds() * 1000))
if not latencies:
return None
return sum(latencies) / len(latencies)
def _is_parser_drift_run(run: SourceRun) -> bool:
if run.status != SourceRun.Status.SUCCESS:
return False
if _int(run.extracted_count) > SOURCE_HEALTH_PARSER_DRIFT_EXTRACTED_MAX:
return False
if _int(run.created_count) > SOURCE_HEALTH_PARSER_DRIFT_EXTRACTED_MAX:
return False
if _int(run.updated_count) > SOURCE_HEALTH_PARSER_DRIFT_EXTRACTED_MAX:
return False
if _parser_from_run(run) in {None, "not-modified"}:
return False
warnings = _warnings_from_run(run)
return len(warnings) >= SOURCE_HEALTH_PARSER_MIN_WARNINGS
def _determine_health_action(runs: list[SourceRun]) -> tuple[str | None, str | None]:
recent_runs = runs[:SOURCE_HEALTH_RUN_WINDOW]
considered_failures = [
run for run in recent_runs if run.status in {SourceRun.Status.FAILED, SourceRun.Status.SKIPPED}
]
if any(
run.error_category in POLICY_ERROR_CATEGORIES
for run in considered_failures[:3]
if run.error_category
):
return "quarantine", "Herhaald beleid-/securityprobleem in bronruns."
if len(considered_failures) >= SOURCE_HEALTH_MIN_RUNS_FOR_TEMPORARY_QUARANTINE:
temporary_count = sum(
1
for run in considered_failures
if (run.error_category or "") in TEMPORARY_ERROR_CATEGORIES
)
ratio = _float(temporary_count) / _float(len(considered_failures))
if ratio >= SOURCE_HEALTH_TEMPORARY_ERROR_RATIO:
return "quarantine", "Herhaald tijdelijk foutgedrag tijdens bronruns."
streak = 0
for run in recent_runs:
if _is_parser_drift_run(run):
streak += 1
if streak >= SOURCE_HEALTH_PARSER_DRIFT_SUCCESS_STREAK:
return (
"quarantine",
"Parserdrift vermoed: opeenvolgende succesvolle runs met minimale output.",
)
continue
streak = 0
return None, None
def collect_source_health(*, runs_to_consider: int = SOURCE_HEALTH_RUN_WINDOW) -> list[SourceHealth]:
sources = Source.objects.order_by("name").all()
rows: list[SourceHealth] = []
for source in sources:
run_queryset = SourceRun.objects.filter(source=source).order_by("-started_at")
runs = list(run_queryset[:runs_to_consider])
monitored_runs = len(runs)
success_runs = [run for run in runs if run.status == SourceRun.Status.SUCCESS]
success_ratio = _float(len(success_runs) / monitored_runs) if monitored_runs else 0.0
avg_latency_ms = _latency_ms(runs)
error_counts, http_status_counts = _run_counts(runs)
extracted_count = sum(_int(run.extracted_count) for run in runs)
updated_count = sum(_int(run.updated_count) for run in runs)
duplicate_count = sum(_int(run.duplicate_count) for run in runs)
last_parser, last_warnings = _last_parser_output(runs)
action, reason = _determine_health_action(runs)
rows.append(
SourceHealth(
source_id=source.pk,
source_name=source.name,
source_type=source.source_type,
source_status=source.status,
monitored_runs=monitored_runs,
success_ratio=success_ratio,
avg_latency_ms=avg_latency_ms,
error_counts=error_counts,
http_status_counts=http_status_counts,
extracted_count=extracted_count,
updated_count=updated_count,
duplicate_count=duplicate_count,
last_parser=last_parser,
last_parser_warnings=last_warnings,
last_health_action=action,
last_health_reason=reason,
)
)
return rows
def evaluate_source_health(*, now: datetime | None = None) -> dict[str, int]:
now = now or timezone.now()
rows = [row for row in collect_source_health() if row.source_status in {Source.Status.ACTIVE, Source.Status.TRIAL}]
counts = {"evaluated": len(rows), "quarantined": 0}
for row in rows:
if row.last_health_action != "quarantine":
continue
counts["evaluated"] += 1
source = Source.objects.get(pk=row.source_id)
source.status = Source.Status.QUARANTINED
source.policy = Source.Policy.DENY
source.policy_reason = row.last_health_reason or "Bronhealth detecteert instabiele bron"
_set_health_metadata(
source,
{
"state": "quarantined",
"quarantine_reason": source.policy_reason,
"quarantined_at": _to_iso(now),
"recovery_due_at": _to_iso(
now + timedelta(hours=SOURCE_HEALTH_RECOVERY_COOLDOWN_HOURS)
),
"canary_started": False,
},
)
source.save(update_fields=["status", "policy", "policy_reason", "updated_at"])
latest_review = source.policy_reviews.order_by("-created_at").first()
if latest_review is None or latest_review.decision != SourcePolicyReview.Decision.DENY:
create_policy_review(
source,
actor=None,
decision=SourcePolicyReview.Decision.DENY,
reason=source.policy_reason,
scope=SourcePolicyReview.Scope.SOURCE,
notes="Automatische bronhealth",
)
counts["quarantined"] += 1
return counts
def canary_recovery_sources(*, now: datetime | None = None) -> list[Source]:
now = now or timezone.now()
sources: list[Source] = []
for source in Source.objects.filter(status=Source.Status.QUARANTINED):
health = _health_metadata(source)
if health.get("state") != "quarantined":
continue
if bool(health.get("canary_started", False)):
continue
recovery_due = _from_iso(health.get("recovery_due_at") if isinstance(health, dict) else None)
if recovery_due and recovery_due > now:
continue
sources.append(source)
return sources
def start_health_canary(source: Source, *, now: datetime | None = None) -> None:
now = now or timezone.now()
health = _health_metadata(source)
latest_review = source.policy_reviews.order_by("-created_at").first()
if latest_review is None or latest_review.decision != SourcePolicyReview.Decision.TRIAL:
create_policy_review(
source,
actor=None,
decision=SourcePolicyReview.Decision.TRIAL,
reason="Automatische bronrecovery via canary",
scope=SourcePolicyReview.Scope.SOURCE,
notes="Canaryherstel",
)
source.status = Source.Status.TRIAL
source.policy = Source.Policy.REVIEW
source.policy_reason = "Bronherstel via geautomatiseerde canary"
source.next_run_at = now
_set_health_metadata(
source,
{
**health,
"state": "canary_in_progress",
"canary_started": True,
"canary_started_at": _to_iso(now),
},
)
source.save(update_fields=["status", "policy", "policy_reason", "next_run_at", "updated_at"])