Fix release blockers and deployment build
deploy / deploy (push) Canceled after 0s

This commit is contained in:
Jens
2026-07-21 21:22:29 +02:00
parent b8091e59bd
commit a4eced8be5
64 changed files with 1011 additions and 675 deletions
+15 -11
View File
@@ -4,11 +4,11 @@ import ipaddress
import json
import re
from dataclasses import dataclass
from datetime import datetime, timezone
from datetime import UTC, datetime
from urllib.parse import urljoin, urlsplit
from xml.etree import ElementTree
from bs4 import BeautifulSoup
from defusedxml import ElementTree
from django.db import transaction
from apps.sources.adapters.email_alert import EmailAlertAdapter
@@ -166,15 +166,14 @@ def persist_discovery_candidates(candidates: list[SourceCandidate]) -> tuple[int
evidence = metadata.get("discovery", [])
if not isinstance(evidence, list):
evidence = []
if not any(item.get("url") == candidate.url for item in evidence if isinstance(item, dict)):
if not any(
item.get("url") == candidate.url for item in evidence if isinstance(item, dict)
):
evidence.append(provenance)
metadata["discovery"] = evidence[-20:]
metadata["discovered_at"] = now
source.metadata = metadata
updated_fields.append("metadata")
else:
skipped += 1
if not source.name:
source.name = _candidate_name(candidate)
updated_fields.append("name")
@@ -223,19 +222,24 @@ def _discover_html(
for anchor in soup.find_all("a", href=True):
label = " ".join(anchor.get_text(" ", strip=True).split())
raw_url = urljoin(base_url, str(anchor["href"]))
candidate_host = _hostname(raw_url)
same_domain = bool(base_domain and domain_matches(candidate_host, base_domain))
candidate = _build_candidate(
raw_url=urljoin(base_url, str(anchor["href"])),
raw_url=raw_url,
base_domain=base_domain,
source_type=Source.Type.EMPLOYER,
label=label,
reason="career-link",
discovered_from="html",
confidence=0.9,
allow_off_domain=False,
confidence=0.9 if same_domain else 0.7,
allow_off_domain=True,
)
if not candidate:
continue
if not CAREER_PATTERN.search(label) and not CAREER_PATTERN.search(urlsplit(candidate.url).path):
if not CAREER_PATTERN.search(label) and not CAREER_PATTERN.search(
urlsplit(candidate.url).path
):
continue
candidates.append(candidate)
@@ -388,4 +392,4 @@ def _to_domain_root_url(url: str) -> str:
def _utc_now() -> str:
return datetime.now(timezone.utc).isoformat()
return datetime.now(UTC).isoformat()
+12 -3
View File
@@ -1,10 +1,11 @@
from __future__ import annotations
import hashlib
from datetime import datetime, timezone as utc
from email.utils import parsedate_to_datetime
from collections.abc import Mapping
from dataclasses import dataclass
from datetime import datetime
from datetime import timezone as utc
from email.utils import parsedate_to_datetime
from urllib.parse import urljoin
import httpx
@@ -13,7 +14,7 @@ from django.conf import settings
from apps.sources.models import Source
from .policy import assess_url
from .url_security import validate_public_url
from .url_security import ValidatedUrl, validate_public_url
ALLOWED_CONTENT_TYPES = (
"text/html",
@@ -124,12 +125,19 @@ def fetch_url(
headers=headers,
)
current_url = url
previous_validation: ValidatedUrl | None = None
try:
for _ in range(settings.FETCHER_MAX_REDIRECTS + 1):
validation = validate_public_url(
current_url,
allow_nonstandard_ports=settings.FETCHER_ALLOW_NONSTANDARD_PORTS,
)
if (
previous_validation is not None
and validation.hostname == previous_validation.hostname
and not set(validation.addresses).intersection(previous_validation.addresses)
):
raise FetchError("DNS-rebindcontrole faalde bij redirect naar dezelfde host.")
response = http_client.get(current_url, headers=headers)
post_validation = validate_public_url(
str(response.url) if response.url else current_url,
@@ -145,6 +153,7 @@ def fetch_url(
next_decision = assess_url(current_url, source=source)
if not next_decision.allowed:
raise PolicyBlockedError(next_decision.reason)
previous_validation = validation
continue
if response.status_code == 304:
return FetchedDocument(url, current_url, 304, dict(response.headers), b"")
+14 -4
View File
@@ -174,7 +174,9 @@ def _is_parser_drift_run(run: SourceRun) -> bool:
def _determine_health_action(runs: list[SourceRun]) -> tuple[str | None, str | None]:
recent_runs = runs[:SOURCE_HEALTH_RUN_WINDOW]
considered_failures = [
run for run in recent_runs if run.status in {SourceRun.Status.FAILED, SourceRun.Status.SKIPPED}
run
for run in recent_runs
if run.status in {SourceRun.Status.FAILED, SourceRun.Status.SKIPPED}
]
if any(
@@ -209,7 +211,9 @@ def _determine_health_action(runs: list[SourceRun]) -> tuple[str | None, str | N
return None, None
def collect_source_health(*, runs_to_consider: int = SOURCE_HEALTH_RUN_WINDOW) -> list[SourceHealth]:
def collect_source_health(
*, runs_to_consider: int = SOURCE_HEALTH_RUN_WINDOW
) -> list[SourceHealth]:
sources = Source.objects.order_by("name").all()
rows: list[SourceHealth] = []
@@ -256,7 +260,11 @@ def collect_source_health(*, runs_to_consider: int = SOURCE_HEALTH_RUN_WINDOW) -
def evaluate_source_health(*, now: datetime | None = None) -> dict[str, int]:
now = now or timezone.now()
rows = [row for row in collect_source_health() if row.source_status in {Source.Status.ACTIVE, Source.Status.TRIAL}]
rows = [
row
for row in collect_source_health()
if row.source_status in {Source.Status.ACTIVE, Source.Status.TRIAL}
]
counts = {"evaluated": len(rows), "quarantined": 0}
for row in rows:
@@ -309,7 +317,9 @@ def canary_recovery_sources(*, now: datetime | None = None) -> list[Source]:
if bool(health.get("canary_started", False)):
continue
recovery_due = _from_iso(health.get("recovery_due_at") if isinstance(health, dict) else None)
recovery_due = _from_iso(
health.get("recovery_due_at") if isinstance(health, dict) else None
)
if recovery_due and recovery_due > now:
continue
+6 -13
View File
@@ -18,9 +18,9 @@ from apps.jobs.services.pipeline import process_raw_document
from apps.sources.models import RawDocument, Source, SourcePolicyReview, SourceRun
from apps.sources.services.canonicalize import canonicalize_url
from apps.sources.services.fetcher import (
FetchedDocument,
FetchError,
FetchTimeoutError,
FetchedDocument,
PolicyBlockedError,
RateLimitedError,
fetch_url,
@@ -88,11 +88,7 @@ def _mode_source_name(domain: str, *, mode: str) -> str:
def _ensure_manual_review(source: Source, actor) -> None:
review = source.latest_policy_review
if (
review
and not review.is_expired
and review.decision == SourcePolicyReview.Decision.ALLOW
):
if review and not review.is_expired and review.decision == SourcePolicyReview.Decision.ALLOW:
return
create_policy_review(
source,
@@ -120,9 +116,7 @@ def _ensure_manual_source(*, domain: str, source_url: str, actor, mode: str) ->
source.base_url = source_url
source.status = Source.Status.CANDIDATE
source.policy = Source.Policy.ALLOW
source.save(
update_fields=["name", "base_url", "status", "policy", "updated_at"]
)
source.save(update_fields=["name", "base_url", "status", "policy", "updated_at"])
_ensure_manual_review(source, actor=actor)
return source
@@ -175,7 +169,7 @@ def _run_pipeline(document: RawDocument, *, source_run: SourceRun) -> ManualImpo
source = document.source
source_name = source.name if source else ""
warnings: list[str] = list(metrics.get("warnings", []))
warnings_count = len(warnings)
len(warnings)
source_run.finish(
SourceRun.Status.SUCCESS,
http_status=document.source_run.http_status if document.source_run else None,
@@ -186,9 +180,8 @@ def _run_pipeline(document: RawDocument, *, source_run: SourceRun) -> ManualImpo
metrics={"parser": metrics["parser"], "warnings": warnings},
)
jobs = _build_jobs_from_document(document)
if metrics["created"] == 0 and metrics["updated"] == 0:
if not warnings:
warnings.append("De bron leverde geen herkenbare vacaturedata op.")
if metrics["created"] == 0 and metrics["updated"] == 0 and not warnings:
warnings.append("De bron leverde geen herkenbare vacaturedata op.")
if not warnings:
# keep stable, machine-readable payload shape
warnings = []
+15 -5
View File
@@ -79,18 +79,28 @@ def _check_review_gate(source: Source) -> PolicyDecision | None:
if not review:
return PolicyDecision(False, Source.Policy.REVIEW, "Review vereist")
if review.decision == SourcePolicyReview.Decision.DENY:
return PolicyDecision(False, Source.Policy.DENY, review.reason or "Review blokkeert bron")
return PolicyDecision(
False, Source.Policy.DENY, review.reason or "Review blokkeert bron"
)
if review.decision == SourcePolicyReview.Decision.PAUSE:
return PolicyDecision(False, Source.Policy.REVIEW, review.reason or "Review vraagt pauze")
return PolicyDecision(
False, Source.Policy.REVIEW, review.reason or "Review vraagt pauze"
)
return None
if source.policy == Source.Policy.ALLOW:
if not review:
return PolicyDecision(False, Source.Policy.REVIEW, "Review ontbreekt voor actief beleid")
return PolicyDecision(
False, Source.Policy.REVIEW, "Review ontbreekt voor actief beleid"
)
if review.decision == SourcePolicyReview.Decision.DENY:
return PolicyDecision(False, Source.Policy.DENY, review.reason or "Review blokkeert bron")
return PolicyDecision(
False, Source.Policy.DENY, review.reason or "Review blokkeert bron"
)
if review.decision == SourcePolicyReview.Decision.PAUSE:
return PolicyDecision(False, Source.Policy.REVIEW, review.reason or "Review vraagt pauze")
return PolicyDecision(
False, Source.Policy.REVIEW, review.reason or "Review vraagt pauze"
)
return None
+23 -16
View File
@@ -10,6 +10,7 @@ from django.conf import settings
from django.utils import timezone
from apps.sources.models import Source, SourceRobotsCache
from .url_security import UnsafeUrlError, validate_public_url
ALLOW = "allow"
@@ -30,7 +31,11 @@ def _origin_for(url: str) -> str:
if not host:
raise ValueError("Host ontbreekt voor robotscontrole.")
port = parts.port
if (parts.scheme == "http" and port == 80) or (parts.scheme == "https" and port == 443) or not port:
if (
(parts.scheme == "http" and port == 80)
or (parts.scheme == "https" and port == 443)
or not port
):
netloc = host
else:
netloc = f"{host}:{port}"
@@ -61,10 +66,7 @@ def _rules_from_text(text: str) -> dict[str, dict[str, list[str]]]:
key_lower = key.lower()
if key_lower == "user-agent":
token = value.lower()
if token:
active_agents = {token}
else:
active_agents = set()
active_agents = {token} if token else set()
continue
if key_lower not in {ALLOW, DISALLOW}:
continue
@@ -85,7 +87,7 @@ def _pick_rules(rules: dict[str, dict[str, list[str]]], user_agent: str) -> dict
selected = {ALLOW: [], DISALLOW: []}
for agent, values in rules.items():
if agent == "*" or agent and agent in normalized:
if agent == "*" or (agent and agent in normalized):
selected[ALLOW].extend(values[ALLOW])
selected[DISALLOW].extend(values[DISALLOW])
@@ -95,10 +97,14 @@ def _pick_rules(rules: dict[str, dict[str, list[str]]], user_agent: str) -> dict
def _longest_prefix(path: str, rules: Iterable[str]) -> int:
return max((len(rule.rstrip("/")) for rule in rules if rule and path.startswith(rule)), default=0)
return max(
(len(rule.rstrip("/")) for rule in rules if rule and path.startswith(rule)), default=0
)
def _evaluate_path(path: str, user_agent: str, rules: dict[str, dict[str, list[str]]]) -> RobotsDecision:
def _evaluate_path(
path: str, user_agent: str, rules: dict[str, dict[str, list[str]]]
) -> RobotsDecision:
selected = _pick_rules(rules, user_agent=user_agent)
allow_len = _longest_prefix(path, selected[ALLOW])
disallow_len = _longest_prefix(path, selected[DISALLOW])
@@ -184,12 +190,7 @@ def _persist_cache(
},
)[0]
if status_code in {404, 410}:
rules = {}
elif status_code >= 400:
rules = {}
else:
rules = _rules_from_text(content)
rules = {} if status_code in {404, 410} or status_code >= 400 else _rules_from_text(content)
return SourceRobotsCache.objects.update_or_create(
origin=origin,
@@ -257,12 +258,18 @@ def assess_robots(
return RobotsDecision(False, f"Robotscontrole mislukt: {exc}")
except httpx.HTTPError as exc:
if stale is not None:
return _evaluate_path(path, user_agent=user_agent or getattr(settings, "FETCHER_USER_AGENT", ""), rules=_build_ruleset(stale))
return _evaluate_path(
path,
user_agent=user_agent or getattr(settings, "FETCHER_USER_AGENT", ""),
rules=_build_ruleset(stale),
)
return RobotsDecision(True, f"Robotscontrole tijdelijk niet beschikbaar: {exc}")
if cache.error:
return RobotsDecision(True, cache.error)
rules = _build_ruleset(cache)
decision = _evaluate_path(path, user_agent=user_agent or getattr(settings, "FETCHER_USER_AGENT", ""), rules=rules)
decision = _evaluate_path(
path, user_agent=user_agent or getattr(settings, "FETCHER_USER_AGENT", ""), rules=rules
)
return decision
+13 -13
View File
@@ -76,14 +76,10 @@ def calculate_failure_backoff_seconds(
def calculate_success_jitter_seconds(source: Source) -> int:
return calculate_jitter_seconds(
source, max_seconds=settings.SOURCE_SUCCESS_JITTER_SECONDS
)
return calculate_jitter_seconds(source, max_seconds=settings.SOURCE_SUCCESS_JITTER_SECONDS)
def acquire_source_lease(
*, source_id: int, worker_token: str, now=None
) -> SourceLease | None:
def acquire_source_lease(*, source_id: int, worker_token: str, now=None) -> SourceLease | None:
now = now or timezone.now()
with transaction.atomic():
source = Source.objects.select_for_update().get(pk=source_id)
@@ -101,9 +97,8 @@ def acquire_source_lease(
return None
lease = SourceLease.objects.select_for_update().filter(source=source).first()
if lease is not None and not lease.is_expired:
if lease.token != worker_token:
return None
if lease is not None and not lease.is_expired and lease.token != worker_token:
return None
active_leases = SourceLease.objects.select_for_update().filter(
source__domain=domain, expires_at__gt=now
@@ -118,7 +113,10 @@ def acquire_source_lease(
lease.token = worker_token
lease.worker_id = worker_token
lease.expires_at = now + timedelta(seconds=_lease_ttl_seconds())
lease.save(update_fields=["token", "worker_id", "expires_at", "updated_at"])
if lease.pk is None:
lease.save()
else:
lease.save(update_fields=["token", "worker_id", "expires_at", "updated_at"])
return lease
@@ -129,9 +127,11 @@ def release_source_lease(*, source_id: int, worker_token: str, now=None) -> bool
if not source.domain:
return False
lease = SourceLease.objects.select_for_update().filter(
source=source, token=worker_token
).first()
lease = (
SourceLease.objects.select_for_update()
.filter(source=source, token=worker_token)
.first()
)
if not lease:
return False