Fix release blockers and deployment build
deploy / deploy (push) Canceled after 0s

This commit is contained in:
Jens
2026-07-21 21:22:29 +02:00
parent b8091e59bd
commit a4eced8be5
64 changed files with 1011 additions and 675 deletions
+8 -3
View File
@@ -1,6 +1,7 @@
from __future__ import annotations
from dataclasses import dataclass
from django.core.cache import cache
from django.http import HttpRequest
from django.utils import timezone
@@ -17,7 +18,9 @@ def _identity_for_user(request: HttpRequest) -> str:
user = getattr(request, "user", None)
if user and getattr(user, "is_authenticated", False):
return f"user:{user.pk}"
username = (request.POST.get("username", "") if request.method == "POST" else "").strip().lower()
username = (
(request.POST.get("username", "") if request.method == "POST" else "").strip().lower()
)
return f"anon:{username or _client_ip(request)}"
@@ -47,7 +50,7 @@ def is_rate_limited(
now = timezone.now().timestamp()
identity = _identity_for_user(request)
block_until = cache.get(_block_key(namespace, identity))
if block_until and isinstance(block_until, (int, float)) and block_until > now:
if block_until and isinstance(block_until, int | float) and block_until > now:
return RateLimitState(True, max(0, int(block_until - now)), 0)
if cache.get(_attempts_key(namespace, identity), 0) >= max_attempts:
@@ -84,7 +87,9 @@ def register_rate_limit_failure(
attempts = int(cache.get(_attempts_key(namespace, identity), 0)) + 1
if attempts >= max_attempts:
now = timezone.now().timestamp()
cache.set(_block_key(namespace, identity), now + max(block_seconds, 1), timeout=block_seconds)
cache.set(
_block_key(namespace, identity), now + max(block_seconds, 1), timeout=block_seconds
)
cache.delete(_attempts_key(namespace, identity))
return RateLimitState(True, max(block_seconds, 1), attempts)
+5 -4
View File
@@ -1,20 +1,20 @@
from __future__ import annotations
from django.conf import settings
from django.contrib.auth.mixins import LoginRequiredMixin
from django.contrib import messages
from django.contrib.auth.mixins import LoginRequiredMixin
from django.contrib.auth.views import LoginView
from django.db.models import Count, Q
from django.http import JsonResponse
from django.utils import timezone
from django.views.generic import TemplateView
from apps.core.rate_limit import clear_rate_limit, is_rate_limited, register_rate_limit_failure
from apps.jobs.models import Application, JobPosting, ScoreRun
from apps.sources.models import Source
from apps.core.rate_limit import clear_rate_limit, is_rate_limited, register_rate_limit_failure
from apps.sources.services.health import collect_source_health
from .health import readiness
from apps.sources.services.health import collect_source_health
class SecurityAwareLoginView(LoginView):
@@ -39,7 +39,8 @@ class SecurityAwareLoginView(LoginView):
request,
(
"Te veel inlogpogingen op dit account. Wacht "
f"{self._remaining_minutes(state.remaining_seconds)} en probeer daarna opnieuw."
f"{self._remaining_minutes(state.remaining_seconds)} en probeer daarna "
"opnieuw."
),
)
return self.form_invalid(self.get_form())
+2 -2
View File
@@ -1,15 +1,15 @@
from django.contrib import admin
from .models import (
AiAnalysisCache,
Application,
ApplicationTimelineEvent,
Employer,
Feedback,
FieldProvenance,
AiAnalysisCache,
GeocodeLocationLookup,
JobPosting,
JobSourceAlias,
GeocodeLocationLookup,
JobVersion,
ScoreRun,
)
@@ -2,13 +2,13 @@ from __future__ import annotations
from pathlib import Path
from urllib.parse import urlparse
from django.db.models import Q
from django.core.management.base import BaseCommand, CommandError
from django.db.models import Q
from apps.jobs.models import JobPosting
from apps.jobs.services.scoring import rescore_jobs_with_profiles
from apps.jobs.services.geocoding import import_csv_geodata, validate_csv_geodata
from apps.jobs.services.scoring import rescore_jobs_with_profiles
class Command(BaseCommand):
+16 -9
View File
@@ -1,7 +1,7 @@
from __future__ import annotations
from decimal import Decimal
import uuid
from decimal import Decimal
from django.conf import settings
from django.db import models
@@ -193,8 +193,12 @@ class GeocodeLocationLookup(TimeStampedModel):
class Meta:
ordering = ["query_kind", "query_value", "municipality"]
indexes = [
models.Index(fields=["query_kind", "query_value"]),
models.Index(fields=["source_name", "source_version"]),
models.Index(
fields=["query_kind", "query_value"], name="geocode_lookup_kind_query_idx"
),
models.Index(
fields=["source_name", "source_version"], name="geocode_lookup_source_idx"
),
]
constraints = [
models.UniqueConstraint(
@@ -272,7 +276,7 @@ class AiAnalysisCache(TimeStampedModel):
status = models.CharField(max_length=16, choices=Status.choices)
error_category = models.CharField(max_length=120, blank=True)
summary_nl = models.TextField(blank=True)
features = models.JSONField(default=dict, blank=True)
features = models.JSONField(default=dict)
warnings = models.JSONField(default=list, blank=True)
class Meta:
@@ -284,8 +288,11 @@ class AiAnalysisCache(TimeStampedModel):
)
]
indexes = [
models.Index(fields=["content_hash"]),
models.Index(fields=["model_name", "prompt_version", "schema_version"]),
models.Index(fields=["content_hash"], name="jobs_aianal_cache_content_idx"),
models.Index(
fields=["model_name", "prompt_version", "schema_version"],
name="jobs_aianal_cache_model_idx",
),
]
def __str__(self) -> str:
@@ -364,9 +371,9 @@ class ApplicationTimelineEvent(TimeStampedModel):
class Meta:
ordering = ["-created_at"]
indexes = [
models.Index(fields=["application", "created_at"]),
models.Index(fields=["user", "created_at"]),
models.Index(fields=["application", "created_at"], name="jobs_app_timeline_app_idx"),
models.Index(fields=["user", "created_at"], name="jobs_app_timeline_user_idx"),
]
def __str__(self) -> str:
return f"{self.application}{self.event_type}"
return f"{self.application}{self.event_type}"
+15 -5
View File
@@ -57,7 +57,13 @@ def _schema() -> dict[str, Any]:
"seniority": {"type": "string"},
"evidence": {"type": "array", "items": {"type": "string"}},
},
"required": ["support_ratio", "consultancy_ratio", "travel_ratio", "seniority", "evidence"],
"required": [
"support_ratio",
"consultancy_ratio",
"travel_ratio",
"seniority",
"evidence",
],
},
"warnings": {"type": "array", "items": {"type": "string"}},
},
@@ -73,8 +79,8 @@ def _normalize_text(title: str, description: str) -> str:
def _coerce_ratio(name: str, value: Any) -> float:
try:
ratio = float(value)
except (TypeError, ValueError):
raise ValueError(f"ratio:{name}")
except (TypeError, ValueError) as exc:
raise ValueError(f"ratio:{name}") from exc
if not (MIN_FEATURE_VALUE <= ratio <= MAX_FEATURE_VALUE):
raise ValueError(f"ratio:{name}")
return round(ratio, 6)
@@ -108,7 +114,9 @@ def _parse_and_validate_payload(payload: Any, *, title: str, description: str) -
evidence_values = features.get("evidence")
if not isinstance(evidence_values, list):
raise ValueError("evidence")
normalized_evidence = [normalize_token(item) for item in evidence_values if isinstance(item, str)]
normalized_evidence = [
normalize_token(item) for item in evidence_values if isinstance(item, str)
]
normalized_evidence = [item for item in normalized_evidence if item]
if not normalized_evidence:
raise ValueError("evidence")
@@ -289,7 +297,9 @@ def analyze_job_text(
)
try:
return _cache_get(content_hash=content_hash, model=model_name, prompt_version=prompt_version)
return _cache_get(
content_hash=content_hash, model=model_name, prompt_version=prompt_version
)
except AiUnavailable:
pass
+45 -21
View File
@@ -26,7 +26,7 @@ SNAPSHOT_VERSION = "1.0.0"
def _coerce_json_value(value: Any) -> Any:
if isinstance(value, (date,)):
if isinstance(value, date):
return value.isoformat()
if isinstance(value, Decimal):
return float(value)
@@ -85,7 +85,17 @@ def _latest_score_snapshot(job: JobPosting, user) -> dict[str, Any] | None:
run = (
ScoreRun.objects.filter(job=job, profile=profile)
.order_by("-created_at")
.only("score", "recommendation", "confidence", "components", "positives", "concerns", "hard_exclusions", "evidence", "profile_version")
.only(
"score",
"recommendation",
"confidence",
"components",
"positives",
"concerns",
"hard_exclusions",
"evidence",
"profile_version",
)
.first()
)
if not run:
@@ -121,7 +131,9 @@ def _build_application_snapshot_payload(application: Application) -> dict[str, A
}
def _record_timeline_event(*, application: Application, user, event_type: str, metadata: dict[str, Any] | None = None) -> ApplicationTimelineEvent:
def _record_timeline_event(
*, application: Application, user, event_type: str, metadata: dict[str, Any] | None = None
) -> ApplicationTimelineEvent:
payload: dict[str, Any] = {}
if metadata:
payload.update({key: value for key, value in metadata.items() if value is not None})
@@ -243,8 +255,11 @@ def track_application_changes(
)
events += 1
if (_normalize_str(previous.get("contact_name")) != _normalize_str(current.get("contact_name"))) or (
_normalize_str(previous.get("contact_email")) != _normalize_str(current.get("contact_email"))
if (
_normalize_str(previous.get("contact_name")) != _normalize_str(current.get("contact_name"))
) or (
_normalize_str(previous.get("contact_email"))
!= _normalize_str(current.get("contact_email"))
):
_record_timeline_event(
application=application,
@@ -266,16 +281,14 @@ def build_print_html(application: Application) -> str:
safe_rows = []
for row in timeline_rows:
safe_rows.append(
(
f"<tr>"
f"<td>{escape(row['timestamp'])}</td>"
f"<td>{escape(row['type'])}</td>"
f"<td>{escape(row['actor'])}</td>"
f"<td>{escape(row['from'])}</td>"
f"<td>{escape(row['to'])}</td>"
f"<td>{escape(row['note'])}</td>"
"</tr>"
)
f"<tr>"
f"<td>{escape(row['timestamp'])}</td>"
f"<td>{escape(row['type'])}</td>"
f"<td>{escape(row['actor'])}</td>"
f"<td>{escape(row['from'])}</td>"
f"<td>{escape(row['to'])}</td>"
f"<td>{escape(row['note'])}</td>"
"</tr>"
)
source_rows = []
@@ -284,20 +297,26 @@ def build_print_html(application: Application) -> str:
source_url = escape(str(source.get("url", "")))
source_rows.append(f"<li>{source_name}: {source_url}</li>")
source_section = "".join(source_rows) if source_rows else "<li>Niet beschikbaar</li>"
frozen_at = (
application.snapshot.get("frozen_at", "") if isinstance(application.snapshot, dict) else ""
)
snapshot_json = json.dumps(snapshot, indent=2, ensure_ascii=False, sort_keys=True)
return (
"<!doctype html><html><head><meta charset='utf-8'>"
"<title>Sollicitatiedossier</title><style>body{font-family:Arial,sans-serif;margin:24px}"
"table{border-collapse:collapse;width:100%}th,td{border:1px solid #ddd;padding:8px;text-align:left}"
"table{border-collapse:collapse;width:100%}"
"th,td{border:1px solid #ddd;padding:8px;text-align:left}"
"th{background:#f2f2f2} </style></head><body>"
f"<h1>{escape(job.original_title)}</h1>"
f"<p>Vacature: {escape(job.canonical_url)}</p>"
f"<p>Status: {escape(application.get_status_display())}</p>"
f"<p>Geslaagd op: {escape(str(application.snapshot.get('frozen_at') if isinstance(application.snapshot, dict) else ''))}</p>"
f"<p>Geslaagd op: {escape(str(frozen_at))}</p>"
f"<h2>Bronnen</h2><ul>{source_section}</ul>"
f"<h2>Timeline</h2><table><thead><tr><th>Tijd</th><th>Type</th><th>Actor</th><th>Van</th><th>Naar</th><th>Notitie</th></tr></thead><tbody>"
"<h2>Timeline</h2><table><thead><tr><th>Tijd</th><th>Type</th>"
"<th>Actor</th><th>Van</th><th>Naar</th><th>Notitie</th></tr></thead><tbody>"
f"{''.join(safe_rows)}</tbody></table>"
f"<h2>Snapshot</h2><pre>{escape(json.dumps(snapshot, indent=2, ensure_ascii=False, sort_keys=True))}</pre>"
f"<h2>Snapshot</h2><pre>{escape(snapshot_json)}</pre>"
"</body></html>"
)
@@ -315,7 +334,9 @@ def build_application_export(application: Application) -> tuple[bytes, str]:
"snapshot_version": SNAPSHOT_VERSION,
"exported_at": timezone.now().isoformat(),
"applied_at": application.applied_at.isoformat() if application.applied_at else None,
"follow_up_date": application.follow_up_date.isoformat() if application.follow_up_date else None,
"follow_up_date": application.follow_up_date.isoformat()
if application.follow_up_date
else None,
},
"snapshot": _coerce_json_value(application.snapshot),
"timeline": _timeline_dict_rows(application),
@@ -332,7 +353,10 @@ def build_application_export(application: Application) -> tuple[bytes, str]:
content_buffer = io.BytesIO()
with zipfile.ZipFile(content_buffer, "w", compression=zipfile.ZIP_DEFLATED) as zf:
zf.writestr("application.json", json.dumps(_coerce_json_value(payload), indent=2, ensure_ascii=False, sort_keys=True))
zf.writestr(
"application.json",
json.dumps(_coerce_json_value(payload), indent=2, ensure_ascii=False, sort_keys=True),
)
zf.writestr("timeline.csv", timeline_buffer.getvalue())
zf.writestr("application_print.html", build_print_html(application))
+1 -1
View File
@@ -3,10 +3,10 @@ from __future__ import annotations
from dataclasses import dataclass, field
from difflib import SequenceMatcher
from apps.sources.models import Source
from django.db.models import Q
from apps.jobs.models import JobPosting, JobSourceAlias
from apps.sources.models import Source
from .employer_resolution import EmployerResolutionDecision, resolve_direct_employer_match
from .normalization import CanonicalJobDraft, normalize_token
+1 -2
View File
@@ -27,8 +27,7 @@ class CommuteEstimator(Protocol):
name: str
version: str
def estimate(self, distance_km: float) -> CommuteEstimate:
...
def estimate(self, distance_km: float) -> CommuteEstimate: ...
@dataclass(frozen=True)
+40 -16
View File
@@ -9,7 +9,6 @@ from apps.sources.models import Source
from .normalization import CanonicalJobDraft, normalize_token
MERGE_THRESHOLD = 0.96
TITLE_MIN_THRESHOLD = 0.92
CONFLICT_TITLE_THRESHOLD = 0.70
@@ -46,11 +45,13 @@ def _weighted_similarity(
employer_domain_match: bool,
canonical_host_match: bool,
) -> float:
weights: list[tuple[float, float]] = [(title_score, 0.68), (employer_domain_match and 1.0 or 0.0, 0.12)]
weights: list[tuple[float, float]] = [(title_score, 0.68)]
if location_score is not None:
weights.append((location_score, 0.12))
if employer_score is not None:
weights.append((employer_score, 0.06))
if employer_domain_match:
weights.append((1.0, 0.12))
if canonical_host_match:
weights.append((1.0, 0.05))
total_weight = sum(weight for _, weight in weights)
@@ -62,7 +63,10 @@ def _weighted_similarity(
def _domain_match(value: str, candidate: str) -> bool:
if not value or not candidate:
return False
return _normalize_similarity(value).strip(".").lower() == _normalize_similarity(candidate).strip(".").lower()
return (
_normalize_similarity(value).strip(".").lower()
== _normalize_similarity(candidate).strip(".").lower()
)
def _host(value: str) -> str:
@@ -73,12 +77,18 @@ def resolve_direct_employer_match(
draft: CanonicalJobDraft, *, source: Source | None
) -> EmployerResolutionDecision:
if source is None or source.source_type == Source.Type.EMPLOYER or not draft.normalized_title:
return EmployerResolutionDecision(None, "no_direct_resolution", 0.0, canonical_url=None, conflict=False)
return EmployerResolutionDecision(
None, "no_direct_resolution", 0.0, canonical_url=None, conflict=False
)
candidates = JobPosting.objects.filter(
status__in=[JobPosting.Status.ACTIVE, JobPosting.Status.NEW],
direct_employer=True,
).select_related("employer").order_by("id")
candidates = (
JobPosting.objects.filter(
status__in=[JobPosting.Status.ACTIVE, JobPosting.Status.NEW],
direct_employer=True,
)
.select_related("employer")
.order_by("id")
)
best: JobPosting | None = None
best_score = 0.0
@@ -92,7 +102,13 @@ def resolve_direct_employer_match(
continue
location_score: float | None = None
if draft.location_text and candidate.raw_location:
if (
draft.municipality
and candidate.municipality
and normalize_token(draft.municipality) == normalize_token(candidate.municipality)
):
location_score = 1.0
elif draft.location_text and candidate.raw_location:
location_score = _token_similarity(draft.location_text, candidate.raw_location)
employer_score: float | None = None
@@ -112,15 +128,23 @@ def resolve_direct_employer_match(
canonical_host_match=canonical_host_match,
)
has_conflict = False
if draft.location_text and candidate.raw_location:
if location_score is not None and location_score < CONFLICT_LOCATION_THRESHOLD:
has_conflict = True
if draft.employer_name and candidate.employer_name and (
employer_score is not None and employer_score < CONFLICT_EMPLOYER_THRESHOLD
has_conflict = bool(
draft.location_text
and candidate.raw_location
and location_score is not None
and location_score < CONFLICT_LOCATION_THRESHOLD
)
if (
draft.employer_name
and candidate.employer_name
and (employer_score is not None and employer_score < CONFLICT_EMPLOYER_THRESHOLD)
):
has_conflict = True
if draft.employer_name and not candidate.employer_name and title_score < CONFLICT_TITLE_THRESHOLD:
if (
draft.employer_name
and not candidate.employer_name
and title_score < CONFLICT_TITLE_THRESHOLD
):
has_conflict = True
if score > best_score:
+20 -9
View File
@@ -1,18 +1,15 @@
from __future__ import annotations
from dataclasses import dataclass
from datetime import timedelta
from typing import Any
from django.db import transaction
from django.utils import timezone
from apps.jobs.models import Application, Feedback, JobPosting, ScoreRun
from apps.jobs.models import Feedback, JobPosting, ScoreRun
from apps.jobs.services.applications import apply_application_on_feedback
from apps.profiles.models import SearchProfile
from apps.profiles.services import apply_feedback_delta
LEARNING_MIN_SAMPLES = 2
LEARNING_DELTA_BY_ACTION = {
Feedback.Action.INTERESTING: 1.0,
@@ -47,15 +44,21 @@ def _latest_score_run(profile: SearchProfile, job: JobPosting) -> ScoreRun | Non
def _best_signal_feature(score_run: ScoreRun | None) -> str:
if score_run is None:
return "content"
components = {
key: value for key, value in (score_run.components or {}).items() if key in LEARNING_FEATURES
key: value
for key, value in (score_run.components or {}).items()
if key in LEARNING_FEATURES
}
if not components:
return "content"
return max(components, key=components.get)
def _classify_hide_signal(profile: SearchProfile, score_run: ScoreRun | None, reason: str) -> _LearningSignal:
def _classify_hide_signal(
profile: SearchProfile, score_run: ScoreRun | None, reason: str
) -> _LearningSignal:
normalized = _normalize_reason_text(reason)
if not normalized:
return _LearningSignal(
@@ -71,7 +74,9 @@ def _classify_hide_signal(profile: SearchProfile, score_run: ScoreRun | None, re
reason_code="non_learning_title",
learnable=False,
)
if any(token in normalized for token in ("afstand", "afstands", "km", "locatie", "verplaatsing")):
if any(
token in normalized for token in ("afstand", "afstands", "km", "locatie", "verplaatsing")
):
return _LearningSignal(
feature="",
delta=0.0,
@@ -128,7 +133,9 @@ def _learning_signal_count(profile: SearchProfile, feature: str) -> int:
return count
def _apply_learning_metadata(feedback: Feedback, signal: _LearningSignal, *, samples: int, applied: bool) -> None:
def _apply_learning_metadata(
feedback: Feedback, signal: _LearningSignal, *, samples: int, applied: bool
) -> None:
metadata: dict[str, Any] = dict(feedback.metadata or {})
metadata["learning"] = {
"status": "applied" if applied else "queued",
@@ -187,7 +194,11 @@ def record_feedback(
action=action,
reason=reason[:200],
)
signal = _classify_learning_signal(profile=profile, job=job, action=action, reason=reason) if profile else None
signal = (
_classify_learning_signal(profile=profile, job=job, action=action, reason=reason)
if profile
else None
)
if signal:
_evaluate_learning(profile=profile, feedback=feedback, signal=signal)
if action == Feedback.Action.APPLIED:
+17 -16
View File
@@ -20,8 +20,7 @@ class GeocodeProvider(Protocol):
confidence: float
metadata: dict[str, Any]
def resolve(self, query: str) -> list["LocationMatch"]:
...
def resolve(self, query: str) -> list[LocationMatch]: ...
@dataclass(frozen=True)
@@ -63,10 +62,7 @@ def parse_belgian_location_query(raw: str) -> tuple[str | None, str | None]:
for chunk in re.findall(r"\b\d{4}\b", normalized):
postal = chunk
break
if "," in normalized:
municipality_part = normalized.split(",", 1)[0]
else:
municipality_part = normalized
municipality_part = normalized.split(",", 1)[0] if "," in normalized else normalized
municipality_part = re.sub(r"\b\d{4}\b", " ", municipality_part)
municipality_part = re.sub(r"[^a-z0-9 ]", " ", municipality_part)
municipality = " ".join(municipality_part.split())
@@ -100,11 +96,12 @@ def _read_rows(path: str | Path) -> list[_ParsedRow]:
with csv_path.open("r", encoding="utf-8-sig", newline="") as handle:
reader = csv.DictReader(handle)
headers = set((reader.fieldnames or []))
headers = set(reader.fieldnames or [])
required = {"postal_code", "municipality", "region", "latitude", "longitude"}
if not required.issubset(headers):
raise CommandError(
"Verplichte kolommen ontbreken: postal_code, municipality, region, latitude, longitude"
"Verplichte kolommen ontbreken: postal_code, municipality, region, "
"latitude, longitude"
)
for row_number, raw_row in enumerate(reader, start=2):
@@ -116,7 +113,9 @@ def _read_rows(path: str | Path) -> list[_ParsedRow]:
if not postal_code:
raise CommandError(f"regel {row_number}: postal_code mag niet leeg zijn")
if len(postal_code) != 4 or not postal_code.isdigit():
raise CommandError(f"regel {row_number}: ongeldige Belgische postcode {postal_code}")
raise CommandError(
f"regel {row_number}: ongeldige Belgische postcode {postal_code}"
)
if not municipality:
raise CommandError(f"regel {row_number}: municipality mag niet leeg zijn")
@@ -136,7 +135,8 @@ def _read_rows(path: str | Path) -> list[_ParsedRow]:
row_key = (postal_code, normalized_municipality)
if row_key in seen:
raise CommandError(
f"regel {row_number}: dubbel record in bestand voor {postal_code} {municipality}"
f"regel {row_number}: dubbel record in bestand voor "
f"{postal_code} {municipality}"
)
seen.add(row_key)
@@ -171,7 +171,9 @@ class CsvGeocodeProvider:
self.metadata: dict[str, Any] = metadata or {}
@staticmethod
def _load_candidates(query_value: str, query_kind: str, *, source_name: str, source_version: str):
def _load_candidates(
query_value: str, query_kind: str, *, source_name: str, source_version: str
):
return GeocodeLocationLookup.objects.filter(
source_name=source_name,
source_version=source_version,
@@ -307,7 +309,9 @@ def import_csv_geodata(
rows = _read_rows(path)
if len(rows) > 50000:
raise CommandError("Importbestand bevat meer dan 50.000 records; import in delen aanbevolen.")
raise CommandError(
"Importbestand bevat meer dan 50.000 records; import in delen aanbevolen."
)
existing_rows = GeocodeLocationLookup.objects.filter(
source_name=source_name,
@@ -347,10 +351,7 @@ def import_csv_geodata(
row.postal_code,
row.municipality,
)
if (not replace) and (
municipal_key in existing_keys
or postal_key in existing_keys
):
if (not replace) and (municipal_key in existing_keys or postal_key in existing_keys):
raise CommandError(
"Import zou bestaande lookuprecords overschrijven zonder --replace."
)
+3 -1
View File
@@ -271,7 +271,9 @@ def persist_draft(
else:
alias.last_seen = timezone.now()
alias.raw_document = document
alias.payload = _alias_payload(alias.payload, decision, fallback_canonical_url=draft.canonical_url)
alias.payload = _alias_payload(
alias.payload, decision, fallback_canonical_url=draft.canonical_url
)
if direct:
alias.is_canonical = True
alias.save(
+35 -15
View File
@@ -1,15 +1,16 @@
from __future__ import annotations
from collections.abc import Iterable
from dataclasses import dataclass
from difflib import SequenceMatcher
from typing import Any, Iterable
from typing import Any
from django.db import transaction
from apps.jobs.models import JobPosting, ScoreRun
from apps.jobs.services.ai import AiAnalysis, AiAnalysisCache, analyze_job_text
from apps.jobs.services.distance import estimate_commute, haversine_km
from apps.jobs.services.geocoding import resolve_cached_location
from apps.jobs.models import JobPosting, ScoreRun
from apps.profiles.models import SearchProfile
from .normalization import normalize_token
@@ -106,7 +107,9 @@ def _profile_reference(profile: SearchProfile) -> _GeoReference | None:
def _title_fit(job: JobPosting, profile: SearchProfile) -> float:
if not profile.desired_titles:
return 0.65
return max(_similarity(job.normalized_title, desired_title) for desired_title in profile.desired_titles)
return max(
_similarity(job.normalized_title, desired_title) for desired_title in profile.desired_titles
)
def _skill_fit(job: JobPosting, profile: SearchProfile) -> tuple[float, list[str], list[str]]:
@@ -191,9 +194,15 @@ def _hard_exclusions(
):
reasons.append(f"Uitgesloten regio: {job.region or job.municipality}")
if distance.exact_distance_km is not None and distance.exact_distance_km > distance_limit:
if job.workplace_type != "remote":
reasons.append(f"Afstand {distance.exact_distance_km:.0f} km boven maximum {profile.max_distance_km} km")
if (
distance.exact_distance_km is not None
and distance.exact_distance_km > distance_limit
and job.workplace_type != "remote"
):
reasons.append(
f"Afstand {distance.exact_distance_km:.0f} km boven maximum "
f"{profile.max_distance_km} km"
)
max_commute_minutes = profile.hard_rules.get("max_commute_minutes")
try:
@@ -206,7 +215,8 @@ def _hard_exclusions(
and distance.commute_minutes > max_commute_limit
):
reasons.append(
f"Geschatte reistijd {distance.commute_minutes} minuten boven limiet van {max_commute_limit}"
f"Geschatte reistijd {distance.commute_minutes} minuten boven limiet van "
f"{max_commute_limit}"
)
excluded_skills = {
@@ -244,7 +254,9 @@ def _ai_feature_score(features: dict[str, Any]) -> float:
"unknown": 0.03,
"": 0.03,
}.get(seniority, 0.05)
score = 0.5 * (1.0 - support_ratio) + 0.25 * (1.0 - consultancy_ratio) + 0.15 * (1.0 - travel_ratio)
score = (
0.5 * (1.0 - support_ratio) + 0.25 * (1.0 - consultancy_ratio) + 0.15 * (1.0 - travel_ratio)
)
return max(0.0, min(1.0, score + seniority_boost))
@@ -373,7 +385,10 @@ def calculate_score(job: JobPosting, profile: SearchProfile) -> ScoreResult:
positives.append("Herkenbare skills: " + ", ".join(present_skills[:6]))
if job.direct_employer and not job.recruiter:
positives.append("Rechtstreekse werkgeversbron.")
if distance.exact_distance_km is not None and distance.exact_distance_km <= profile.max_distance_km:
if (
distance.exact_distance_km is not None
and distance.exact_distance_km <= profile.max_distance_km
):
positives.append(f"Binnen de ingestelde afstand ({distance.exact_distance_km:.0f} km).")
if (
distance.exact_distance_km is None
@@ -382,13 +397,18 @@ def calculate_score(job: JobPosting, profile: SearchProfile) -> ScoreResult:
):
estimate_label = "geschatte" if distance.commute_estimate else "ingeschatte"
concerns.append(
f"Schatting: {estimate_label} reistijd ca. {distance.commute_minutes} min (conservatief)."
f"Schatting: {estimate_label} reistijd ca. {distance.commute_minutes} min "
"(conservatief)."
)
if support_ratio >= 0.5:
concerns.append("Vacature bevat sterke first-line/helpdesksignalen.")
if missing_skills:
concerns.append("Niet duidelijk teruggevonden: " + ", ".join(missing_skills[:6]))
if distance.exact_distance_km is None and distance.has_distance_data and job.workplace_type != "remote":
if (
distance.exact_distance_km is None
and distance.has_distance_data
and job.workplace_type != "remote"
):
concerns.append("Afstand kon nog niet exact betrouwbaar worden berekend.")
if not job.compensation:
concerns.append("Salaris of barema is niet vermeld.")
@@ -436,7 +456,9 @@ def calculate_score(job: JobPosting, profile: SearchProfile) -> ScoreResult:
"warnings": ai_analysis.warnings,
"features": ai_analysis.features,
"weight_requested": float(profile.weights.get("ai", 0) or 0),
"weight_applied": ai_weight if ai_analysis.status == AiAnalysisCache.Status.OK else 0.0,
"weight_applied": ai_weight
if ai_analysis.status == AiAnalysisCache.Status.OK
else 0.0,
},
},
model_version=ai_analysis.model,
@@ -451,9 +473,7 @@ def _iter_active_profiles(profile_id: int | None):
return profiles
def rescore_jobs_with_profiles(
jobs: Iterable[JobPosting], *, profile_id: int | None = None
) -> int:
def rescore_jobs_with_profiles(jobs: Iterable[JobPosting], *, profile_id: int | None = None) -> int:
count = 0
for profile in _iter_active_profiles(profile_id):
for job in jobs:
+2 -2
View File
@@ -3,11 +3,11 @@ from django.urls import path
from .views import (
ApplicationListView,
ApplicationUpdateView,
JobDetailView,
JobListView,
application_delete,
application_export,
application_print,
JobDetailView,
JobListView,
job_feedback,
)
+8 -7
View File
@@ -4,7 +4,7 @@ from django.contrib import messages
from django.contrib.auth.decorators import login_required
from django.contrib.auth.mixins import LoginRequiredMixin
from django.db.models import Q
from django.http import HttpResponse
from django.http import Http404, HttpResponse
from django.shortcuts import get_object_or_404, redirect
from django.urls import reverse
from django.views.decorators.http import require_POST
@@ -118,12 +118,11 @@ class ApplicationUpdateView(LoginRequiredMixin, UpdateView):
return Application.objects.filter(user=self.request.user).select_related("job")
def form_valid(self, form):
previous = {
"status": self.object.status,
"notes": self.object.notes,
"contact_name": self.object.contact_name,
"contact_email": self.object.contact_email,
}
previous = (
Application.objects.filter(pk=self.object.pk)
.values("status", "notes", "contact_name", "contact_email")
.get()
)
response = super().form_valid(form)
track_application_changes(
application=self.object,
@@ -165,6 +164,8 @@ def application_print(request, pk: int):
@login_required
@require_POST
def application_delete(request, pk: int):
if Application.objects.filter(pk=pk).exclude(user=request.user).exists():
raise Http404
deleted = delete_application_dossier(application_id=pk, user=request.user)
if deleted:
messages.success(request, "Sollicitatiedossier verwijderd.")
+10 -2
View File
@@ -68,10 +68,18 @@ class ReminderOutbox(TimeStampedModel):
payload = models.JSONField(default=dict)
status = models.CharField(max_length=16, choices=Status.choices, default=Status.PENDING)
job = models.ForeignKey(
"jobs.JobPosting", on_delete=models.SET_NULL, null=True, blank=True, related_name="reminders"
"jobs.JobPosting",
on_delete=models.SET_NULL,
null=True,
blank=True,
related_name="reminders",
)
application = models.ForeignKey(
"jobs.Application", on_delete=models.SET_NULL, null=True, blank=True, related_name="reminders"
"jobs.Application",
on_delete=models.SET_NULL,
null=True,
blank=True,
related_name="reminders",
)
score_run = models.ForeignKey(
"jobs.ScoreRun", on_delete=models.SET_NULL, null=True, blank=True, related_name="reminders"
+11 -8
View File
@@ -15,7 +15,6 @@ from apps.profiles.models import SearchProfile
from .models import DigestOutbox, ReminderOutbox
TOP_MATCH_MIN_CONFIDENCE = Decimal("0.85")
@@ -27,7 +26,8 @@ def _to_local(profile: SearchProfile, *, now: datetime | None = None) -> datetim
def _hidden_job_ids(profile: SearchProfile) -> set[str]:
return set(
str(job_id) for job_id in Feedback.objects.filter(
str(job_id)
for job_id in Feedback.objects.filter(
user=profile.user, action=Feedback.Action.HIDE
).values_list("job_id", flat=True)
)
@@ -35,7 +35,8 @@ def _hidden_job_ids(profile: SearchProfile) -> set[str]:
def _applied_job_ids(profile: SearchProfile) -> set[str]:
return set(
str(job_id) for job_id in Application.objects.filter(
str(job_id)
for job_id in Application.objects.filter(
user=profile.user, status=Application.Status.APPLIED
).values_list("job_id", flat=True)
)
@@ -234,7 +235,9 @@ def create_closing_reminders(profile: SearchProfile, *, now=None) -> list[Remind
if not profile.reminders_enabled:
return []
now_local = _to_local(profile, now=now)
due_start = datetime.combine(now_local.date(), time.min, tzinfo=now_local.tzinfo).astimezone(UTC)
due_start = datetime.combine(now_local.date(), time.min, tzinfo=now_local.tzinfo).astimezone(
UTC
)
due_end = due_start + timedelta(days=1)
hidden_ids = _hidden_job_ids(profile)
applied_ids = _applied_job_ids(profile)
@@ -255,9 +258,7 @@ def create_closing_reminders(profile: SearchProfile, *, now=None) -> list[Remind
local_valid_through = (
job.valid_through.astimezone(now_local.tzinfo) if job.valid_through else now_local
)
dedupe_key = (
f"closing:{profile.pk}:{job.pk}:{local_valid_through.date().isoformat()}"
)
dedupe_key = f"closing:{profile.pk}:{job.pk}:{local_valid_through.date().isoformat()}"
outboxes.append(
_build_reminder_outbox(
profile,
@@ -324,7 +325,9 @@ def create_follow_up_reminders(profile: SearchProfile, *, now=None) -> list[Remi
"employer": application.job.employer_name,
"link": application.job.canonical_url,
"follow_up_date": (
application.follow_up_date.isoformat() if application.follow_up_date else None
application.follow_up_date.isoformat()
if application.follow_up_date
else None
),
},
job=application.job,
+6 -6
View File
@@ -1,7 +1,3 @@
from .base import ExtractedJob, ExtractionResult
from .generic_html import GenericHtmlAdapter
from .jsonld import JsonLdJobPostingAdapter
from .rss import RssAdapter
from .ats import (
GreenhouseAdapter,
LeverAdapter,
@@ -9,16 +5,20 @@ from .ats import (
SmartRecruitersAdapter,
WorkableAdapter,
)
from .base import ExtractedJob, ExtractionResult
from .generic_html import GenericHtmlAdapter
from .jsonld import JsonLdJobPostingAdapter
from .rss import RssAdapter
__all__ = [
"ExtractedJob",
"ExtractionResult",
"GenericHtmlAdapter",
"JsonLdJobPostingAdapter",
"RssAdapter",
"GreenhouseAdapter",
"JsonLdJobPostingAdapter",
"LeverAdapter",
"RecruiteeAdapter",
"RssAdapter",
"SmartRecruitersAdapter",
"WorkableAdapter",
]
+16 -10
View File
@@ -1,4 +1,4 @@
from __future__ import annotations
from __future__ import annotations
import json
from abc import ABC, abstractmethod
@@ -8,7 +8,6 @@ from bs4 import BeautifulSoup
from .base import ExtractedJob, ExtractionResult, FieldEvidence
CLOSED_STATUSES = {
"closed",
"inactive",
@@ -51,7 +50,7 @@ def _first_text(data, *candidates):
value = data.get(candidate) if isinstance(data, dict) else None
if value is None:
continue
if isinstance(value, (list, tuple)):
if isinstance(value, list | tuple):
for item in value:
text = _to_text(item)
if text:
@@ -114,7 +113,7 @@ def _as_text(value) -> str:
def _extract_payload(content: str):
try:
return json.loads(content)
return json.loads(content.lstrip("\ufeff"))
except json.JSONDecodeError:
pass
soup = BeautifulSoup(content, "lxml")
@@ -123,7 +122,7 @@ def _extract_payload(content: str):
if not script_text:
continue
try:
return json.loads(script_text)
return json.loads(script_text.lstrip("\ufeff"))
except json.JSONDecodeError:
continue
return None
@@ -139,8 +138,7 @@ class _AtsAdapter(ABC):
closed_statuses = CLOSED_STATUSES
@abstractmethod
def _extract_records(self, payload) -> list[dict[str, object]]:
...
def _extract_records(self, payload) -> list[dict[str, object]]: ...
def _supports_url(self, url: str) -> bool:
host = (urlsplit(url).hostname or "").lower()
@@ -191,7 +189,7 @@ class _AtsAdapter(ABC):
location_raw = ", ".join(location_values)
location = location_raw
location_parts = [part.strip() for part in _to_text(location).split(",") if part.strip()]
[part.strip() for part in _to_text(location).split(",") if part.strip()]
region = _first_text(
record,
"region",
@@ -343,7 +341,7 @@ class _AtsAdapter(ABC):
valid_through=valid_through,
employment_types=employment_types,
workplace_type=workplace_type,
raw=record,
raw=record.get("__raw__", record),
evidence=evidence,
)
@@ -389,7 +387,9 @@ class _AtsAdapter(ABC):
warnings: list[str] = []
if not jobs:
warnings.append("Geen actieve ATS-vacatures gevonden")
return ExtractionResult(jobs, self.parser_key, self.parser_version, 0.9 if jobs else 0.0, warnings)
return ExtractionResult(
jobs, self.parser_key, self.parser_version, 0.9 if jobs else 0.0, warnings
)
class GreenhouseAdapter(_AtsAdapter):
@@ -452,6 +452,12 @@ class LeverAdapter(_AtsAdapter):
break
value = value[key]
if isinstance(value, dict):
if path == ("position",):
raw_payload = {
"position": value,
"work_type": _to_text(value.get("workplaceType")),
}
return [{**value, "__raw__": raw_payload}]
return [value]
return []
+6 -3
View File
@@ -1,8 +1,7 @@
from __future__ import annotations
from __future__ import annotations
from apps.sources.models import RawDocument
from .base import ExtractionResult
from .ats import (
GreenhouseAdapter,
LeverAdapter,
@@ -10,6 +9,7 @@ from .ats import (
SmartRecruitersAdapter,
WorkableAdapter,
)
from .base import ExtractionResult
from .generic_html import GenericHtmlAdapter
from .jsonld import JsonLdJobPostingAdapter
from .rss import RssAdapter
@@ -45,7 +45,10 @@ class AdapterRegistry:
result = provider.extract(content, url=url)
if any(
msg in result.warnings
for msg in ("Geen parseerbare ATS-response", "Geen herkenbare ATS-markup voor deze adapter")
for msg in (
"Geen parseerbare ATS-response",
"Geen herkenbare ATS-markup voor deze adapter",
)
):
continue
return result
+17 -8
View File
@@ -1,7 +1,6 @@
from __future__ import annotations
from datetime import timedelta
from uuid import uuid4
from urllib.parse import urlparse
from django.conf import settings
@@ -91,7 +90,7 @@ class Source(TimeStampedModel):
return [entry for entry in entries if isinstance(entry, dict)]
@property
def latest_policy_review(self) -> "SourcePolicyReview | None":
def latest_policy_review(self) -> SourcePolicyReview | None:
return self.policy_reviews.order_by("-created_at").first()
@property
@@ -109,27 +108,37 @@ class Source(TimeStampedModel):
if not review:
return "Geen review geregistreerd"
if review.is_expired:
return f"Review verlopen op {review.expires_at:%Y-%m-%d}" if review.expires_at else "Review verlopen"
return (
f"Review verlopen op {review.expires_at:%Y-%m-%d}"
if review.expires_at
else "Review verlopen"
)
if review.expires_at:
return f"{review.get_decision_display()} geldig tot {review.expires_at:%Y-%m-%d}"
return f"{review.get_decision_display()} zonder vervaldatum"
@property
def requires_terms_review(self) -> bool:
return self.policy == Source.Policy.ALLOW and self.policy_review_state in {"missing", "expired"}
return self.policy == Source.Policy.ALLOW and self.policy_review_state in {
"missing",
"expired",
}
def schedule_after_success(self, *, now=None, jitter_seconds: int = 0) -> None:
now = now or timezone.now()
self.last_success_at = now
self.failure_count = 0
self.next_run_at = now + timedelta(minutes=self.crawl_interval_minutes) + timedelta(
seconds=max(0, jitter_seconds)
self.next_run_at = (
now
+ timedelta(minutes=self.crawl_interval_minutes)
+ timedelta(seconds=max(0, jitter_seconds))
)
self.save(update_fields=["last_success_at", "failure_count", "next_run_at", "updated_at"])
def schedule_after_failure(
self,
*, now=None,
*,
now=None,
backoff_minutes: int | None = None,
backoff_seconds: int | None = None,
) -> None:
@@ -186,7 +195,7 @@ class SourceRun(TimeStampedModel):
class SourceLease(TimeStampedModel):
source = models.OneToOneField(Source, on_delete=models.CASCADE, related_name="lease")
token = models.CharField(max_length=64, default=lambda: str(uuid4()))
token = models.CharField(max_length=64, default="")
worker_id = models.CharField(max_length=128, blank=True)
expires_at = models.DateTimeField(db_index=True)
+15 -11
View File
@@ -4,11 +4,11 @@ import ipaddress
import json
import re
from dataclasses import dataclass
from datetime import datetime, timezone
from datetime import UTC, datetime
from urllib.parse import urljoin, urlsplit
from xml.etree import ElementTree
from bs4 import BeautifulSoup
from defusedxml import ElementTree
from django.db import transaction
from apps.sources.adapters.email_alert import EmailAlertAdapter
@@ -166,15 +166,14 @@ def persist_discovery_candidates(candidates: list[SourceCandidate]) -> tuple[int
evidence = metadata.get("discovery", [])
if not isinstance(evidence, list):
evidence = []
if not any(item.get("url") == candidate.url for item in evidence if isinstance(item, dict)):
if not any(
item.get("url") == candidate.url for item in evidence if isinstance(item, dict)
):
evidence.append(provenance)
metadata["discovery"] = evidence[-20:]
metadata["discovered_at"] = now
source.metadata = metadata
updated_fields.append("metadata")
else:
skipped += 1
if not source.name:
source.name = _candidate_name(candidate)
updated_fields.append("name")
@@ -223,19 +222,24 @@ def _discover_html(
for anchor in soup.find_all("a", href=True):
label = " ".join(anchor.get_text(" ", strip=True).split())
raw_url = urljoin(base_url, str(anchor["href"]))
candidate_host = _hostname(raw_url)
same_domain = bool(base_domain and domain_matches(candidate_host, base_domain))
candidate = _build_candidate(
raw_url=urljoin(base_url, str(anchor["href"])),
raw_url=raw_url,
base_domain=base_domain,
source_type=Source.Type.EMPLOYER,
label=label,
reason="career-link",
discovered_from="html",
confidence=0.9,
allow_off_domain=False,
confidence=0.9 if same_domain else 0.7,
allow_off_domain=True,
)
if not candidate:
continue
if not CAREER_PATTERN.search(label) and not CAREER_PATTERN.search(urlsplit(candidate.url).path):
if not CAREER_PATTERN.search(label) and not CAREER_PATTERN.search(
urlsplit(candidate.url).path
):
continue
candidates.append(candidate)
@@ -388,4 +392,4 @@ def _to_domain_root_url(url: str) -> str:
def _utc_now() -> str:
return datetime.now(timezone.utc).isoformat()
return datetime.now(UTC).isoformat()
+12 -3
View File
@@ -1,10 +1,11 @@
from __future__ import annotations
import hashlib
from datetime import datetime, timezone as utc
from email.utils import parsedate_to_datetime
from collections.abc import Mapping
from dataclasses import dataclass
from datetime import datetime
from datetime import timezone as utc
from email.utils import parsedate_to_datetime
from urllib.parse import urljoin
import httpx
@@ -13,7 +14,7 @@ from django.conf import settings
from apps.sources.models import Source
from .policy import assess_url
from .url_security import validate_public_url
from .url_security import ValidatedUrl, validate_public_url
ALLOWED_CONTENT_TYPES = (
"text/html",
@@ -124,12 +125,19 @@ def fetch_url(
headers=headers,
)
current_url = url
previous_validation: ValidatedUrl | None = None
try:
for _ in range(settings.FETCHER_MAX_REDIRECTS + 1):
validation = validate_public_url(
current_url,
allow_nonstandard_ports=settings.FETCHER_ALLOW_NONSTANDARD_PORTS,
)
if (
previous_validation is not None
and validation.hostname == previous_validation.hostname
and not set(validation.addresses).intersection(previous_validation.addresses)
):
raise FetchError("DNS-rebindcontrole faalde bij redirect naar dezelfde host.")
response = http_client.get(current_url, headers=headers)
post_validation = validate_public_url(
str(response.url) if response.url else current_url,
@@ -145,6 +153,7 @@ def fetch_url(
next_decision = assess_url(current_url, source=source)
if not next_decision.allowed:
raise PolicyBlockedError(next_decision.reason)
previous_validation = validation
continue
if response.status_code == 304:
return FetchedDocument(url, current_url, 304, dict(response.headers), b"")
+14 -4
View File
@@ -174,7 +174,9 @@ def _is_parser_drift_run(run: SourceRun) -> bool:
def _determine_health_action(runs: list[SourceRun]) -> tuple[str | None, str | None]:
recent_runs = runs[:SOURCE_HEALTH_RUN_WINDOW]
considered_failures = [
run for run in recent_runs if run.status in {SourceRun.Status.FAILED, SourceRun.Status.SKIPPED}
run
for run in recent_runs
if run.status in {SourceRun.Status.FAILED, SourceRun.Status.SKIPPED}
]
if any(
@@ -209,7 +211,9 @@ def _determine_health_action(runs: list[SourceRun]) -> tuple[str | None, str | N
return None, None
def collect_source_health(*, runs_to_consider: int = SOURCE_HEALTH_RUN_WINDOW) -> list[SourceHealth]:
def collect_source_health(
*, runs_to_consider: int = SOURCE_HEALTH_RUN_WINDOW
) -> list[SourceHealth]:
sources = Source.objects.order_by("name").all()
rows: list[SourceHealth] = []
@@ -256,7 +260,11 @@ def collect_source_health(*, runs_to_consider: int = SOURCE_HEALTH_RUN_WINDOW) -
def evaluate_source_health(*, now: datetime | None = None) -> dict[str, int]:
now = now or timezone.now()
rows = [row for row in collect_source_health() if row.source_status in {Source.Status.ACTIVE, Source.Status.TRIAL}]
rows = [
row
for row in collect_source_health()
if row.source_status in {Source.Status.ACTIVE, Source.Status.TRIAL}
]
counts = {"evaluated": len(rows), "quarantined": 0}
for row in rows:
@@ -309,7 +317,9 @@ def canary_recovery_sources(*, now: datetime | None = None) -> list[Source]:
if bool(health.get("canary_started", False)):
continue
recovery_due = _from_iso(health.get("recovery_due_at") if isinstance(health, dict) else None)
recovery_due = _from_iso(
health.get("recovery_due_at") if isinstance(health, dict) else None
)
if recovery_due and recovery_due > now:
continue
+6 -13
View File
@@ -18,9 +18,9 @@ from apps.jobs.services.pipeline import process_raw_document
from apps.sources.models import RawDocument, Source, SourcePolicyReview, SourceRun
from apps.sources.services.canonicalize import canonicalize_url
from apps.sources.services.fetcher import (
FetchedDocument,
FetchError,
FetchTimeoutError,
FetchedDocument,
PolicyBlockedError,
RateLimitedError,
fetch_url,
@@ -88,11 +88,7 @@ def _mode_source_name(domain: str, *, mode: str) -> str:
def _ensure_manual_review(source: Source, actor) -> None:
review = source.latest_policy_review
if (
review
and not review.is_expired
and review.decision == SourcePolicyReview.Decision.ALLOW
):
if review and not review.is_expired and review.decision == SourcePolicyReview.Decision.ALLOW:
return
create_policy_review(
source,
@@ -120,9 +116,7 @@ def _ensure_manual_source(*, domain: str, source_url: str, actor, mode: str) ->
source.base_url = source_url
source.status = Source.Status.CANDIDATE
source.policy = Source.Policy.ALLOW
source.save(
update_fields=["name", "base_url", "status", "policy", "updated_at"]
)
source.save(update_fields=["name", "base_url", "status", "policy", "updated_at"])
_ensure_manual_review(source, actor=actor)
return source
@@ -175,7 +169,7 @@ def _run_pipeline(document: RawDocument, *, source_run: SourceRun) -> ManualImpo
source = document.source
source_name = source.name if source else ""
warnings: list[str] = list(metrics.get("warnings", []))
warnings_count = len(warnings)
len(warnings)
source_run.finish(
SourceRun.Status.SUCCESS,
http_status=document.source_run.http_status if document.source_run else None,
@@ -186,9 +180,8 @@ def _run_pipeline(document: RawDocument, *, source_run: SourceRun) -> ManualImpo
metrics={"parser": metrics["parser"], "warnings": warnings},
)
jobs = _build_jobs_from_document(document)
if metrics["created"] == 0 and metrics["updated"] == 0:
if not warnings:
warnings.append("De bron leverde geen herkenbare vacaturedata op.")
if metrics["created"] == 0 and metrics["updated"] == 0 and not warnings:
warnings.append("De bron leverde geen herkenbare vacaturedata op.")
if not warnings:
# keep stable, machine-readable payload shape
warnings = []
+15 -5
View File
@@ -79,18 +79,28 @@ def _check_review_gate(source: Source) -> PolicyDecision | None:
if not review:
return PolicyDecision(False, Source.Policy.REVIEW, "Review vereist")
if review.decision == SourcePolicyReview.Decision.DENY:
return PolicyDecision(False, Source.Policy.DENY, review.reason or "Review blokkeert bron")
return PolicyDecision(
False, Source.Policy.DENY, review.reason or "Review blokkeert bron"
)
if review.decision == SourcePolicyReview.Decision.PAUSE:
return PolicyDecision(False, Source.Policy.REVIEW, review.reason or "Review vraagt pauze")
return PolicyDecision(
False, Source.Policy.REVIEW, review.reason or "Review vraagt pauze"
)
return None
if source.policy == Source.Policy.ALLOW:
if not review:
return PolicyDecision(False, Source.Policy.REVIEW, "Review ontbreekt voor actief beleid")
return PolicyDecision(
False, Source.Policy.REVIEW, "Review ontbreekt voor actief beleid"
)
if review.decision == SourcePolicyReview.Decision.DENY:
return PolicyDecision(False, Source.Policy.DENY, review.reason or "Review blokkeert bron")
return PolicyDecision(
False, Source.Policy.DENY, review.reason or "Review blokkeert bron"
)
if review.decision == SourcePolicyReview.Decision.PAUSE:
return PolicyDecision(False, Source.Policy.REVIEW, review.reason or "Review vraagt pauze")
return PolicyDecision(
False, Source.Policy.REVIEW, review.reason or "Review vraagt pauze"
)
return None
+23 -16
View File
@@ -10,6 +10,7 @@ from django.conf import settings
from django.utils import timezone
from apps.sources.models import Source, SourceRobotsCache
from .url_security import UnsafeUrlError, validate_public_url
ALLOW = "allow"
@@ -30,7 +31,11 @@ def _origin_for(url: str) -> str:
if not host:
raise ValueError("Host ontbreekt voor robotscontrole.")
port = parts.port
if (parts.scheme == "http" and port == 80) or (parts.scheme == "https" and port == 443) or not port:
if (
(parts.scheme == "http" and port == 80)
or (parts.scheme == "https" and port == 443)
or not port
):
netloc = host
else:
netloc = f"{host}:{port}"
@@ -61,10 +66,7 @@ def _rules_from_text(text: str) -> dict[str, dict[str, list[str]]]:
key_lower = key.lower()
if key_lower == "user-agent":
token = value.lower()
if token:
active_agents = {token}
else:
active_agents = set()
active_agents = {token} if token else set()
continue
if key_lower not in {ALLOW, DISALLOW}:
continue
@@ -85,7 +87,7 @@ def _pick_rules(rules: dict[str, dict[str, list[str]]], user_agent: str) -> dict
selected = {ALLOW: [], DISALLOW: []}
for agent, values in rules.items():
if agent == "*" or agent and agent in normalized:
if agent == "*" or (agent and agent in normalized):
selected[ALLOW].extend(values[ALLOW])
selected[DISALLOW].extend(values[DISALLOW])
@@ -95,10 +97,14 @@ def _pick_rules(rules: dict[str, dict[str, list[str]]], user_agent: str) -> dict
def _longest_prefix(path: str, rules: Iterable[str]) -> int:
return max((len(rule.rstrip("/")) for rule in rules if rule and path.startswith(rule)), default=0)
return max(
(len(rule.rstrip("/")) for rule in rules if rule and path.startswith(rule)), default=0
)
def _evaluate_path(path: str, user_agent: str, rules: dict[str, dict[str, list[str]]]) -> RobotsDecision:
def _evaluate_path(
path: str, user_agent: str, rules: dict[str, dict[str, list[str]]]
) -> RobotsDecision:
selected = _pick_rules(rules, user_agent=user_agent)
allow_len = _longest_prefix(path, selected[ALLOW])
disallow_len = _longest_prefix(path, selected[DISALLOW])
@@ -184,12 +190,7 @@ def _persist_cache(
},
)[0]
if status_code in {404, 410}:
rules = {}
elif status_code >= 400:
rules = {}
else:
rules = _rules_from_text(content)
rules = {} if status_code in {404, 410} or status_code >= 400 else _rules_from_text(content)
return SourceRobotsCache.objects.update_or_create(
origin=origin,
@@ -257,12 +258,18 @@ def assess_robots(
return RobotsDecision(False, f"Robotscontrole mislukt: {exc}")
except httpx.HTTPError as exc:
if stale is not None:
return _evaluate_path(path, user_agent=user_agent or getattr(settings, "FETCHER_USER_AGENT", ""), rules=_build_ruleset(stale))
return _evaluate_path(
path,
user_agent=user_agent or getattr(settings, "FETCHER_USER_AGENT", ""),
rules=_build_ruleset(stale),
)
return RobotsDecision(True, f"Robotscontrole tijdelijk niet beschikbaar: {exc}")
if cache.error:
return RobotsDecision(True, cache.error)
rules = _build_ruleset(cache)
decision = _evaluate_path(path, user_agent=user_agent or getattr(settings, "FETCHER_USER_AGENT", ""), rules=rules)
decision = _evaluate_path(
path, user_agent=user_agent or getattr(settings, "FETCHER_USER_AGENT", ""), rules=rules
)
return decision
+13 -13
View File
@@ -76,14 +76,10 @@ def calculate_failure_backoff_seconds(
def calculate_success_jitter_seconds(source: Source) -> int:
return calculate_jitter_seconds(
source, max_seconds=settings.SOURCE_SUCCESS_JITTER_SECONDS
)
return calculate_jitter_seconds(source, max_seconds=settings.SOURCE_SUCCESS_JITTER_SECONDS)
def acquire_source_lease(
*, source_id: int, worker_token: str, now=None
) -> SourceLease | None:
def acquire_source_lease(*, source_id: int, worker_token: str, now=None) -> SourceLease | None:
now = now or timezone.now()
with transaction.atomic():
source = Source.objects.select_for_update().get(pk=source_id)
@@ -101,9 +97,8 @@ def acquire_source_lease(
return None
lease = SourceLease.objects.select_for_update().filter(source=source).first()
if lease is not None and not lease.is_expired:
if lease.token != worker_token:
return None
if lease is not None and not lease.is_expired and lease.token != worker_token:
return None
active_leases = SourceLease.objects.select_for_update().filter(
source__domain=domain, expires_at__gt=now
@@ -118,7 +113,10 @@ def acquire_source_lease(
lease.token = worker_token
lease.worker_id = worker_token
lease.expires_at = now + timedelta(seconds=_lease_ttl_seconds())
lease.save(update_fields=["token", "worker_id", "expires_at", "updated_at"])
if lease.pk is None:
lease.save()
else:
lease.save(update_fields=["token", "worker_id", "expires_at", "updated_at"])
return lease
@@ -129,9 +127,11 @@ def release_source_lease(*, source_id: int, worker_token: str, now=None) -> bool
if not source.domain:
return False
lease = SourceLease.objects.select_for_update().filter(
source=source, token=worker_token
).first()
lease = (
SourceLease.objects.select_for_update()
.filter(source=source, token=worker_token)
.first()
)
if not lease:
return False
+8 -10
View File
@@ -20,7 +20,11 @@ from apps.sources.services.fetcher import (
RateLimitedError,
fetch_url,
)
from apps.sources.services.health import canary_recovery_sources, evaluate_source_health, start_health_canary
from apps.sources.services.health import (
canary_recovery_sources,
evaluate_source_health,
start_health_canary,
)
from apps.sources.services.policy import assess_url
from apps.sources.services.scheduling import (
acquire_source_lease,
@@ -82,9 +86,7 @@ def fetch_source(self, source_id: int, force: bool = False) -> dict[str, object]
try:
fetched = fetch_url(source.base_url, source=source, conditional_headers=headers)
if fetched.status_code == 304:
source.schedule_after_success(
jitter_seconds=calculate_success_jitter_seconds(source)
)
source.schedule_after_success(jitter_seconds=calculate_success_jitter_seconds(source))
run.finish(SourceRun.Status.SUCCESS, http_status=304)
return {"status": "not_modified"}
content_type = fetched.headers.get("content-type", "")
@@ -110,9 +112,7 @@ def fetch_source(self, source_id: int, force: bool = False) -> dict[str, object]
source.etag = fetched.headers.get("etag", source.etag)
source.last_modified = fetched.headers.get("last-modified", source.last_modified)
source.save(update_fields=["etag", "last_modified", "updated_at"])
source.schedule_after_success(
jitter_seconds=calculate_success_jitter_seconds(source)
)
source.schedule_after_success(jitter_seconds=calculate_success_jitter_seconds(source))
run.finish(
SourceRun.Status.SUCCESS,
http_status=fetched.status_code,
@@ -159,9 +159,7 @@ def fetch_source(self, source_id: int, force: bool = False) -> dict[str, object]
)
return {"status": "timeout", "backoff": backoff}
except FetchError as exc:
backoff = calculate_failure_backoff_seconds(
source, failure_count=source.failure_count + 1
)
backoff = calculate_failure_backoff_seconds(source, failure_count=source.failure_count + 1)
source.schedule_after_failure(now=timezone.now(), backoff_seconds=backoff)
run.finish(
SourceRun.Status.FAILED,
+5 -7
View File
@@ -11,22 +11,20 @@ from django.views.decorators.cache import never_cache
from django.views.decorators.http import require_POST
from django.views.generic import ListView
from apps.core.rate_limit import clear_rate_limit, is_rate_limited, register_rate_limit_failure
from .forms import ManualImportForm
from .models import Source
from .models import SourcePolicyReview
from .models import Source, SourcePolicyReview
from .services.manual_import import ManualImportError, ManualImportSummary, import_manual_source
from .services.policy import create_policy_review
from .tasks import fetch_source
from apps.core.rate_limit import clear_rate_limit, is_rate_limited, register_rate_limit_failure
def _source_list_queryset():
return Source.objects.prefetch_related("policy_reviews").all().order_by("name")
def _manual_import_context(
request: HttpRequest, *, form: ManualImportForm, manual_result=None
):
def _manual_import_context(request: HttpRequest, *, form: ManualImportForm, manual_result=None):
base_queryset = _source_list_queryset()
return {
"sources": base_queryset,
@@ -134,7 +132,7 @@ def bulk_candidate_action(request):
source.policy_reason = reason
source.save(update_fields=["status", "policy", "policy_reason", "updated_at"])
messages.success(request, f"{sources.count()} bron(nen) naar { _bulk_summary(action) }.")
messages.success(request, f"{sources.count()} bron(nen) naar {_bulk_summary(action)}.")
return redirect("sources:list")