diff --git a/Dockerfile b/Dockerfile index 1913462..8d7b09e 100644 --- a/Dockerfile +++ b/Dockerfile @@ -5,6 +5,7 @@ ENV PYTHONDONTWRITEBYTECODE=1 \ PYTHONUNBUFFERED=1 \ UV_COMPILE_BYTECODE=1 \ UV_LINK_MODE=copy \ + UV_DEFAULT_INDEX=https://pypi.org/simple \ PATH="/app/.venv/bin:$PATH" RUN apt-get update \ diff --git a/apps/core/rate_limit.py b/apps/core/rate_limit.py index 6cee792..5485f4a 100644 --- a/apps/core/rate_limit.py +++ b/apps/core/rate_limit.py @@ -1,6 +1,7 @@ from __future__ import annotations from dataclasses import dataclass + from django.core.cache import cache from django.http import HttpRequest from django.utils import timezone @@ -17,7 +18,9 @@ def _identity_for_user(request: HttpRequest) -> str: user = getattr(request, "user", None) if user and getattr(user, "is_authenticated", False): return f"user:{user.pk}" - username = (request.POST.get("username", "") if request.method == "POST" else "").strip().lower() + username = ( + (request.POST.get("username", "") if request.method == "POST" else "").strip().lower() + ) return f"anon:{username or _client_ip(request)}" @@ -47,7 +50,7 @@ def is_rate_limited( now = timezone.now().timestamp() identity = _identity_for_user(request) block_until = cache.get(_block_key(namespace, identity)) - if block_until and isinstance(block_until, (int, float)) and block_until > now: + if block_until and isinstance(block_until, int | float) and block_until > now: return RateLimitState(True, max(0, int(block_until - now)), 0) if cache.get(_attempts_key(namespace, identity), 0) >= max_attempts: @@ -84,7 +87,9 @@ def register_rate_limit_failure( attempts = int(cache.get(_attempts_key(namespace, identity), 0)) + 1 if attempts >= max_attempts: now = timezone.now().timestamp() - cache.set(_block_key(namespace, identity), now + max(block_seconds, 1), timeout=block_seconds) + cache.set( + _block_key(namespace, identity), now + max(block_seconds, 1), timeout=block_seconds + ) cache.delete(_attempts_key(namespace, identity)) return RateLimitState(True, max(block_seconds, 1), attempts) diff --git a/apps/core/views.py b/apps/core/views.py index 691b175..201781b 100644 --- a/apps/core/views.py +++ b/apps/core/views.py @@ -1,20 +1,20 @@ from __future__ import annotations from django.conf import settings -from django.contrib.auth.mixins import LoginRequiredMixin from django.contrib import messages +from django.contrib.auth.mixins import LoginRequiredMixin from django.contrib.auth.views import LoginView from django.db.models import Count, Q from django.http import JsonResponse from django.utils import timezone from django.views.generic import TemplateView +from apps.core.rate_limit import clear_rate_limit, is_rate_limited, register_rate_limit_failure from apps.jobs.models import Application, JobPosting, ScoreRun from apps.sources.models import Source -from apps.core.rate_limit import clear_rate_limit, is_rate_limited, register_rate_limit_failure +from apps.sources.services.health import collect_source_health from .health import readiness -from apps.sources.services.health import collect_source_health class SecurityAwareLoginView(LoginView): @@ -39,7 +39,8 @@ class SecurityAwareLoginView(LoginView): request, ( "Te veel inlogpogingen op dit account. Wacht " - f"{self._remaining_minutes(state.remaining_seconds)} en probeer daarna opnieuw." + f"{self._remaining_minutes(state.remaining_seconds)} en probeer daarna " + "opnieuw." ), ) return self.form_invalid(self.get_form()) diff --git a/apps/jobs/admin.py b/apps/jobs/admin.py index 5a4e5d0..76e4360 100644 --- a/apps/jobs/admin.py +++ b/apps/jobs/admin.py @@ -1,15 +1,15 @@ from django.contrib import admin from .models import ( + AiAnalysisCache, Application, ApplicationTimelineEvent, Employer, Feedback, FieldProvenance, - AiAnalysisCache, + GeocodeLocationLookup, JobPosting, JobSourceAlias, - GeocodeLocationLookup, JobVersion, ScoreRun, ) diff --git a/apps/jobs/management/commands/import_geodata.py b/apps/jobs/management/commands/import_geodata.py index 9b5657f..6f31c37 100644 --- a/apps/jobs/management/commands/import_geodata.py +++ b/apps/jobs/management/commands/import_geodata.py @@ -2,13 +2,13 @@ from __future__ import annotations from pathlib import Path from urllib.parse import urlparse -from django.db.models import Q from django.core.management.base import BaseCommand, CommandError +from django.db.models import Q from apps.jobs.models import JobPosting -from apps.jobs.services.scoring import rescore_jobs_with_profiles from apps.jobs.services.geocoding import import_csv_geodata, validate_csv_geodata +from apps.jobs.services.scoring import rescore_jobs_with_profiles class Command(BaseCommand): diff --git a/apps/jobs/models.py b/apps/jobs/models.py index 50e418e..f22fd24 100644 --- a/apps/jobs/models.py +++ b/apps/jobs/models.py @@ -1,7 +1,7 @@ from __future__ import annotations -from decimal import Decimal import uuid +from decimal import Decimal from django.conf import settings from django.db import models @@ -193,8 +193,12 @@ class GeocodeLocationLookup(TimeStampedModel): class Meta: ordering = ["query_kind", "query_value", "municipality"] indexes = [ - models.Index(fields=["query_kind", "query_value"]), - models.Index(fields=["source_name", "source_version"]), + models.Index( + fields=["query_kind", "query_value"], name="geocode_lookup_kind_query_idx" + ), + models.Index( + fields=["source_name", "source_version"], name="geocode_lookup_source_idx" + ), ] constraints = [ models.UniqueConstraint( @@ -272,7 +276,7 @@ class AiAnalysisCache(TimeStampedModel): status = models.CharField(max_length=16, choices=Status.choices) error_category = models.CharField(max_length=120, blank=True) summary_nl = models.TextField(blank=True) - features = models.JSONField(default=dict, blank=True) + features = models.JSONField(default=dict) warnings = models.JSONField(default=list, blank=True) class Meta: @@ -284,8 +288,11 @@ class AiAnalysisCache(TimeStampedModel): ) ] indexes = [ - models.Index(fields=["content_hash"]), - models.Index(fields=["model_name", "prompt_version", "schema_version"]), + models.Index(fields=["content_hash"], name="jobs_aianal_cache_content_idx"), + models.Index( + fields=["model_name", "prompt_version", "schema_version"], + name="jobs_aianal_cache_model_idx", + ), ] def __str__(self) -> str: @@ -364,9 +371,9 @@ class ApplicationTimelineEvent(TimeStampedModel): class Meta: ordering = ["-created_at"] indexes = [ - models.Index(fields=["application", "created_at"]), - models.Index(fields=["user", "created_at"]), + models.Index(fields=["application", "created_at"], name="jobs_app_timeline_app_idx"), + models.Index(fields=["user", "created_at"], name="jobs_app_timeline_user_idx"), ] def __str__(self) -> str: - return f"{self.application} — {self.event_type}" + return f"{self.application} — {self.event_type}" diff --git a/apps/jobs/services/ai.py b/apps/jobs/services/ai.py index 38105ee..647e28e 100644 --- a/apps/jobs/services/ai.py +++ b/apps/jobs/services/ai.py @@ -57,7 +57,13 @@ def _schema() -> dict[str, Any]: "seniority": {"type": "string"}, "evidence": {"type": "array", "items": {"type": "string"}}, }, - "required": ["support_ratio", "consultancy_ratio", "travel_ratio", "seniority", "evidence"], + "required": [ + "support_ratio", + "consultancy_ratio", + "travel_ratio", + "seniority", + "evidence", + ], }, "warnings": {"type": "array", "items": {"type": "string"}}, }, @@ -73,8 +79,8 @@ def _normalize_text(title: str, description: str) -> str: def _coerce_ratio(name: str, value: Any) -> float: try: ratio = float(value) - except (TypeError, ValueError): - raise ValueError(f"ratio:{name}") + except (TypeError, ValueError) as exc: + raise ValueError(f"ratio:{name}") from exc if not (MIN_FEATURE_VALUE <= ratio <= MAX_FEATURE_VALUE): raise ValueError(f"ratio:{name}") return round(ratio, 6) @@ -108,7 +114,9 @@ def _parse_and_validate_payload(payload: Any, *, title: str, description: str) - evidence_values = features.get("evidence") if not isinstance(evidence_values, list): raise ValueError("evidence") - normalized_evidence = [normalize_token(item) for item in evidence_values if isinstance(item, str)] + normalized_evidence = [ + normalize_token(item) for item in evidence_values if isinstance(item, str) + ] normalized_evidence = [item for item in normalized_evidence if item] if not normalized_evidence: raise ValueError("evidence") @@ -289,7 +297,9 @@ def analyze_job_text( ) try: - return _cache_get(content_hash=content_hash, model=model_name, prompt_version=prompt_version) + return _cache_get( + content_hash=content_hash, model=model_name, prompt_version=prompt_version + ) except AiUnavailable: pass diff --git a/apps/jobs/services/applications.py b/apps/jobs/services/applications.py index 9060333..96dea8c 100644 --- a/apps/jobs/services/applications.py +++ b/apps/jobs/services/applications.py @@ -26,7 +26,7 @@ SNAPSHOT_VERSION = "1.0.0" def _coerce_json_value(value: Any) -> Any: - if isinstance(value, (date,)): + if isinstance(value, date): return value.isoformat() if isinstance(value, Decimal): return float(value) @@ -85,7 +85,17 @@ def _latest_score_snapshot(job: JobPosting, user) -> dict[str, Any] | None: run = ( ScoreRun.objects.filter(job=job, profile=profile) .order_by("-created_at") - .only("score", "recommendation", "confidence", "components", "positives", "concerns", "hard_exclusions", "evidence", "profile_version") + .only( + "score", + "recommendation", + "confidence", + "components", + "positives", + "concerns", + "hard_exclusions", + "evidence", + "profile_version", + ) .first() ) if not run: @@ -121,7 +131,9 @@ def _build_application_snapshot_payload(application: Application) -> dict[str, A } -def _record_timeline_event(*, application: Application, user, event_type: str, metadata: dict[str, Any] | None = None) -> ApplicationTimelineEvent: +def _record_timeline_event( + *, application: Application, user, event_type: str, metadata: dict[str, Any] | None = None +) -> ApplicationTimelineEvent: payload: dict[str, Any] = {} if metadata: payload.update({key: value for key, value in metadata.items() if value is not None}) @@ -243,8 +255,11 @@ def track_application_changes( ) events += 1 - if (_normalize_str(previous.get("contact_name")) != _normalize_str(current.get("contact_name"))) or ( - _normalize_str(previous.get("contact_email")) != _normalize_str(current.get("contact_email")) + if ( + _normalize_str(previous.get("contact_name")) != _normalize_str(current.get("contact_name")) + ) or ( + _normalize_str(previous.get("contact_email")) + != _normalize_str(current.get("contact_email")) ): _record_timeline_event( application=application, @@ -266,16 +281,14 @@ def build_print_html(application: Application) -> str: safe_rows = [] for row in timeline_rows: safe_rows.append( - ( - f"" - f"{escape(row['timestamp'])}" - f"{escape(row['type'])}" - f"{escape(row['actor'])}" - f"{escape(row['from'])}" - f"{escape(row['to'])}" - f"{escape(row['note'])}" - "" - ) + f"" + f"{escape(row['timestamp'])}" + f"{escape(row['type'])}" + f"{escape(row['actor'])}" + f"{escape(row['from'])}" + f"{escape(row['to'])}" + f"{escape(row['note'])}" + "" ) source_rows = [] @@ -284,20 +297,26 @@ def build_print_html(application: Application) -> str: source_url = escape(str(source.get("url", ""))) source_rows.append(f"
  • {source_name}: {source_url}
  • ") source_section = "".join(source_rows) if source_rows else "
  • Niet beschikbaar
  • " + frozen_at = ( + application.snapshot.get("frozen_at", "") if isinstance(application.snapshot, dict) else "" + ) + snapshot_json = json.dumps(snapshot, indent=2, ensure_ascii=False, sort_keys=True) return ( "" "Sollicitatiedossier" f"

    {escape(job.original_title)}

    " f"

    Vacature: {escape(job.canonical_url)}

    " f"

    Status: {escape(application.get_status_display())}

    " - f"

    Geslaagd op: {escape(str(application.snapshot.get('frozen_at') if isinstance(application.snapshot, dict) else ''))}

    " + f"

    Geslaagd op: {escape(str(frozen_at))}

    " f"

    Bronnen

    " - f"

    Timeline

    " + "

    Timeline

    TijdTypeActorVanNaarNotitie
    " + "" f"{''.join(safe_rows)}
    TijdTypeActorVanNaarNotitie
    " - f"

    Snapshot

    {escape(json.dumps(snapshot, indent=2, ensure_ascii=False, sort_keys=True))}
    " + f"

    Snapshot

    {escape(snapshot_json)}
    " "" ) @@ -315,7 +334,9 @@ def build_application_export(application: Application) -> tuple[bytes, str]: "snapshot_version": SNAPSHOT_VERSION, "exported_at": timezone.now().isoformat(), "applied_at": application.applied_at.isoformat() if application.applied_at else None, - "follow_up_date": application.follow_up_date.isoformat() if application.follow_up_date else None, + "follow_up_date": application.follow_up_date.isoformat() + if application.follow_up_date + else None, }, "snapshot": _coerce_json_value(application.snapshot), "timeline": _timeline_dict_rows(application), @@ -332,7 +353,10 @@ def build_application_export(application: Application) -> tuple[bytes, str]: content_buffer = io.BytesIO() with zipfile.ZipFile(content_buffer, "w", compression=zipfile.ZIP_DEFLATED) as zf: - zf.writestr("application.json", json.dumps(_coerce_json_value(payload), indent=2, ensure_ascii=False, sort_keys=True)) + zf.writestr( + "application.json", + json.dumps(_coerce_json_value(payload), indent=2, ensure_ascii=False, sort_keys=True), + ) zf.writestr("timeline.csv", timeline_buffer.getvalue()) zf.writestr("application_print.html", build_print_html(application)) diff --git a/apps/jobs/services/dedupe.py b/apps/jobs/services/dedupe.py index 2c6cdee..b9648b4 100644 --- a/apps/jobs/services/dedupe.py +++ b/apps/jobs/services/dedupe.py @@ -3,10 +3,10 @@ from __future__ import annotations from dataclasses import dataclass, field from difflib import SequenceMatcher -from apps.sources.models import Source from django.db.models import Q from apps.jobs.models import JobPosting, JobSourceAlias +from apps.sources.models import Source from .employer_resolution import EmployerResolutionDecision, resolve_direct_employer_match from .normalization import CanonicalJobDraft, normalize_token diff --git a/apps/jobs/services/distance.py b/apps/jobs/services/distance.py index e0357f0..4b35b91 100644 --- a/apps/jobs/services/distance.py +++ b/apps/jobs/services/distance.py @@ -27,8 +27,7 @@ class CommuteEstimator(Protocol): name: str version: str - def estimate(self, distance_km: float) -> CommuteEstimate: - ... + def estimate(self, distance_km: float) -> CommuteEstimate: ... @dataclass(frozen=True) diff --git a/apps/jobs/services/employer_resolution.py b/apps/jobs/services/employer_resolution.py index 91f3fff..248fc26 100644 --- a/apps/jobs/services/employer_resolution.py +++ b/apps/jobs/services/employer_resolution.py @@ -9,7 +9,6 @@ from apps.sources.models import Source from .normalization import CanonicalJobDraft, normalize_token - MERGE_THRESHOLD = 0.96 TITLE_MIN_THRESHOLD = 0.92 CONFLICT_TITLE_THRESHOLD = 0.70 @@ -46,11 +45,13 @@ def _weighted_similarity( employer_domain_match: bool, canonical_host_match: bool, ) -> float: - weights: list[tuple[float, float]] = [(title_score, 0.68), (employer_domain_match and 1.0 or 0.0, 0.12)] + weights: list[tuple[float, float]] = [(title_score, 0.68)] if location_score is not None: weights.append((location_score, 0.12)) if employer_score is not None: weights.append((employer_score, 0.06)) + if employer_domain_match: + weights.append((1.0, 0.12)) if canonical_host_match: weights.append((1.0, 0.05)) total_weight = sum(weight for _, weight in weights) @@ -62,7 +63,10 @@ def _weighted_similarity( def _domain_match(value: str, candidate: str) -> bool: if not value or not candidate: return False - return _normalize_similarity(value).strip(".").lower() == _normalize_similarity(candidate).strip(".").lower() + return ( + _normalize_similarity(value).strip(".").lower() + == _normalize_similarity(candidate).strip(".").lower() + ) def _host(value: str) -> str: @@ -73,12 +77,18 @@ def resolve_direct_employer_match( draft: CanonicalJobDraft, *, source: Source | None ) -> EmployerResolutionDecision: if source is None or source.source_type == Source.Type.EMPLOYER or not draft.normalized_title: - return EmployerResolutionDecision(None, "no_direct_resolution", 0.0, canonical_url=None, conflict=False) + return EmployerResolutionDecision( + None, "no_direct_resolution", 0.0, canonical_url=None, conflict=False + ) - candidates = JobPosting.objects.filter( - status__in=[JobPosting.Status.ACTIVE, JobPosting.Status.NEW], - direct_employer=True, - ).select_related("employer").order_by("id") + candidates = ( + JobPosting.objects.filter( + status__in=[JobPosting.Status.ACTIVE, JobPosting.Status.NEW], + direct_employer=True, + ) + .select_related("employer") + .order_by("id") + ) best: JobPosting | None = None best_score = 0.0 @@ -92,7 +102,13 @@ def resolve_direct_employer_match( continue location_score: float | None = None - if draft.location_text and candidate.raw_location: + if ( + draft.municipality + and candidate.municipality + and normalize_token(draft.municipality) == normalize_token(candidate.municipality) + ): + location_score = 1.0 + elif draft.location_text and candidate.raw_location: location_score = _token_similarity(draft.location_text, candidate.raw_location) employer_score: float | None = None @@ -112,15 +128,23 @@ def resolve_direct_employer_match( canonical_host_match=canonical_host_match, ) - has_conflict = False - if draft.location_text and candidate.raw_location: - if location_score is not None and location_score < CONFLICT_LOCATION_THRESHOLD: - has_conflict = True - if draft.employer_name and candidate.employer_name and ( - employer_score is not None and employer_score < CONFLICT_EMPLOYER_THRESHOLD + has_conflict = bool( + draft.location_text + and candidate.raw_location + and location_score is not None + and location_score < CONFLICT_LOCATION_THRESHOLD + ) + if ( + draft.employer_name + and candidate.employer_name + and (employer_score is not None and employer_score < CONFLICT_EMPLOYER_THRESHOLD) ): has_conflict = True - if draft.employer_name and not candidate.employer_name and title_score < CONFLICT_TITLE_THRESHOLD: + if ( + draft.employer_name + and not candidate.employer_name + and title_score < CONFLICT_TITLE_THRESHOLD + ): has_conflict = True if score > best_score: diff --git a/apps/jobs/services/feedback.py b/apps/jobs/services/feedback.py index ca80855..9c5c1d1 100644 --- a/apps/jobs/services/feedback.py +++ b/apps/jobs/services/feedback.py @@ -1,18 +1,15 @@ from __future__ import annotations from dataclasses import dataclass -from datetime import timedelta from typing import Any from django.db import transaction -from django.utils import timezone -from apps.jobs.models import Application, Feedback, JobPosting, ScoreRun +from apps.jobs.models import Feedback, JobPosting, ScoreRun from apps.jobs.services.applications import apply_application_on_feedback from apps.profiles.models import SearchProfile from apps.profiles.services import apply_feedback_delta - LEARNING_MIN_SAMPLES = 2 LEARNING_DELTA_BY_ACTION = { Feedback.Action.INTERESTING: 1.0, @@ -47,15 +44,21 @@ def _latest_score_run(profile: SearchProfile, job: JobPosting) -> ScoreRun | Non def _best_signal_feature(score_run: ScoreRun | None) -> str: + if score_run is None: + return "content" components = { - key: value for key, value in (score_run.components or {}).items() if key in LEARNING_FEATURES + key: value + for key, value in (score_run.components or {}).items() + if key in LEARNING_FEATURES } if not components: return "content" return max(components, key=components.get) -def _classify_hide_signal(profile: SearchProfile, score_run: ScoreRun | None, reason: str) -> _LearningSignal: +def _classify_hide_signal( + profile: SearchProfile, score_run: ScoreRun | None, reason: str +) -> _LearningSignal: normalized = _normalize_reason_text(reason) if not normalized: return _LearningSignal( @@ -71,7 +74,9 @@ def _classify_hide_signal(profile: SearchProfile, score_run: ScoreRun | None, re reason_code="non_learning_title", learnable=False, ) - if any(token in normalized for token in ("afstand", "afstands", "km", "locatie", "verplaatsing")): + if any( + token in normalized for token in ("afstand", "afstands", "km", "locatie", "verplaatsing") + ): return _LearningSignal( feature="", delta=0.0, @@ -128,7 +133,9 @@ def _learning_signal_count(profile: SearchProfile, feature: str) -> int: return count -def _apply_learning_metadata(feedback: Feedback, signal: _LearningSignal, *, samples: int, applied: bool) -> None: +def _apply_learning_metadata( + feedback: Feedback, signal: _LearningSignal, *, samples: int, applied: bool +) -> None: metadata: dict[str, Any] = dict(feedback.metadata or {}) metadata["learning"] = { "status": "applied" if applied else "queued", @@ -187,7 +194,11 @@ def record_feedback( action=action, reason=reason[:200], ) - signal = _classify_learning_signal(profile=profile, job=job, action=action, reason=reason) if profile else None + signal = ( + _classify_learning_signal(profile=profile, job=job, action=action, reason=reason) + if profile + else None + ) if signal: _evaluate_learning(profile=profile, feedback=feedback, signal=signal) if action == Feedback.Action.APPLIED: diff --git a/apps/jobs/services/geocoding.py b/apps/jobs/services/geocoding.py index d28563e..4a99082 100644 --- a/apps/jobs/services/geocoding.py +++ b/apps/jobs/services/geocoding.py @@ -20,8 +20,7 @@ class GeocodeProvider(Protocol): confidence: float metadata: dict[str, Any] - def resolve(self, query: str) -> list["LocationMatch"]: - ... + def resolve(self, query: str) -> list[LocationMatch]: ... @dataclass(frozen=True) @@ -63,10 +62,7 @@ def parse_belgian_location_query(raw: str) -> tuple[str | None, str | None]: for chunk in re.findall(r"\b\d{4}\b", normalized): postal = chunk break - if "," in normalized: - municipality_part = normalized.split(",", 1)[0] - else: - municipality_part = normalized + municipality_part = normalized.split(",", 1)[0] if "," in normalized else normalized municipality_part = re.sub(r"\b\d{4}\b", " ", municipality_part) municipality_part = re.sub(r"[^a-z0-9 ]", " ", municipality_part) municipality = " ".join(municipality_part.split()) @@ -100,11 +96,12 @@ def _read_rows(path: str | Path) -> list[_ParsedRow]: with csv_path.open("r", encoding="utf-8-sig", newline="") as handle: reader = csv.DictReader(handle) - headers = set((reader.fieldnames or [])) + headers = set(reader.fieldnames or []) required = {"postal_code", "municipality", "region", "latitude", "longitude"} if not required.issubset(headers): raise CommandError( - "Verplichte kolommen ontbreken: postal_code, municipality, region, latitude, longitude" + "Verplichte kolommen ontbreken: postal_code, municipality, region, " + "latitude, longitude" ) for row_number, raw_row in enumerate(reader, start=2): @@ -116,7 +113,9 @@ def _read_rows(path: str | Path) -> list[_ParsedRow]: if not postal_code: raise CommandError(f"regel {row_number}: postal_code mag niet leeg zijn") if len(postal_code) != 4 or not postal_code.isdigit(): - raise CommandError(f"regel {row_number}: ongeldige Belgische postcode {postal_code}") + raise CommandError( + f"regel {row_number}: ongeldige Belgische postcode {postal_code}" + ) if not municipality: raise CommandError(f"regel {row_number}: municipality mag niet leeg zijn") @@ -136,7 +135,8 @@ def _read_rows(path: str | Path) -> list[_ParsedRow]: row_key = (postal_code, normalized_municipality) if row_key in seen: raise CommandError( - f"regel {row_number}: dubbel record in bestand voor {postal_code} {municipality}" + f"regel {row_number}: dubbel record in bestand voor " + f"{postal_code} {municipality}" ) seen.add(row_key) @@ -171,7 +171,9 @@ class CsvGeocodeProvider: self.metadata: dict[str, Any] = metadata or {} @staticmethod - def _load_candidates(query_value: str, query_kind: str, *, source_name: str, source_version: str): + def _load_candidates( + query_value: str, query_kind: str, *, source_name: str, source_version: str + ): return GeocodeLocationLookup.objects.filter( source_name=source_name, source_version=source_version, @@ -307,7 +309,9 @@ def import_csv_geodata( rows = _read_rows(path) if len(rows) > 50000: - raise CommandError("Importbestand bevat meer dan 50.000 records; import in delen aanbevolen.") + raise CommandError( + "Importbestand bevat meer dan 50.000 records; import in delen aanbevolen." + ) existing_rows = GeocodeLocationLookup.objects.filter( source_name=source_name, @@ -347,10 +351,7 @@ def import_csv_geodata( row.postal_code, row.municipality, ) - if (not replace) and ( - municipal_key in existing_keys - or postal_key in existing_keys - ): + if (not replace) and (municipal_key in existing_keys or postal_key in existing_keys): raise CommandError( "Import zou bestaande lookuprecords overschrijven zonder --replace." ) diff --git a/apps/jobs/services/pipeline.py b/apps/jobs/services/pipeline.py index 13bc0ca..f8a0820 100644 --- a/apps/jobs/services/pipeline.py +++ b/apps/jobs/services/pipeline.py @@ -271,7 +271,9 @@ def persist_draft( else: alias.last_seen = timezone.now() alias.raw_document = document - alias.payload = _alias_payload(alias.payload, decision, fallback_canonical_url=draft.canonical_url) + alias.payload = _alias_payload( + alias.payload, decision, fallback_canonical_url=draft.canonical_url + ) if direct: alias.is_canonical = True alias.save( diff --git a/apps/jobs/services/scoring.py b/apps/jobs/services/scoring.py index a833efc..07b6a1f 100644 --- a/apps/jobs/services/scoring.py +++ b/apps/jobs/services/scoring.py @@ -1,15 +1,16 @@ from __future__ import annotations +from collections.abc import Iterable from dataclasses import dataclass from difflib import SequenceMatcher -from typing import Any, Iterable +from typing import Any from django.db import transaction +from apps.jobs.models import JobPosting, ScoreRun from apps.jobs.services.ai import AiAnalysis, AiAnalysisCache, analyze_job_text from apps.jobs.services.distance import estimate_commute, haversine_km from apps.jobs.services.geocoding import resolve_cached_location -from apps.jobs.models import JobPosting, ScoreRun from apps.profiles.models import SearchProfile from .normalization import normalize_token @@ -106,7 +107,9 @@ def _profile_reference(profile: SearchProfile) -> _GeoReference | None: def _title_fit(job: JobPosting, profile: SearchProfile) -> float: if not profile.desired_titles: return 0.65 - return max(_similarity(job.normalized_title, desired_title) for desired_title in profile.desired_titles) + return max( + _similarity(job.normalized_title, desired_title) for desired_title in profile.desired_titles + ) def _skill_fit(job: JobPosting, profile: SearchProfile) -> tuple[float, list[str], list[str]]: @@ -191,9 +194,15 @@ def _hard_exclusions( ): reasons.append(f"Uitgesloten regio: {job.region or job.municipality}") - if distance.exact_distance_km is not None and distance.exact_distance_km > distance_limit: - if job.workplace_type != "remote": - reasons.append(f"Afstand {distance.exact_distance_km:.0f} km boven maximum {profile.max_distance_km} km") + if ( + distance.exact_distance_km is not None + and distance.exact_distance_km > distance_limit + and job.workplace_type != "remote" + ): + reasons.append( + f"Afstand {distance.exact_distance_km:.0f} km boven maximum " + f"{profile.max_distance_km} km" + ) max_commute_minutes = profile.hard_rules.get("max_commute_minutes") try: @@ -206,7 +215,8 @@ def _hard_exclusions( and distance.commute_minutes > max_commute_limit ): reasons.append( - f"Geschatte reistijd {distance.commute_minutes} minuten boven limiet van {max_commute_limit}" + f"Geschatte reistijd {distance.commute_minutes} minuten boven limiet van " + f"{max_commute_limit}" ) excluded_skills = { @@ -244,7 +254,9 @@ def _ai_feature_score(features: dict[str, Any]) -> float: "unknown": 0.03, "": 0.03, }.get(seniority, 0.05) - score = 0.5 * (1.0 - support_ratio) + 0.25 * (1.0 - consultancy_ratio) + 0.15 * (1.0 - travel_ratio) + score = ( + 0.5 * (1.0 - support_ratio) + 0.25 * (1.0 - consultancy_ratio) + 0.15 * (1.0 - travel_ratio) + ) return max(0.0, min(1.0, score + seniority_boost)) @@ -373,7 +385,10 @@ def calculate_score(job: JobPosting, profile: SearchProfile) -> ScoreResult: positives.append("Herkenbare skills: " + ", ".join(present_skills[:6])) if job.direct_employer and not job.recruiter: positives.append("Rechtstreekse werkgeversbron.") - if distance.exact_distance_km is not None and distance.exact_distance_km <= profile.max_distance_km: + if ( + distance.exact_distance_km is not None + and distance.exact_distance_km <= profile.max_distance_km + ): positives.append(f"Binnen de ingestelde afstand ({distance.exact_distance_km:.0f} km).") if ( distance.exact_distance_km is None @@ -382,13 +397,18 @@ def calculate_score(job: JobPosting, profile: SearchProfile) -> ScoreResult: ): estimate_label = "geschatte" if distance.commute_estimate else "ingeschatte" concerns.append( - f"Schatting: {estimate_label} reistijd ca. {distance.commute_minutes} min (conservatief)." + f"Schatting: {estimate_label} reistijd ca. {distance.commute_minutes} min " + "(conservatief)." ) if support_ratio >= 0.5: concerns.append("Vacature bevat sterke first-line/helpdesksignalen.") if missing_skills: concerns.append("Niet duidelijk teruggevonden: " + ", ".join(missing_skills[:6])) - if distance.exact_distance_km is None and distance.has_distance_data and job.workplace_type != "remote": + if ( + distance.exact_distance_km is None + and distance.has_distance_data + and job.workplace_type != "remote" + ): concerns.append("Afstand kon nog niet exact betrouwbaar worden berekend.") if not job.compensation: concerns.append("Salaris of barema is niet vermeld.") @@ -436,7 +456,9 @@ def calculate_score(job: JobPosting, profile: SearchProfile) -> ScoreResult: "warnings": ai_analysis.warnings, "features": ai_analysis.features, "weight_requested": float(profile.weights.get("ai", 0) or 0), - "weight_applied": ai_weight if ai_analysis.status == AiAnalysisCache.Status.OK else 0.0, + "weight_applied": ai_weight + if ai_analysis.status == AiAnalysisCache.Status.OK + else 0.0, }, }, model_version=ai_analysis.model, @@ -451,9 +473,7 @@ def _iter_active_profiles(profile_id: int | None): return profiles -def rescore_jobs_with_profiles( - jobs: Iterable[JobPosting], *, profile_id: int | None = None -) -> int: +def rescore_jobs_with_profiles(jobs: Iterable[JobPosting], *, profile_id: int | None = None) -> int: count = 0 for profile in _iter_active_profiles(profile_id): for job in jobs: diff --git a/apps/jobs/urls.py b/apps/jobs/urls.py index 4c0a603..c5fd904 100644 --- a/apps/jobs/urls.py +++ b/apps/jobs/urls.py @@ -3,11 +3,11 @@ from django.urls import path from .views import ( ApplicationListView, ApplicationUpdateView, + JobDetailView, + JobListView, application_delete, application_export, application_print, - JobDetailView, - JobListView, job_feedback, ) diff --git a/apps/jobs/views.py b/apps/jobs/views.py index 5fcb477..3ebe4a5 100644 --- a/apps/jobs/views.py +++ b/apps/jobs/views.py @@ -4,7 +4,7 @@ from django.contrib import messages from django.contrib.auth.decorators import login_required from django.contrib.auth.mixins import LoginRequiredMixin from django.db.models import Q -from django.http import HttpResponse +from django.http import Http404, HttpResponse from django.shortcuts import get_object_or_404, redirect from django.urls import reverse from django.views.decorators.http import require_POST @@ -118,12 +118,11 @@ class ApplicationUpdateView(LoginRequiredMixin, UpdateView): return Application.objects.filter(user=self.request.user).select_related("job") def form_valid(self, form): - previous = { - "status": self.object.status, - "notes": self.object.notes, - "contact_name": self.object.contact_name, - "contact_email": self.object.contact_email, - } + previous = ( + Application.objects.filter(pk=self.object.pk) + .values("status", "notes", "contact_name", "contact_email") + .get() + ) response = super().form_valid(form) track_application_changes( application=self.object, @@ -165,6 +164,8 @@ def application_print(request, pk: int): @login_required @require_POST def application_delete(request, pk: int): + if Application.objects.filter(pk=pk).exclude(user=request.user).exists(): + raise Http404 deleted = delete_application_dossier(application_id=pk, user=request.user) if deleted: messages.success(request, "Sollicitatiedossier verwijderd.") diff --git a/apps/notifications/models.py b/apps/notifications/models.py index 8987ff4..200d5b6 100644 --- a/apps/notifications/models.py +++ b/apps/notifications/models.py @@ -68,10 +68,18 @@ class ReminderOutbox(TimeStampedModel): payload = models.JSONField(default=dict) status = models.CharField(max_length=16, choices=Status.choices, default=Status.PENDING) job = models.ForeignKey( - "jobs.JobPosting", on_delete=models.SET_NULL, null=True, blank=True, related_name="reminders" + "jobs.JobPosting", + on_delete=models.SET_NULL, + null=True, + blank=True, + related_name="reminders", ) application = models.ForeignKey( - "jobs.Application", on_delete=models.SET_NULL, null=True, blank=True, related_name="reminders" + "jobs.Application", + on_delete=models.SET_NULL, + null=True, + blank=True, + related_name="reminders", ) score_run = models.ForeignKey( "jobs.ScoreRun", on_delete=models.SET_NULL, null=True, blank=True, related_name="reminders" diff --git a/apps/notifications/services.py b/apps/notifications/services.py index 5b67d7a..9bfc847 100644 --- a/apps/notifications/services.py +++ b/apps/notifications/services.py @@ -15,7 +15,6 @@ from apps.profiles.models import SearchProfile from .models import DigestOutbox, ReminderOutbox - TOP_MATCH_MIN_CONFIDENCE = Decimal("0.85") @@ -27,7 +26,8 @@ def _to_local(profile: SearchProfile, *, now: datetime | None = None) -> datetim def _hidden_job_ids(profile: SearchProfile) -> set[str]: return set( - str(job_id) for job_id in Feedback.objects.filter( + str(job_id) + for job_id in Feedback.objects.filter( user=profile.user, action=Feedback.Action.HIDE ).values_list("job_id", flat=True) ) @@ -35,7 +35,8 @@ def _hidden_job_ids(profile: SearchProfile) -> set[str]: def _applied_job_ids(profile: SearchProfile) -> set[str]: return set( - str(job_id) for job_id in Application.objects.filter( + str(job_id) + for job_id in Application.objects.filter( user=profile.user, status=Application.Status.APPLIED ).values_list("job_id", flat=True) ) @@ -234,7 +235,9 @@ def create_closing_reminders(profile: SearchProfile, *, now=None) -> list[Remind if not profile.reminders_enabled: return [] now_local = _to_local(profile, now=now) - due_start = datetime.combine(now_local.date(), time.min, tzinfo=now_local.tzinfo).astimezone(UTC) + due_start = datetime.combine(now_local.date(), time.min, tzinfo=now_local.tzinfo).astimezone( + UTC + ) due_end = due_start + timedelta(days=1) hidden_ids = _hidden_job_ids(profile) applied_ids = _applied_job_ids(profile) @@ -255,9 +258,7 @@ def create_closing_reminders(profile: SearchProfile, *, now=None) -> list[Remind local_valid_through = ( job.valid_through.astimezone(now_local.tzinfo) if job.valid_through else now_local ) - dedupe_key = ( - f"closing:{profile.pk}:{job.pk}:{local_valid_through.date().isoformat()}" - ) + dedupe_key = f"closing:{profile.pk}:{job.pk}:{local_valid_through.date().isoformat()}" outboxes.append( _build_reminder_outbox( profile, @@ -324,7 +325,9 @@ def create_follow_up_reminders(profile: SearchProfile, *, now=None) -> list[Remi "employer": application.job.employer_name, "link": application.job.canonical_url, "follow_up_date": ( - application.follow_up_date.isoformat() if application.follow_up_date else None + application.follow_up_date.isoformat() + if application.follow_up_date + else None ), }, job=application.job, diff --git a/apps/sources/adapters/__init__.py b/apps/sources/adapters/__init__.py index 444976e..ce189e1 100644 --- a/apps/sources/adapters/__init__.py +++ b/apps/sources/adapters/__init__.py @@ -1,7 +1,3 @@ -from .base import ExtractedJob, ExtractionResult -from .generic_html import GenericHtmlAdapter -from .jsonld import JsonLdJobPostingAdapter -from .rss import RssAdapter from .ats import ( GreenhouseAdapter, LeverAdapter, @@ -9,16 +5,20 @@ from .ats import ( SmartRecruitersAdapter, WorkableAdapter, ) +from .base import ExtractedJob, ExtractionResult +from .generic_html import GenericHtmlAdapter +from .jsonld import JsonLdJobPostingAdapter +from .rss import RssAdapter __all__ = [ "ExtractedJob", "ExtractionResult", "GenericHtmlAdapter", - "JsonLdJobPostingAdapter", - "RssAdapter", "GreenhouseAdapter", + "JsonLdJobPostingAdapter", "LeverAdapter", "RecruiteeAdapter", + "RssAdapter", "SmartRecruitersAdapter", "WorkableAdapter", ] diff --git a/apps/sources/adapters/ats.py b/apps/sources/adapters/ats.py index 4b3a64c..e932dfc 100644 --- a/apps/sources/adapters/ats.py +++ b/apps/sources/adapters/ats.py @@ -1,4 +1,4 @@ -from __future__ import annotations +from __future__ import annotations import json from abc import ABC, abstractmethod @@ -8,7 +8,6 @@ from bs4 import BeautifulSoup from .base import ExtractedJob, ExtractionResult, FieldEvidence - CLOSED_STATUSES = { "closed", "inactive", @@ -51,7 +50,7 @@ def _first_text(data, *candidates): value = data.get(candidate) if isinstance(data, dict) else None if value is None: continue - if isinstance(value, (list, tuple)): + if isinstance(value, list | tuple): for item in value: text = _to_text(item) if text: @@ -114,7 +113,7 @@ def _as_text(value) -> str: def _extract_payload(content: str): try: - return json.loads(content) + return json.loads(content.lstrip("\ufeff")) except json.JSONDecodeError: pass soup = BeautifulSoup(content, "lxml") @@ -123,7 +122,7 @@ def _extract_payload(content: str): if not script_text: continue try: - return json.loads(script_text) + return json.loads(script_text.lstrip("\ufeff")) except json.JSONDecodeError: continue return None @@ -139,8 +138,7 @@ class _AtsAdapter(ABC): closed_statuses = CLOSED_STATUSES @abstractmethod - def _extract_records(self, payload) -> list[dict[str, object]]: - ... + def _extract_records(self, payload) -> list[dict[str, object]]: ... def _supports_url(self, url: str) -> bool: host = (urlsplit(url).hostname or "").lower() @@ -191,7 +189,7 @@ class _AtsAdapter(ABC): location_raw = ", ".join(location_values) location = location_raw - location_parts = [part.strip() for part in _to_text(location).split(",") if part.strip()] + [part.strip() for part in _to_text(location).split(",") if part.strip()] region = _first_text( record, "region", @@ -343,7 +341,7 @@ class _AtsAdapter(ABC): valid_through=valid_through, employment_types=employment_types, workplace_type=workplace_type, - raw=record, + raw=record.get("__raw__", record), evidence=evidence, ) @@ -389,7 +387,9 @@ class _AtsAdapter(ABC): warnings: list[str] = [] if not jobs: warnings.append("Geen actieve ATS-vacatures gevonden") - return ExtractionResult(jobs, self.parser_key, self.parser_version, 0.9 if jobs else 0.0, warnings) + return ExtractionResult( + jobs, self.parser_key, self.parser_version, 0.9 if jobs else 0.0, warnings + ) class GreenhouseAdapter(_AtsAdapter): @@ -452,6 +452,12 @@ class LeverAdapter(_AtsAdapter): break value = value[key] if isinstance(value, dict): + if path == ("position",): + raw_payload = { + "position": value, + "work_type": _to_text(value.get("workplaceType")), + } + return [{**value, "__raw__": raw_payload}] return [value] return [] diff --git a/apps/sources/adapters/registry.py b/apps/sources/adapters/registry.py index f811d36..472bb7a 100644 --- a/apps/sources/adapters/registry.py +++ b/apps/sources/adapters/registry.py @@ -1,8 +1,7 @@ -from __future__ import annotations +from __future__ import annotations from apps.sources.models import RawDocument -from .base import ExtractionResult from .ats import ( GreenhouseAdapter, LeverAdapter, @@ -10,6 +9,7 @@ from .ats import ( SmartRecruitersAdapter, WorkableAdapter, ) +from .base import ExtractionResult from .generic_html import GenericHtmlAdapter from .jsonld import JsonLdJobPostingAdapter from .rss import RssAdapter @@ -45,7 +45,10 @@ class AdapterRegistry: result = provider.extract(content, url=url) if any( msg in result.warnings - for msg in ("Geen parseerbare ATS-response", "Geen herkenbare ATS-markup voor deze adapter") + for msg in ( + "Geen parseerbare ATS-response", + "Geen herkenbare ATS-markup voor deze adapter", + ) ): continue return result diff --git a/apps/sources/models.py b/apps/sources/models.py index 67a9777..6bf03ba 100644 --- a/apps/sources/models.py +++ b/apps/sources/models.py @@ -1,7 +1,6 @@ from __future__ import annotations from datetime import timedelta -from uuid import uuid4 from urllib.parse import urlparse from django.conf import settings @@ -91,7 +90,7 @@ class Source(TimeStampedModel): return [entry for entry in entries if isinstance(entry, dict)] @property - def latest_policy_review(self) -> "SourcePolicyReview | None": + def latest_policy_review(self) -> SourcePolicyReview | None: return self.policy_reviews.order_by("-created_at").first() @property @@ -109,27 +108,37 @@ class Source(TimeStampedModel): if not review: return "Geen review geregistreerd" if review.is_expired: - return f"Review verlopen op {review.expires_at:%Y-%m-%d}" if review.expires_at else "Review verlopen" + return ( + f"Review verlopen op {review.expires_at:%Y-%m-%d}" + if review.expires_at + else "Review verlopen" + ) if review.expires_at: return f"{review.get_decision_display()} geldig tot {review.expires_at:%Y-%m-%d}" return f"{review.get_decision_display()} zonder vervaldatum" @property def requires_terms_review(self) -> bool: - return self.policy == Source.Policy.ALLOW and self.policy_review_state in {"missing", "expired"} + return self.policy == Source.Policy.ALLOW and self.policy_review_state in { + "missing", + "expired", + } def schedule_after_success(self, *, now=None, jitter_seconds: int = 0) -> None: now = now or timezone.now() self.last_success_at = now self.failure_count = 0 - self.next_run_at = now + timedelta(minutes=self.crawl_interval_minutes) + timedelta( - seconds=max(0, jitter_seconds) + self.next_run_at = ( + now + + timedelta(minutes=self.crawl_interval_minutes) + + timedelta(seconds=max(0, jitter_seconds)) ) self.save(update_fields=["last_success_at", "failure_count", "next_run_at", "updated_at"]) def schedule_after_failure( self, - *, now=None, + *, + now=None, backoff_minutes: int | None = None, backoff_seconds: int | None = None, ) -> None: @@ -186,7 +195,7 @@ class SourceRun(TimeStampedModel): class SourceLease(TimeStampedModel): source = models.OneToOneField(Source, on_delete=models.CASCADE, related_name="lease") - token = models.CharField(max_length=64, default=lambda: str(uuid4())) + token = models.CharField(max_length=64, default="") worker_id = models.CharField(max_length=128, blank=True) expires_at = models.DateTimeField(db_index=True) diff --git a/apps/sources/services/discovery.py b/apps/sources/services/discovery.py index ea42c2f..22952e3 100644 --- a/apps/sources/services/discovery.py +++ b/apps/sources/services/discovery.py @@ -4,11 +4,11 @@ import ipaddress import json import re from dataclasses import dataclass -from datetime import datetime, timezone +from datetime import UTC, datetime from urllib.parse import urljoin, urlsplit -from xml.etree import ElementTree from bs4 import BeautifulSoup +from defusedxml import ElementTree from django.db import transaction from apps.sources.adapters.email_alert import EmailAlertAdapter @@ -166,15 +166,14 @@ def persist_discovery_candidates(candidates: list[SourceCandidate]) -> tuple[int evidence = metadata.get("discovery", []) if not isinstance(evidence, list): evidence = [] - if not any(item.get("url") == candidate.url for item in evidence if isinstance(item, dict)): + if not any( + item.get("url") == candidate.url for item in evidence if isinstance(item, dict) + ): evidence.append(provenance) metadata["discovery"] = evidence[-20:] metadata["discovered_at"] = now source.metadata = metadata updated_fields.append("metadata") - else: - skipped += 1 - if not source.name: source.name = _candidate_name(candidate) updated_fields.append("name") @@ -223,19 +222,24 @@ def _discover_html( for anchor in soup.find_all("a", href=True): label = " ".join(anchor.get_text(" ", strip=True).split()) + raw_url = urljoin(base_url, str(anchor["href"])) + candidate_host = _hostname(raw_url) + same_domain = bool(base_domain and domain_matches(candidate_host, base_domain)) candidate = _build_candidate( - raw_url=urljoin(base_url, str(anchor["href"])), + raw_url=raw_url, base_domain=base_domain, source_type=Source.Type.EMPLOYER, label=label, reason="career-link", discovered_from="html", - confidence=0.9, - allow_off_domain=False, + confidence=0.9 if same_domain else 0.7, + allow_off_domain=True, ) if not candidate: continue - if not CAREER_PATTERN.search(label) and not CAREER_PATTERN.search(urlsplit(candidate.url).path): + if not CAREER_PATTERN.search(label) and not CAREER_PATTERN.search( + urlsplit(candidate.url).path + ): continue candidates.append(candidate) @@ -388,4 +392,4 @@ def _to_domain_root_url(url: str) -> str: def _utc_now() -> str: - return datetime.now(timezone.utc).isoformat() + return datetime.now(UTC).isoformat() diff --git a/apps/sources/services/fetcher.py b/apps/sources/services/fetcher.py index a6bea8f..0587a47 100644 --- a/apps/sources/services/fetcher.py +++ b/apps/sources/services/fetcher.py @@ -1,10 +1,11 @@ from __future__ import annotations import hashlib -from datetime import datetime, timezone as utc -from email.utils import parsedate_to_datetime from collections.abc import Mapping from dataclasses import dataclass +from datetime import datetime +from datetime import timezone as utc +from email.utils import parsedate_to_datetime from urllib.parse import urljoin import httpx @@ -13,7 +14,7 @@ from django.conf import settings from apps.sources.models import Source from .policy import assess_url -from .url_security import validate_public_url +from .url_security import ValidatedUrl, validate_public_url ALLOWED_CONTENT_TYPES = ( "text/html", @@ -124,12 +125,19 @@ def fetch_url( headers=headers, ) current_url = url + previous_validation: ValidatedUrl | None = None try: for _ in range(settings.FETCHER_MAX_REDIRECTS + 1): validation = validate_public_url( current_url, allow_nonstandard_ports=settings.FETCHER_ALLOW_NONSTANDARD_PORTS, ) + if ( + previous_validation is not None + and validation.hostname == previous_validation.hostname + and not set(validation.addresses).intersection(previous_validation.addresses) + ): + raise FetchError("DNS-rebindcontrole faalde bij redirect naar dezelfde host.") response = http_client.get(current_url, headers=headers) post_validation = validate_public_url( str(response.url) if response.url else current_url, @@ -145,6 +153,7 @@ def fetch_url( next_decision = assess_url(current_url, source=source) if not next_decision.allowed: raise PolicyBlockedError(next_decision.reason) + previous_validation = validation continue if response.status_code == 304: return FetchedDocument(url, current_url, 304, dict(response.headers), b"") diff --git a/apps/sources/services/health.py b/apps/sources/services/health.py index 04ea2bc..b87a2a1 100644 --- a/apps/sources/services/health.py +++ b/apps/sources/services/health.py @@ -174,7 +174,9 @@ def _is_parser_drift_run(run: SourceRun) -> bool: def _determine_health_action(runs: list[SourceRun]) -> tuple[str | None, str | None]: recent_runs = runs[:SOURCE_HEALTH_RUN_WINDOW] considered_failures = [ - run for run in recent_runs if run.status in {SourceRun.Status.FAILED, SourceRun.Status.SKIPPED} + run + for run in recent_runs + if run.status in {SourceRun.Status.FAILED, SourceRun.Status.SKIPPED} ] if any( @@ -209,7 +211,9 @@ def _determine_health_action(runs: list[SourceRun]) -> tuple[str | None, str | N return None, None -def collect_source_health(*, runs_to_consider: int = SOURCE_HEALTH_RUN_WINDOW) -> list[SourceHealth]: +def collect_source_health( + *, runs_to_consider: int = SOURCE_HEALTH_RUN_WINDOW +) -> list[SourceHealth]: sources = Source.objects.order_by("name").all() rows: list[SourceHealth] = [] @@ -256,7 +260,11 @@ def collect_source_health(*, runs_to_consider: int = SOURCE_HEALTH_RUN_WINDOW) - def evaluate_source_health(*, now: datetime | None = None) -> dict[str, int]: now = now or timezone.now() - rows = [row for row in collect_source_health() if row.source_status in {Source.Status.ACTIVE, Source.Status.TRIAL}] + rows = [ + row + for row in collect_source_health() + if row.source_status in {Source.Status.ACTIVE, Source.Status.TRIAL} + ] counts = {"evaluated": len(rows), "quarantined": 0} for row in rows: @@ -309,7 +317,9 @@ def canary_recovery_sources(*, now: datetime | None = None) -> list[Source]: if bool(health.get("canary_started", False)): continue - recovery_due = _from_iso(health.get("recovery_due_at") if isinstance(health, dict) else None) + recovery_due = _from_iso( + health.get("recovery_due_at") if isinstance(health, dict) else None + ) if recovery_due and recovery_due > now: continue diff --git a/apps/sources/services/manual_import.py b/apps/sources/services/manual_import.py index 04392e1..325fcae 100644 --- a/apps/sources/services/manual_import.py +++ b/apps/sources/services/manual_import.py @@ -18,9 +18,9 @@ from apps.jobs.services.pipeline import process_raw_document from apps.sources.models import RawDocument, Source, SourcePolicyReview, SourceRun from apps.sources.services.canonicalize import canonicalize_url from apps.sources.services.fetcher import ( + FetchedDocument, FetchError, FetchTimeoutError, - FetchedDocument, PolicyBlockedError, RateLimitedError, fetch_url, @@ -88,11 +88,7 @@ def _mode_source_name(domain: str, *, mode: str) -> str: def _ensure_manual_review(source: Source, actor) -> None: review = source.latest_policy_review - if ( - review - and not review.is_expired - and review.decision == SourcePolicyReview.Decision.ALLOW - ): + if review and not review.is_expired and review.decision == SourcePolicyReview.Decision.ALLOW: return create_policy_review( source, @@ -120,9 +116,7 @@ def _ensure_manual_source(*, domain: str, source_url: str, actor, mode: str) -> source.base_url = source_url source.status = Source.Status.CANDIDATE source.policy = Source.Policy.ALLOW - source.save( - update_fields=["name", "base_url", "status", "policy", "updated_at"] - ) + source.save(update_fields=["name", "base_url", "status", "policy", "updated_at"]) _ensure_manual_review(source, actor=actor) return source @@ -175,7 +169,7 @@ def _run_pipeline(document: RawDocument, *, source_run: SourceRun) -> ManualImpo source = document.source source_name = source.name if source else "" warnings: list[str] = list(metrics.get("warnings", [])) - warnings_count = len(warnings) + len(warnings) source_run.finish( SourceRun.Status.SUCCESS, http_status=document.source_run.http_status if document.source_run else None, @@ -186,9 +180,8 @@ def _run_pipeline(document: RawDocument, *, source_run: SourceRun) -> ManualImpo metrics={"parser": metrics["parser"], "warnings": warnings}, ) jobs = _build_jobs_from_document(document) - if metrics["created"] == 0 and metrics["updated"] == 0: - if not warnings: - warnings.append("De bron leverde geen herkenbare vacaturedata op.") + if metrics["created"] == 0 and metrics["updated"] == 0 and not warnings: + warnings.append("De bron leverde geen herkenbare vacaturedata op.") if not warnings: # keep stable, machine-readable payload shape warnings = [] diff --git a/apps/sources/services/policy.py b/apps/sources/services/policy.py index e2288c7..da738e8 100644 --- a/apps/sources/services/policy.py +++ b/apps/sources/services/policy.py @@ -79,18 +79,28 @@ def _check_review_gate(source: Source) -> PolicyDecision | None: if not review: return PolicyDecision(False, Source.Policy.REVIEW, "Review vereist") if review.decision == SourcePolicyReview.Decision.DENY: - return PolicyDecision(False, Source.Policy.DENY, review.reason or "Review blokkeert bron") + return PolicyDecision( + False, Source.Policy.DENY, review.reason or "Review blokkeert bron" + ) if review.decision == SourcePolicyReview.Decision.PAUSE: - return PolicyDecision(False, Source.Policy.REVIEW, review.reason or "Review vraagt pauze") + return PolicyDecision( + False, Source.Policy.REVIEW, review.reason or "Review vraagt pauze" + ) return None if source.policy == Source.Policy.ALLOW: if not review: - return PolicyDecision(False, Source.Policy.REVIEW, "Review ontbreekt voor actief beleid") + return PolicyDecision( + False, Source.Policy.REVIEW, "Review ontbreekt voor actief beleid" + ) if review.decision == SourcePolicyReview.Decision.DENY: - return PolicyDecision(False, Source.Policy.DENY, review.reason or "Review blokkeert bron") + return PolicyDecision( + False, Source.Policy.DENY, review.reason or "Review blokkeert bron" + ) if review.decision == SourcePolicyReview.Decision.PAUSE: - return PolicyDecision(False, Source.Policy.REVIEW, review.reason or "Review vraagt pauze") + return PolicyDecision( + False, Source.Policy.REVIEW, review.reason or "Review vraagt pauze" + ) return None diff --git a/apps/sources/services/robots.py b/apps/sources/services/robots.py index 901b13d..284f0d8 100644 --- a/apps/sources/services/robots.py +++ b/apps/sources/services/robots.py @@ -10,6 +10,7 @@ from django.conf import settings from django.utils import timezone from apps.sources.models import Source, SourceRobotsCache + from .url_security import UnsafeUrlError, validate_public_url ALLOW = "allow" @@ -30,7 +31,11 @@ def _origin_for(url: str) -> str: if not host: raise ValueError("Host ontbreekt voor robotscontrole.") port = parts.port - if (parts.scheme == "http" and port == 80) or (parts.scheme == "https" and port == 443) or not port: + if ( + (parts.scheme == "http" and port == 80) + or (parts.scheme == "https" and port == 443) + or not port + ): netloc = host else: netloc = f"{host}:{port}" @@ -61,10 +66,7 @@ def _rules_from_text(text: str) -> dict[str, dict[str, list[str]]]: key_lower = key.lower() if key_lower == "user-agent": token = value.lower() - if token: - active_agents = {token} - else: - active_agents = set() + active_agents = {token} if token else set() continue if key_lower not in {ALLOW, DISALLOW}: continue @@ -85,7 +87,7 @@ def _pick_rules(rules: dict[str, dict[str, list[str]]], user_agent: str) -> dict selected = {ALLOW: [], DISALLOW: []} for agent, values in rules.items(): - if agent == "*" or agent and agent in normalized: + if agent == "*" or (agent and agent in normalized): selected[ALLOW].extend(values[ALLOW]) selected[DISALLOW].extend(values[DISALLOW]) @@ -95,10 +97,14 @@ def _pick_rules(rules: dict[str, dict[str, list[str]]], user_agent: str) -> dict def _longest_prefix(path: str, rules: Iterable[str]) -> int: - return max((len(rule.rstrip("/")) for rule in rules if rule and path.startswith(rule)), default=0) + return max( + (len(rule.rstrip("/")) for rule in rules if rule and path.startswith(rule)), default=0 + ) -def _evaluate_path(path: str, user_agent: str, rules: dict[str, dict[str, list[str]]]) -> RobotsDecision: +def _evaluate_path( + path: str, user_agent: str, rules: dict[str, dict[str, list[str]]] +) -> RobotsDecision: selected = _pick_rules(rules, user_agent=user_agent) allow_len = _longest_prefix(path, selected[ALLOW]) disallow_len = _longest_prefix(path, selected[DISALLOW]) @@ -184,12 +190,7 @@ def _persist_cache( }, )[0] - if status_code in {404, 410}: - rules = {} - elif status_code >= 400: - rules = {} - else: - rules = _rules_from_text(content) + rules = {} if status_code in {404, 410} or status_code >= 400 else _rules_from_text(content) return SourceRobotsCache.objects.update_or_create( origin=origin, @@ -257,12 +258,18 @@ def assess_robots( return RobotsDecision(False, f"Robotscontrole mislukt: {exc}") except httpx.HTTPError as exc: if stale is not None: - return _evaluate_path(path, user_agent=user_agent or getattr(settings, "FETCHER_USER_AGENT", ""), rules=_build_ruleset(stale)) + return _evaluate_path( + path, + user_agent=user_agent or getattr(settings, "FETCHER_USER_AGENT", ""), + rules=_build_ruleset(stale), + ) return RobotsDecision(True, f"Robotscontrole tijdelijk niet beschikbaar: {exc}") if cache.error: return RobotsDecision(True, cache.error) rules = _build_ruleset(cache) - decision = _evaluate_path(path, user_agent=user_agent or getattr(settings, "FETCHER_USER_AGENT", ""), rules=rules) + decision = _evaluate_path( + path, user_agent=user_agent or getattr(settings, "FETCHER_USER_AGENT", ""), rules=rules + ) return decision diff --git a/apps/sources/services/scheduling.py b/apps/sources/services/scheduling.py index d8a2e63..a8a537c 100644 --- a/apps/sources/services/scheduling.py +++ b/apps/sources/services/scheduling.py @@ -76,14 +76,10 @@ def calculate_failure_backoff_seconds( def calculate_success_jitter_seconds(source: Source) -> int: - return calculate_jitter_seconds( - source, max_seconds=settings.SOURCE_SUCCESS_JITTER_SECONDS - ) + return calculate_jitter_seconds(source, max_seconds=settings.SOURCE_SUCCESS_JITTER_SECONDS) -def acquire_source_lease( - *, source_id: int, worker_token: str, now=None -) -> SourceLease | None: +def acquire_source_lease(*, source_id: int, worker_token: str, now=None) -> SourceLease | None: now = now or timezone.now() with transaction.atomic(): source = Source.objects.select_for_update().get(pk=source_id) @@ -101,9 +97,8 @@ def acquire_source_lease( return None lease = SourceLease.objects.select_for_update().filter(source=source).first() - if lease is not None and not lease.is_expired: - if lease.token != worker_token: - return None + if lease is not None and not lease.is_expired and lease.token != worker_token: + return None active_leases = SourceLease.objects.select_for_update().filter( source__domain=domain, expires_at__gt=now @@ -118,7 +113,10 @@ def acquire_source_lease( lease.token = worker_token lease.worker_id = worker_token lease.expires_at = now + timedelta(seconds=_lease_ttl_seconds()) - lease.save(update_fields=["token", "worker_id", "expires_at", "updated_at"]) + if lease.pk is None: + lease.save() + else: + lease.save(update_fields=["token", "worker_id", "expires_at", "updated_at"]) return lease @@ -129,9 +127,11 @@ def release_source_lease(*, source_id: int, worker_token: str, now=None) -> bool if not source.domain: return False - lease = SourceLease.objects.select_for_update().filter( - source=source, token=worker_token - ).first() + lease = ( + SourceLease.objects.select_for_update() + .filter(source=source, token=worker_token) + .first() + ) if not lease: return False diff --git a/apps/sources/tasks.py b/apps/sources/tasks.py index 7a068bd..aaa1b13 100644 --- a/apps/sources/tasks.py +++ b/apps/sources/tasks.py @@ -20,7 +20,11 @@ from apps.sources.services.fetcher import ( RateLimitedError, fetch_url, ) -from apps.sources.services.health import canary_recovery_sources, evaluate_source_health, start_health_canary +from apps.sources.services.health import ( + canary_recovery_sources, + evaluate_source_health, + start_health_canary, +) from apps.sources.services.policy import assess_url from apps.sources.services.scheduling import ( acquire_source_lease, @@ -82,9 +86,7 @@ def fetch_source(self, source_id: int, force: bool = False) -> dict[str, object] try: fetched = fetch_url(source.base_url, source=source, conditional_headers=headers) if fetched.status_code == 304: - source.schedule_after_success( - jitter_seconds=calculate_success_jitter_seconds(source) - ) + source.schedule_after_success(jitter_seconds=calculate_success_jitter_seconds(source)) run.finish(SourceRun.Status.SUCCESS, http_status=304) return {"status": "not_modified"} content_type = fetched.headers.get("content-type", "") @@ -110,9 +112,7 @@ def fetch_source(self, source_id: int, force: bool = False) -> dict[str, object] source.etag = fetched.headers.get("etag", source.etag) source.last_modified = fetched.headers.get("last-modified", source.last_modified) source.save(update_fields=["etag", "last_modified", "updated_at"]) - source.schedule_after_success( - jitter_seconds=calculate_success_jitter_seconds(source) - ) + source.schedule_after_success(jitter_seconds=calculate_success_jitter_seconds(source)) run.finish( SourceRun.Status.SUCCESS, http_status=fetched.status_code, @@ -159,9 +159,7 @@ def fetch_source(self, source_id: int, force: bool = False) -> dict[str, object] ) return {"status": "timeout", "backoff": backoff} except FetchError as exc: - backoff = calculate_failure_backoff_seconds( - source, failure_count=source.failure_count + 1 - ) + backoff = calculate_failure_backoff_seconds(source, failure_count=source.failure_count + 1) source.schedule_after_failure(now=timezone.now(), backoff_seconds=backoff) run.finish( SourceRun.Status.FAILED, diff --git a/apps/sources/views.py b/apps/sources/views.py index 9b5e9ee..a34e011 100644 --- a/apps/sources/views.py +++ b/apps/sources/views.py @@ -11,22 +11,20 @@ from django.views.decorators.cache import never_cache from django.views.decorators.http import require_POST from django.views.generic import ListView +from apps.core.rate_limit import clear_rate_limit, is_rate_limited, register_rate_limit_failure + from .forms import ManualImportForm -from .models import Source -from .models import SourcePolicyReview +from .models import Source, SourcePolicyReview from .services.manual_import import ManualImportError, ManualImportSummary, import_manual_source from .services.policy import create_policy_review from .tasks import fetch_source -from apps.core.rate_limit import clear_rate_limit, is_rate_limited, register_rate_limit_failure def _source_list_queryset(): return Source.objects.prefetch_related("policy_reviews").all().order_by("name") -def _manual_import_context( - request: HttpRequest, *, form: ManualImportForm, manual_result=None -): +def _manual_import_context(request: HttpRequest, *, form: ManualImportForm, manual_result=None): base_queryset = _source_list_queryset() return { "sources": base_queryset, @@ -134,7 +132,7 @@ def bulk_candidate_action(request): source.policy_reason = reason source.save(update_fields=["status", "policy", "policy_reason", "updated_at"]) - messages.success(request, f"{sources.count()} bron(nen) naar { _bulk_summary(action) }.") + messages.success(request, f"{sources.count()} bron(nen) naar {_bulk_summary(action)}.") return redirect("sources:list") diff --git a/config/settings.py b/config/settings.py index d2141f8..528d424 100644 --- a/config/settings.py +++ b/config/settings.py @@ -3,6 +3,7 @@ from __future__ import annotations import os from pathlib import Path from urllib.parse import urlparse + from django.core.exceptions import ImproperlyConfigured BASE_DIR = Path(__file__).resolve().parent.parent @@ -33,23 +34,15 @@ def _validate_production_security( "Onveilige DJANGO_SECRET_KEY in productie. Stel een sterke waarde in via environment." ) if len(secret_key.strip()) < 50: - raise ImproperlyConfigured( - "DJANGO_SECRET_KEY is te kort voor een productieomgeving." - ) + raise ImproperlyConfigured("DJANGO_SECRET_KEY is te kort voor een productieomgeving.") if not allowed_hosts: raise ImproperlyConfigured("ALLOWED_HOSTS mag in productie niet leeg zijn.") if not csrf_trusted_origins: - raise ImproperlyConfigured( - "CSRF_TRUSTED_ORIGINS is verplicht bij DEBUG=False." - ) + raise ImproperlyConfigured("CSRF_TRUSTED_ORIGINS is verplicht bij DEBUG=False.") if not any(origin.lower().startswith("https://") for origin in csrf_trusted_origins): - raise ImproperlyConfigured( - "CSRF_TRUSTED_ORIGINS moet HTTPS-origins bevatten in productie." - ) + raise ImproperlyConfigured("CSRF_TRUSTED_ORIGINS moet HTTPS-origins bevatten in productie.") if not session_cookie_secure or not csrf_cookie_secure: - raise ImproperlyConfigured( - "Session- en CSRF-cookies moeten Secure=True zijn in productie." - ) + raise ImproperlyConfigured("Session- en CSRF-cookies moeten Secure=True zijn in productie.") if not secure_ssl_redirect: raise ImproperlyConfigured("SECURE_SSL_REDIRECT moet True zijn in productie.") @@ -148,7 +141,7 @@ if DATABASE_URL: "OPTIONS": {"connect_timeout": 10}, } } -elif (postgres_cfg := _postgres_database_from_env()): +elif postgres_cfg := _postgres_database_from_env(): DATABASES = {"default": postgres_cfg} else: DATABASES = { @@ -299,15 +292,9 @@ OLLAMA_TIMEOUT_SECONDS = float(os.getenv("OLLAMA_TIMEOUT_SECONDS", "60")) DIGEST_RECIPIENT = os.getenv("DIGEST_RECIPIENT", "") -AUTH_LOGIN_RATE_LIMIT_MAX_ATTEMPTS = int( - os.getenv("AUTH_LOGIN_RATE_LIMIT_MAX_ATTEMPTS", "8") -) -AUTH_LOGIN_RATE_LIMIT_WINDOW_SECONDS = int( - os.getenv("AUTH_LOGIN_RATE_LIMIT_WINDOW_SECONDS", "300") -) -AUTH_LOGIN_RATE_LIMIT_BLOCK_SECONDS = int( - os.getenv("AUTH_LOGIN_RATE_LIMIT_BLOCK_SECONDS", "300") -) +AUTH_LOGIN_RATE_LIMIT_MAX_ATTEMPTS = int(os.getenv("AUTH_LOGIN_RATE_LIMIT_MAX_ATTEMPTS", "8")) +AUTH_LOGIN_RATE_LIMIT_WINDOW_SECONDS = int(os.getenv("AUTH_LOGIN_RATE_LIMIT_WINDOW_SECONDS", "300")) +AUTH_LOGIN_RATE_LIMIT_BLOCK_SECONDS = int(os.getenv("AUTH_LOGIN_RATE_LIMIT_BLOCK_SECONDS", "300")) MANUAL_IMPORT_RATE_LIMIT_MAX_ATTEMPTS = int( os.getenv("MANUAL_IMPORT_RATE_LIMIT_MAX_ATTEMPTS", "12") ) diff --git a/config/urls.py b/config/urls.py index d18b0eb..5999752 100644 --- a/config/urls.py +++ b/config/urls.py @@ -2,8 +2,8 @@ from __future__ import annotations from django.conf import settings from django.conf.urls.static import static -from django.contrib.auth import views as auth_views from django.contrib import admin +from django.contrib.auth import views as auth_views from django.urls import include, path from apps.core.views import SecurityAwareLoginView diff --git a/docs/ai/PROJECT_STATE.md b/docs/ai/PROJECT_STATE.md index b6754a1..f656a5e 100644 --- a/docs/ai/PROJECT_STATE.md +++ b/docs/ai/PROJECT_STATE.md @@ -21,13 +21,16 @@ De repository bevat een uitvoerbare Django-MVP met: ## Laatste geverifieerde baseline -De definitieve projectbasis en een schoon uit de ZIP opgebouwde checkout zijn beide geverifieerd met: +Release-audit op 2026-07-21: -- 43 geslaagde tests; -- 80,42% branch-aware codedekking over `apps` en `config`; +- 144 geslaagde tests en 2 lokaal overgeslagen Playwright-varianten; de HTML/a11y-fallback is geslaagd; +- 81,98% branch-aware codedekking over `apps` en `config`; - Ruff en Django system checks geslaagd; - geen ontbrekende migraties; -- clean-room bootstrap, demo-import en HTTP-smoke voor liveness, readiness en login geslaagd. +- backlog- en repositoryvalidatie geslaagd; +- de lockfile gebruikt publieke PyPI-bronnen in plaats van een niet-overdraagbare interne registry; +- releaseblokkers hersteld in template-rendering, JSON/BOM-verwerking, leases, feedback, dossierautorisatie, ATS-provenance, employer-resolutie en DNS-rebindcontrole; +- een echte Dockerimagebuild en live server-smoke blijven afhankelijk van de Gitea-runner/server, omdat Docker lokaal niet beschikbaar is. ## Laatste uitgevoerde backlogtaak diff --git a/docs/quality/THREAT_MODEL.md b/docs/quality/THREAT_MODEL.md index 51cef96..ed997d8 100644 --- a/docs/quality/THREAT_MODEL.md +++ b/docs/quality/THREAT_MODEL.md @@ -71,6 +71,7 @@ Django/Celery | T-23 | Onbevoegde admin op thuisnetwerk | volledige datatoegang | uniek wachtwoord, TLS/VPN, geen defaultcredentials, sessiebeveiliging | Unraidchecklist | | T-24 | Source terms wijzigen | ongewenste voortgezette crawling | reviewdatum, source health en policy expiry gepland | `VR-102`, `VR-112` | | T-31 | Sollicitatiedossier-export bevat niet-gewenste payload | data-lek of onbedoeld dossieroverdragen | export bevat alleen het gekozen dossier, ZIP-inhoud is beperkt tot snapshot/tijdlijn/print-HTML, bestandsnaam is geslugified zonder padseparators | `tests/integration/test_applications.py` | +| T-32 | XML external entity/entity-expansion via sitemap | lokale data-uitlezing of parseruitputting | sitemap-XML wordt met `defusedxml` verwerkt; fetcherlimieten blijven vóór parsing gelden | `tests/unit/test_discovery_rss.py`, Ruff securitycheck | ## Misbruikscenario's diff --git a/pyproject.toml b/pyproject.toml index 33ecd5d..a587467 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -17,6 +17,7 @@ dependencies = [ "feedparser==6.0.12", "PyYAML==6.0.3", "python-dateutil==2.9.0.post0", + "defusedxml==0.7.1", "gunicorn==26.0.0", "whitenoise==6.11.0", ] diff --git a/scripts/benchmark.py b/scripts/benchmark.py index 835c0b3..cabad70 100644 --- a/scripts/benchmark.py +++ b/scripts/benchmark.py @@ -4,6 +4,7 @@ from __future__ import annotations import argparse import hashlib import json +import math import os import platform import tracemalloc @@ -58,7 +59,7 @@ def _percentile(values: list[float], p: float) -> float: if not values: return 0.0 ordered = sorted(values) - index = round((len(ordered) - 1) * p / 100) + index = max(0, math.ceil(len(ordered) * p / 100) - 1) return ordered[index] @@ -98,7 +99,7 @@ class BenchmarkArtifact: def _load_text(path: Path) -> str: _is_supported_file_path(path) - return path.read_text(encoding="utf-8") + return path.read_text(encoding="utf-8-sig") def load_benchmark_dataset(path: Path) -> dict[str, Any]: @@ -297,8 +298,6 @@ def _create_seed_jobs( ) job = JobPosting.objects.create( employer=employer, - requirements=payload["job"]["requirements"], - benefits=payload["job"]["benefits"], description_html_sanitized=payload["job"]["description_text"], extraction_confidence=Decimal("0.95"), analysis_features={}, diff --git a/scripts/feedback_learning_report.py b/scripts/feedback_learning_report.py index 50896c4..fd2953a 100644 --- a/scripts/feedback_learning_report.py +++ b/scripts/feedback_learning_report.py @@ -11,7 +11,7 @@ import django os.environ.setdefault("DJANGO_SETTINGS_MODULE", "config.settings") django.setup() -from apps.profiles.models import SearchProfile +from apps.profiles.models import SearchProfile # noqa: E402 def _load_profile(profile_id: int | None, username: str | None) -> SearchProfile: @@ -35,7 +35,9 @@ def _collect_feedback_churn(profile: SearchProfile) -> dict: false_negatives = [] for feedback in all_feedback: - learning = feedback.metadata.get("learning") if isinstance(feedback.metadata, dict) else None + learning = ( + feedback.metadata.get("learning") if isinstance(feedback.metadata, dict) else None + ) if not isinstance(learning, dict): continue reason_code = learning.get("reason_code", "") @@ -112,7 +114,10 @@ def main(argv: list[str] | None = None) -> int: ] + [f" - {item['feature']}: {item['delta']:+.3f}" for item in report["weight_churn"]] + [f"\nMogelijke non-learn false negatives (max {args.top}):"] - + [f" - {item['job_id']} ({item['reason_code']})" for item in feedback_report["queued_non_learning_signals"][: args.top]] + + [ + f" - {item['job_id']} ({item['reason_code']})" + for item in feedback_report["queued_non_learning_signals"][: args.top] + ] ) + "\n", encoding="utf-8", diff --git a/scripts/release_smoke.py b/scripts/release_smoke.py index 1af7854..f248425 100644 --- a/scripts/release_smoke.py +++ b/scripts/release_smoke.py @@ -109,24 +109,27 @@ def _run_fixture_import(*, fixture: str, url: str, source_name: str, source_type def _assert_idempotent_import(url: str) -> tuple[JobPosting, int]: - aliases = list( - JobSourceAlias.objects.filter(url=url).select_related("job").order_by("pk") - ) + aliases = list(JobSourceAlias.objects.filter(url=url).select_related("job").order_by("pk")) if not aliases: raise RuntimeError("Release-smoke: fixtureimport leverde geen alias op.") job_ids = {alias.job_id for alias in aliases if alias.job_id} if len(job_ids) != 1: raise RuntimeError( - f"Release-smoke: fixtureimport leidde niet tot één aliascluster (aantal jobids={len(job_ids)})." + "Release-smoke: fixtureimport leidde niet tot één aliascluster " + f"(aantal jobids={len(job_ids)})." ) return aliases[-1].job, len(aliases) -def _build_report(*, label: str, alias_count: int, job: JobPosting, profile: SearchProfile) -> dict[str, object]: +def _build_report( + *, label: str, alias_count: int, job: JobPosting, profile: SearchProfile +) -> dict[str, object]: return { "label": label, "alias_count": alias_count, - "source_job_count": JobPosting.objects.filter(source_aliases__url=job.canonical_url).count(), + "source_job_count": JobPosting.objects.filter( + source_aliases__url=job.canonical_url + ).count(), "job_total_count": JobPosting.objects.count(), "feedback_count": Feedback.objects.filter(user=profile.user, job=job).count(), "application_exists": JobPosting.objects.filter(id=job.id).exists(), diff --git a/templates/sources/list.html b/templates/sources/list.html index 6163d87..c70708d 100644 --- a/templates/sources/list.html +++ b/templates/sources/list.html @@ -236,7 +236,7 @@ {{ error }} {% endfor %} - + {% if manual_import_result %} @@ -339,12 +339,12 @@ {% if latest_run %}{{ latest_run.extracted_count }}{% else %}0{% endif %}
    + {% if source.policy == 'deny' or latest_run and latest_run.status == 'failed' %}bg-error/10 border border-error/20{% elif source.status == 'quarantined' %}bg-error/10 border border-error/20{% elif latest_run and latest_run.status == 'partial' %}bg-orange-500/10 border border-orange-500/20{% else %}bg-secondary/10 border border-secondary/20{% endif %}"> + {% if source.policy == 'deny' or latest_run and latest_run.status == 'failed' %}text-error{% elif latest_run and latest_run.status == 'partial' %}text-orange-400{% else %}text-secondary{% endif %}"> {% if source.policy == "deny" %} Geblokkeerd {% elif latest_run and latest_run.status == 'failed' %} @@ -363,6 +363,16 @@
    + {% if source.discovery_evidence %} +
    +

    Ontdekkingsevidence

    + +
    + {% endif %}