Files
VacatureRadar/scripts/benchmark.py
T
Jens a4eced8be5
deploy / deploy (push) Canceled after 0s
Fix release blockers and deployment build
2026-07-21 21:22:29 +02:00

961 lines
36 KiB
Python

#!/usr/bin/env python3
from __future__ import annotations
import argparse
import hashlib
import json
import math
import os
import platform
import tracemalloc
from contextlib import contextmanager
from dataclasses import dataclass
from decimal import Decimal
from pathlib import Path
from statistics import mean
from time import perf_counter
from typing import Any
import django
from django.conf import settings
from django.contrib.auth import get_user_model
from django.db import connection
from django.test.utils import CaptureQueriesContext
from django.utils import timezone
os.environ.setdefault("DJANGO_SETTINGS_MODULE", "config.settings")
django.setup()
from apps.jobs.models import Employer, JobPosting, JobSourceAlias, ScoreRun # noqa: E402
from apps.jobs.services.dedupe import find_existing_job # noqa: E402
from apps.jobs.services.normalization import CanonicalJobDraft, normalize_token # noqa: E402
from apps.jobs.services.pipeline import process_raw_document # noqa: E402
from apps.jobs.services.scoring import calculate_score, rescore_jobs_with_profiles # noqa: E402
from apps.profiles.models import SearchProfile # noqa: E402
from apps.sources.adapters.registry import registry # noqa: E402
from apps.sources.models import RawDocument, Source, SourceRun # noqa: E402
DEFAULT_DATASET = Path("fixtures/benchmark/quality_benchmark.json")
DEFAULT_PARSER_FIELDS = [
"title",
"employer_name",
"location_text",
"description_text",
"date_posted",
"employment_types",
]
PARSER_EXPECTED_MIN_COVERAGE = 0.90
DEDUPE_EXPECTED_MIN_PRECISION = 0.90
DEDUPE_EXPECTED_MIN_RECALL = 0.90
TOP_N_EXPECTED_MAX_CHURN = 0.35
NFR_007_P95_MS_LIMIT = 1000.0
DEFAULT_TOP_N = 25
QUICK_JOBS = 250
FULL_JOBS = 25_000
PERF_RANKING_JOBS = 120
def _percentile(values: list[float], p: float) -> float:
if not values:
return 0.0
ordered = sorted(values)
index = max(0, math.ceil(len(ordered) * p / 100) - 1)
return ordered[index]
def _to_bool(value: object) -> bool:
return bool(value)
def _value_present(value: object) -> bool:
if value is None:
return False
if isinstance(value, str):
return bool(value.strip())
if isinstance(value, list | tuple | dict | set):
return len(value) > 0
if isinstance(value, bool | int | float | Decimal):
return True
return bool(value)
def _is_supported_file_path(path: Path) -> bool:
if path.exists() and path.is_file():
return True
raise FileNotFoundError(f"Fixturebestand niet gevonden: {path}")
@dataclass
class BenchmarkArtifact:
profile_id: int | None = None
user_id: int | None = None
source_id: int | None = None
source_run_id: int | None = None
job_ids: list[int] | None = None
raw_ids: list[int] | None = None
alias_ids: list[int] | None = None
scorerun_ids: list[int] | None = None
def _load_text(path: Path) -> str:
_is_supported_file_path(path)
return path.read_text(encoding="utf-8-sig")
def load_benchmark_dataset(path: Path) -> dict[str, Any]:
payload = json.loads(_load_text(path))
if not isinstance(payload, dict):
raise ValueError("Benchmark fixturebestand moet een JSON-object bevatten.")
return payload
def _parser_case_to_report(case: dict[str, Any], result) -> dict[str, Any]:
jobs = result.jobs
expected_count = int(case.get("expected_job_count", 1))
expected_fields = list(case.get("expected_fields", DEFAULT_PARSER_FIELDS))
sample_fields = []
for field in expected_fields:
sample_fields.append(any(_value_present(getattr(job, field, None)) for job in jobs))
fields_ok = sum(1 for item in sample_fields if item)
expected_parser_key = case.get("expected_parser_key")
expected_parser_version = case.get("expected_parser_version")
parser_key_match = (
result.parser_key == expected_parser_key if expected_parser_key is not None else True
)
parser_version_match = (
result.parser_version == expected_parser_version
if expected_parser_version is not None
else True
)
passed = (
parser_key_match
and parser_version_match
and len(jobs) >= expected_count
and all(sample_fields)
)
return {
"id": case.get("id"),
"fixture": case.get("fixture"),
"parser_key": result.parser_key,
"parser_version": result.parser_version,
"expected_parser_key": expected_parser_key,
"expected_parser_version": expected_parser_version,
"extracted": len(jobs),
"expected_job_count": expected_count,
"covered_fields": fields_ok,
"expected_fields": len(expected_fields),
"coverage_ratio": round(fields_ok / len(expected_fields), 3) if expected_fields else 1.0,
"warnings": result.warnings,
"passed": passed,
}
def run_parser_benchmark(case_definitions: list[dict[str, Any]]) -> dict[str, Any]:
cases: list[dict[str, Any]] = []
total_fields = 0
covered_fields = 0
for case in case_definitions:
fixture_path = Path(case["fixture"])
content = _load_text(fixture_path)
document = RawDocument(
source=case.get("_source", None),
source_run=case.get("_source_run", None),
url=case["url"],
final_url=case["url"],
kind=case.get("kind", RawDocument.Kind.HTML),
content_type=case.get("content_type", "text/html"),
content_hash=hashlib.sha256(content.encode("utf-8")).hexdigest(),
body_text=content,
byte_length=len(content.encode("utf-8")),
retain_until=timezone.now(),
)
result = registry.extract(document)
report = _parser_case_to_report(case, result)
total_fields += report["expected_fields"]
covered_fields += report["covered_fields"]
cases.append(report)
coverage_ratio = round(covered_fields / total_fields, 3) if total_fields else 0.0
return {
"status": "passed" if coverage_ratio >= PARSER_EXPECTED_MIN_COVERAGE else "failed",
"coverage_ratio": coverage_ratio,
"unknown_data_ratio": round(1.0 - coverage_ratio, 3),
"covered_fields": covered_fields,
"expected_fields": total_fields,
"cases": cases,
}
def _build_draft(case: dict[str, Any]) -> CanonicalJobDraft:
title = case.get("title", "Vacature")
normalized_title = case.get("normalized_title") or normalize_token(title)
employer_name = case.get("employer_name", "Onbekende werkgever")
location_text = case.get("location_text", "")
description = case.get("description_text", "")
payload = (title + normalized_title + employer_name + location_text + description).encode(
"utf-8"
)
return CanonicalJobDraft(
source_url=case.get("source_url", "https://benchmark.vr.local/job"),
canonical_url=case.get("canonical_url", "https://benchmark.vr.local/job"),
external_id=case.get("external_id", ""),
title=title,
normalized_title=normalized_title,
job_family=case.get("job_family", ""),
employer_name=employer_name,
employer_domain=case.get("employer_domain", "vr.example.org"),
location_text=location_text,
region=case.get("region", ""),
municipality=case.get("municipality", location_text.split(",")[0] if location_text else ""),
postal_code=case.get("postal_code", ""),
country=case.get("country", "BE"),
workplace_type=case.get("workplace_type", ""),
employment_types=case.get("employment_types", []),
language=case.get("language", "nl"),
description_html=description,
description_text=description,
date_posted=None,
valid_through=None,
compensation={},
skills_required=case.get("skills_required", []),
skills_preferred=case.get("skills_preferred", []),
content_hash=hashlib.sha256(payload).hexdigest(),
canonical_key=case.get(
"canonical_key",
hashlib.sha256(
(case.get("canonical_url", "benchmark") + title).encode("utf-8")
).hexdigest(),
),
)
def _seed_job_payload(seed: dict[str, Any]) -> tuple[dict[str, Any], dict[str, str]]:
title = seed["title"]
normalized_title = seed.get("normalized_title") or normalize_token(title)
employer_name = seed["employer_name"]
employer_domain = seed.get("employer_domain", "vr.example.org")
normalized_name = seed.get("employer_normalized_name") or normalize_token(employer_name)
location_text = seed.get("location_text", "")
payload = {
"employer": {
"name": employer_name,
"normalized_name": normalized_name,
"domain": employer_domain,
},
"job": {
"original_title": title,
"normalized_title": normalized_title,
"job_family": seed.get("job_family", "it"),
"canonical_url": seed["canonical_url"],
"canonical_key": seed.get(
"canonical_key",
hashlib.sha256(seed["canonical_url"].encode("utf-8")).hexdigest(),
),
"content_hash": hashlib.sha256(
(seed["canonical_url"] + title).encode("utf-8")
).hexdigest(),
"description_text": seed.get("description_text", ""),
"raw_location": location_text,
"region": seed.get("region", ""),
"municipality": seed.get(
"municipality", location_text.split(",")[0] if location_text else ""
),
"postal_code": seed.get("postal_code", ""),
"country": seed.get("country", "BE"),
"workplace_type": seed.get("workplace_type", JobPosting.Workplace.UNKNOWN),
"employment_types": seed.get("employment_types", []),
"requirements": seed.get("requirements", []),
"benefits": seed.get("benefits", []),
"status": JobPosting.Status.ACTIVE,
},
"alias": {
"external_id": seed.get("external_id", ""),
"url": seed["canonical_url"],
},
}
return payload, {"employer_name": employer_name, "employer_domain": employer_domain}
def _create_seed_jobs(
seeds: list[dict[str, Any]],
source: Source,
artifacts: BenchmarkArtifact,
) -> tuple[list[JobPosting], dict[str, JobPosting]]:
id_index = {}
seed_jobs: list[JobPosting] = []
for seed in seeds:
payload, employer_info = _seed_job_payload(seed)
employer, _ = Employer.objects.get_or_create(
normalized_name=employer_info["employer_name"],
defaults={
"name": employer_info["employer_name"],
"domain": employer_info["employer_domain"],
"is_direct_employer": True,
"is_recruiter": False,
"confidence": Decimal("0.95"),
},
)
job = JobPosting.objects.create(
employer=employer,
description_html_sanitized=payload["job"]["description_text"],
extraction_confidence=Decimal("0.95"),
analysis_features={},
direct_employer=True,
recruiter=False,
**payload["job"],
)
alias = JobSourceAlias.objects.create(
job=job,
source=source,
url=seed["canonical_url"],
canonical_url=seed["canonical_url"],
external_id=payload["alias"]["external_id"],
source_title=seed["title"],
source_employer=seed["employer_name"],
is_canonical=True,
extraction_method="benchmark-seed",
extraction_confidence=Decimal("0.95"),
)
seed_jobs.append(job)
if artifacts.job_ids is not None:
artifacts.job_ids.append(job.id)
if artifacts.alias_ids is not None:
artifacts.alias_ids.append(alias.id)
id_index[seed["id"]] = job
return seed_jobs, id_index
def run_dedupe_benchmark(
source: Source,
seed_cases: list[dict[str, Any]],
query_cases: list[dict[str, Any]],
*,
artifacts: BenchmarkArtifact,
) -> dict[str, Any]:
seed_jobs, seed_map = _create_seed_jobs(seed_cases, source, artifacts)
true_positive = 0
false_positive = 0
false_negative = 0
true_negative = 0
case_reports = []
for case in query_cases:
expected_match = case.get("expected_match")
threshold = float(case.get("threshold", 0.92))
draft = _build_draft(case["query"])
decision = find_existing_job(draft, threshold=threshold, source=source)
expected_job = seed_map.get(expected_match) if expected_match else None
matched = decision.job is not None
expected_match_hit = expected_job is not None
expected_match_correct = expected_job is not None and decision.job == expected_job
if expected_match_hit and matched:
if expected_match_correct:
true_positive += 1
else:
false_positive += 1
false_negative += 1
elif expected_match_hit and not matched:
false_negative += 1
elif expected_match_hit is False and matched:
false_positive += 1
else:
true_negative += 1
case_reports.append(
{
"id": case.get("id"),
"expected_match": expected_match,
"expected_match_present": bool(expected_match_hit),
"decision": decision.reason,
"decision_similarity": decision.similarity,
"matched": matched,
"matched_job_id": str(decision.job.pk) if decision.job else None,
"is_expected_match": bool(expected_match_hit and expected_match_correct),
"query": case.get("id"),
}
)
precision = (
round(true_positive / (true_positive + false_positive), 3)
if (true_positive + false_positive)
else 1.0
)
recall = (
round(true_positive / (true_positive + false_negative), 3)
if (true_positive + false_negative)
else 1.0
)
return {
"status": "passed"
if precision >= DEDUPE_EXPECTED_MIN_PRECISION and recall >= DEDUPE_EXPECTED_MIN_RECALL
else "failed",
"precision": precision,
"recall": recall,
"false_merges": false_positive,
"missed_merges": false_negative,
"true_positive": true_positive,
"true_negative": true_negative,
"false_positive": false_positive,
"false_negative": false_negative,
"case_reports": case_reports,
"seed_jobs": len(seed_jobs),
}
def _build_ranking_jobs(
profile: SearchProfile,
source: Source,
count: int,
base_url: str,
artifacts: BenchmarkArtifact,
) -> list[JobPosting]:
titles = [
"Infrastructure Engineer",
"Network Engineer",
"Security Specialist",
"Cloud Architect",
"Sysadmin",
]
employers = ["Example IT", "Example Cloud", "Example Secure"]
locations = ["Hasselt, Limburg", "Aalst, Oost-Vlaanderen", "Genk, Limburg", "Brussel"]
jobs = []
for index in range(count):
title = titles[index % len(titles)]
employer_name = employers[index % len(employers)]
normalized_title = normalize_token(title)
canonical = f"{base_url}/bench/job/{index:06d}"
description = (
f"{title} op projectniveau met ervaring in Microsoft 365." if index % 4 else ""
)
employer, _ = Employer.objects.get_or_create(
normalized_name=normalize_token(employer_name),
defaults={
"name": employer_name,
"domain": f"{normalize_token(employer_name).replace(' ', '-')}.example.org",
"is_direct_employer": True,
"is_recruiter": False,
"confidence": Decimal("0.95"),
},
)
jobs.append(
JobPosting(
employer=employer,
original_title=title,
normalized_title=normalized_title,
job_family=normalized_title.split(" ")[0] or "it",
canonical_url=canonical,
canonical_key=hashlib.sha256(canonical.encode("utf-8")).hexdigest(),
content_hash=hashlib.sha256((canonical + description).encode("utf-8")).hexdigest(),
description_html_sanitized=description,
description_text=description,
raw_location=locations[index % len(locations)],
region="Limburg" if "Limburg" in locations[index % len(locations)] else "Brabant",
municipality=(locations[index % len(locations)].split(",")[0]).strip(),
postal_code="3500" if index % 2 == 0 else "8500",
country="BE",
workplace_type=JobPosting.Workplace.HYBRID
if index % 3 == 0
else JobPosting.Workplace.REMOTE,
employment_types=["full_time"] if index % 2 == 0 else ["permanent"],
extraction_confidence=Decimal("0.95"),
analysis_features={
"support_ratio": 0.01 * (index % 6),
"public_sector_signal": 0.0,
"experience_years_max": 3 + (index % 4),
},
)
)
created_jobs = JobPosting.objects.bulk_create(jobs)
for job in created_jobs:
alias = JobSourceAlias.objects.create(
job=job,
source=source,
url=job.canonical_url,
canonical_url=job.canonical_url,
external_id="",
source_title=job.original_title,
source_employer=job.employer_name,
extraction_method="benchmark",
extraction_confidence=Decimal("0.95"),
is_canonical=True,
)
if artifacts.job_ids is not None:
artifacts.job_ids.append(job.id)
if artifacts.alias_ids is not None:
artifacts.alias_ids.append(alias.id)
return list(created_jobs)
def _score_top_n_jobs(profile: SearchProfile, jobs: list[JobPosting], top_n: int) -> list[str]:
scores: list[tuple[float, int, str]] = []
for job in jobs:
result = calculate_score(job, profile)
scores.append((float(result.score), job.pk, str(job.pk)))
return [
job_id for _, _, job_id in sorted(scores, key=lambda item: item[0], reverse=True)[:top_n]
]
def run_ranking_benchmark(
profile: SearchProfile,
source: Source,
top_n: int,
artifacts: BenchmarkArtifact,
) -> dict[str, Any]:
ranking_jobs = _build_ranking_jobs(
profile, source, PERF_RANKING_JOBS, "https://bench.example.org/rank", artifacts
)
base_weights = dict(profile.weights)
pre_weights = _score_top_n_jobs(profile, ranking_jobs, top_n)
profile.weights = {**base_weights, "content": 30.0, "skills": 35.0}
profile.save(update_fields=["weights", "updated_at"])
post_weights = _score_top_n_jobs(profile, ranking_jobs, top_n)
profile.weights = base_weights
profile.save(update_fields=["weights", "updated_at"])
pre_set = set(pre_weights)
post_set = set(post_weights)
churn = round(1 - (len(pre_set.intersection(post_set)) / top_n), 3) if top_n else 0.0
unknown_fields = {
"description_text": 0,
"raw_location": 0,
"employment_types": 0,
"analysis_features": 0,
}
unknown_total = 0
for job in ranking_jobs:
is_unknown = 0
if not job.description_text:
unknown_fields["description_text"] += 1
is_unknown += 1
if not job.raw_location:
unknown_fields["raw_location"] += 1
is_unknown += 1
if not job.employment_types:
unknown_fields["employment_types"] += 1
is_unknown += 1
if not job.analysis_features:
unknown_fields["analysis_features"] += 1
is_unknown += 1
unknown_total += int(is_unknown > 0)
unknown_ratio = round(unknown_total / len(ranking_jobs), 3) if ranking_jobs else 0.0
return {
"status": "passed" if churn <= TOP_N_EXPECTED_MAX_CHURN else "failed",
"top_n": top_n,
"pre_top_n": pre_weights,
"post_top_n": post_weights,
"churn_ratio": churn,
"unknown_data_ratio": unknown_ratio,
"unknown_fields": unknown_fields,
"seed_jobs_count": len(ranking_jobs),
}
def _clean_benchmark(artifacts: BenchmarkArtifact) -> None:
if artifacts.scorerun_ids:
ScoreRun.objects.filter(pk__in=artifacts.scorerun_ids).delete()
if artifacts.alias_ids:
JobSourceAlias.objects.filter(pk__in=artifacts.alias_ids).delete()
if artifacts.raw_ids:
RawDocument.objects.filter(pk__in=artifacts.raw_ids).delete()
if artifacts.job_ids:
JobPosting.objects.filter(pk__in=artifacts.job_ids).delete()
if artifacts.source_run_id:
SourceRun.objects.filter(pk=artifacts.source_run_id).delete()
if artifacts.source_id:
Source.objects.filter(pk=artifacts.source_id).delete()
if artifacts.profile_id:
SearchProfile.objects.filter(pk=artifacts.profile_id).delete()
if artifacts.user_id:
get_user_model().objects.filter(pk=artifacts.user_id).delete()
@contextmanager
def _query_capture() -> Any:
original_force_debug_cursor = connection.force_debug_cursor
connection.force_debug_cursor = True
try:
with CaptureQueriesContext(connection) as captured:
yield captured
finally:
connection.force_debug_cursor = original_force_debug_cursor
def run_performance_benchmark(
profile: SearchProfile,
source: Source,
*,
jobs_to_process: int,
top_n: int,
run_import: bool = True,
) -> dict[str, Any]:
label = f"vr116-{timezone.now().strftime('%Y%m%d%H%M%S')}"
base_url = "https://bench.example.org"
source_run = SourceRun.objects.create(source=source)
template = _load_text(Path("fixtures/pages/sample_generic_job.html"))
candidates: list[JobPosting] = []
ranking_sample: list[int] = []
import_query_count = 0
rescore_query_count = 0
list_query_count = 0
detail_query_count = 0
current_bytes = 0
peak_bytes = 0
current_ms_total = 0.0
timing_stats: dict[str, float] = {
"p50_ms": 0.0,
"p95_ms": 0.0,
"import_p50_ms": 0.0,
"import_p95_ms": 0.0,
"rescore_p50_ms": 0.0,
"rescore_p95_ms": 0.0,
"list_p95_ms": 0.0,
"detail_p95_ms": 0.0,
}
tracemalloc.start()
import_timings: list[float] = []
rescore_timings: list[float] = []
list_timings: list[float] = []
detail_timings: list[float] = []
query_count_total = 0
try:
with _query_capture() as captured_import:
if run_import:
for index in range(jobs_to_process):
unique_url = f"{base_url}/bench/{label}/{index}"
body = template.replace(
"https://careers.example.net/jobs/workplace-engineer", unique_url
)
body_hash = hashlib.sha256(body.encode("utf-8")).hexdigest()
document = RawDocument.objects.create(
source=source,
source_run=source_run,
url=unique_url,
final_url=unique_url,
kind=RawDocument.Kind.HTML,
content_type="text/html",
content_hash=body_hash,
body_text=body,
byte_length=len(body.encode("utf-8")),
retain_until=timezone.now() + timezone.timedelta(days=7),
)
start = perf_counter()
process_raw_document(document)
import_timings.append((perf_counter() - start) * 1000)
import_query_count = len(captured_import.captured_queries)
query_count_total += import_query_count
candidates = list(
JobPosting.objects.filter(
canonical_url__startswith=f"{base_url}/bench/{label}/",
status=JobPosting.Status.ACTIVE,
).select_related("employer")
)
if not candidates:
raise RuntimeError("Benchmarkimport produceerde geen vacatures.")
with _query_capture() as captured_rescore:
for start in range(0, len(candidates), 250):
batch = candidates[start : start + 250]
batch_start = perf_counter()
rescore_jobs_with_profiles(batch, profile_id=profile.id)
rescore_timings.append((perf_counter() - batch_start) * 1000)
rescore_query_count = len(captured_rescore.captured_queries)
query_count_total += rescore_query_count
with _query_capture() as captured_list:
ranking_sample = list(
JobPosting.objects.filter(source_aliases__source=source)
.order_by("-created_at")
.values_list("id", flat=True)[:top_n]
)
for job_id in ranking_sample:
start = perf_counter()
list(
JobPosting.objects.filter(id=job_id)
.select_related("employer")
.prefetch_related("source_aliases", "scores")
)
list_timings.append((perf_counter() - start) * 1000)
with _query_capture() as captured_detail:
for job_id in ranking_sample:
start = perf_counter()
list(
ScoreRun.objects.filter(
job_id=job_id,
profile=profile,
).order_by("-created_at")
)
detail_timings.append((perf_counter() - start) * 1000)
list_query_count = len(captured_list.captured_queries)
detail_query_count = len(captured_detail.captured_queries)
query_count_total += list_query_count + detail_query_count
current_bytes, peak_bytes = tracemalloc.get_traced_memory()
if query_count_total == 0:
query_count_total = (
len(import_timings) + len(rescore_timings) + len(list_timings) + len(detail_timings)
)
current_ms_total = (
sum(import_timings) + sum(rescore_timings) + sum(list_timings) + sum(detail_timings)
)
operation_times = import_timings + rescore_timings + list_timings + detail_timings
timing_stats = {
"p50_ms": round(_percentile(operation_times, 50), 3),
"p95_ms": round(_percentile(operation_times, 95), 3),
"import_p50_ms": round(_percentile(import_timings, 50), 3),
"import_p95_ms": round(_percentile(import_timings, 95), 3),
"rescore_p50_ms": round(_percentile(rescore_timings, 50), 3),
"rescore_p95_ms": round(_percentile(rescore_timings, 95), 3),
"list_p95_ms": round(_percentile(list_timings, 95), 3),
"detail_p95_ms": round(_percentile(detail_timings, 95), 3),
"import_total_s": round(sum(import_timings) / 1000, 3),
"rescore_total_s": round(sum(rescore_timings) / 1000, 3),
"list_total_s": round(sum(list_timings) / 1000, 3),
"detail_total_s": round(sum(detail_timings) / 1000, 3),
}
finally:
current_bytes, peak_bytes = tracemalloc.get_traced_memory()
tracemalloc.stop()
if query_count_total == 0:
query_count_total = (
len(import_timings) + len(rescore_timings) + len(list_timings) + len(detail_timings)
)
return {
"jobs": len(candidates),
"top_n": top_n,
"query_count": query_count_total,
"source_run_id": source_run.pk,
"import": {
"duration_ms_total": round(sum(import_timings), 3),
"p50_ms": timing_stats["import_p50_ms"],
"p95_ms": timing_stats["import_p95_ms"],
"query_count": import_query_count,
"timing_count": len(import_timings),
},
"rescore": {
"duration_ms_total": round(sum(rescore_timings), 3),
"p50_ms": timing_stats["rescore_p50_ms"],
"p95_ms": timing_stats["rescore_p95_ms"],
"query_count": rescore_query_count,
"timing_count": len(rescore_timings),
},
"dashboard": {
"list_query_count": list_query_count,
"detail_query_count": detail_query_count,
"list_p95_ms": timing_stats["list_p95_ms"],
"detail_p95_ms": timing_stats["detail_p95_ms"],
"timing_count": len(list_timings) + len(detail_timings),
},
"performance": {
"p50_ms": timing_stats["p50_ms"],
"p95_ms": timing_stats["p95_ms"],
"memory_peak_mb": round(peak_bytes / 1024 / 1024, 3),
"memory_current_mb": round(current_bytes / 1024 / 1024, 3),
"nfr_007": _to_bool(timing_stats["p95_ms"] <= NFR_007_P95_MS_LIMIT),
"total_duration_ms": round(current_ms_total, 3),
},
"avg_scores_import_ms": round(mean(import_timings), 3) if import_timings else 0.0,
"avg_scores_rescore_ms": round(mean(rescore_timings), 3) if rescore_timings else 0.0,
"avg_scores_list_ms": round(mean(list_timings), 3) if list_timings else 0.0,
"avg_scores_detail_ms": round(mean(detail_timings), 3) if detail_timings else 0.0,
}
def run_benchmark_report(
dataset_path: Path,
*,
quick: bool = False,
top_n: int = DEFAULT_TOP_N,
jobs: int | None = None,
skip_performance: bool = False,
) -> dict[str, Any]:
dataset = load_benchmark_dataset(dataset_path)
if jobs is None:
jobs = QUICK_JOBS if quick else FULL_JOBS
if top_n <= 0:
raise ValueError("top_n moet groter zijn dan 0.")
artifacts = BenchmarkArtifact(job_ids=[], raw_ids=[], alias_ids=[], scorerun_ids=[])
parser_cases = dataset["parser_coverage"]
parser_report = run_parser_benchmark(parser_cases)
dedupe_definition = dataset.get("dedupe", {})
seed_cases = dedupe_definition.get("seed_jobs", [])
query_cases = dedupe_definition.get("query_cases", [])
dedupe_report: dict[str, Any] = {"status": "skipped"}
ranking_report: dict[str, Any] = {"status": "skipped"}
performance_report: dict[str, Any] = {"status": "skipped"}
nfr_007 = {"measured": False, "passed": False, "p95_ms": None}
try:
User = get_user_model()
profile_user = User.objects.create_user(
username=f"benchmark_{timezone.now().strftime('%Y%m%d%H%M%S')}",
email="benchmark@example.invalid",
password="benchmark-password",
)
profile = SearchProfile.objects.create(
user=profile_user,
name="VR-116 benchmark profile",
is_active=True,
home_municipality="Hasselt",
home_latitude=Decimal("50.9325"),
home_longitude=Decimal("5.3396"),
max_distance_km=75,
desired_titles=["infrastructure engineer", "network engineer", "security specialist"],
desired_skills=["microsoft 365", "cloud", "vmware", "remote support"],
allowed_employment_types=["full_time", "permanent", "remote", "hybrid"],
preferred_workplace=["hybrid", "remote"],
hard_rules={"excluded_title_terms": []},
weights={
"content": 28.0,
"skills": 22.0,
"location": 20.0,
"conditions": 10.0,
"employer": 8.0,
"seniority": 6.0,
"preferences": 6.0,
},
recommendation_threshold=65,
top_match_threshold=90,
)
artifacts.profile_id = profile.pk
artifacts.user_id = profile_user.pk
source = Source.objects.create(
name=f"VR-116 benchmark source {timezone.now().strftime('%H%M%S')}",
source_type=Source.Type.EMPLOYER,
base_url="https://bench.example.org",
domain="bench.example.org",
status=Source.Status.ACTIVE,
policy=Source.Policy.ALLOW,
parser_key="auto",
)
artifacts.source_id = source.pk
if seed_cases and query_cases:
dedupe_report = run_dedupe_benchmark(
source=source,
seed_cases=seed_cases,
query_cases=query_cases,
artifacts=artifacts,
)
ranking_report = run_ranking_benchmark(
profile=profile,
source=source,
top_n=top_n,
artifacts=artifacts,
)
if not skip_performance:
performance_report = run_performance_benchmark(
profile,
source,
jobs_to_process=jobs,
top_n=top_n,
)
if jobs >= FULL_JOBS and not quick:
nfr_007 = {
"measured": True,
"passed": performance_report["performance"]["nfr_007"],
"p95_ms": performance_report["performance"]["p95_ms"],
}
finally:
_clean_benchmark(artifacts)
passed = (
parser_report["status"] == "passed"
and dedupe_report.get("status", "skipped") in {"passed", "skipped"}
and ranking_report.get("status", "skipped") in {"passed", "skipped"}
and (
performance_report.get("status", "skipped") == "skipped"
or performance_report["performance"]["p95_ms"] <= NFR_007_P95_MS_LIMIT
)
)
return {
"dataset_version": dataset.get("dataset_version", "unknown"),
"timestamp_utc": timezone.now().isoformat(),
"parser": parser_report,
"dedupe": dedupe_report,
"ranking": ranking_report,
"performance": performance_report,
"nfr_007": nfr_007,
"top_n": top_n,
"jobs_processed": jobs,
"hardware": {
"platform": platform.platform(),
"python": platform.python_version(),
"cpu_count": os.cpu_count(),
"release": platform.release(),
"debug_mode": settings.DEBUG,
},
"status": "passed" if passed else "failed",
}
def _parse_args(argv: list[str] | None = None) -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Kwaliteits- en performance-benchmarks voor VR-116"
)
parser.add_argument(
"--dataset",
default=str(DEFAULT_DATASET),
help="Pad naar fixturebestand (standaard: fixtures/benchmark/quality_benchmark.json).",
)
parser.add_argument("--quick", action="store_true", help="Kleinere, snelle offline run.")
parser.add_argument(
"--top-n", type=int, default=DEFAULT_TOP_N, help="Top-N grootte voor churnmeting."
)
parser.add_argument(
"--jobs",
type=int,
default=None,
help="Aantal vacatures voor performance-import; default 250 (quick) of 25000.",
)
parser.add_argument("--skip-performance", action="store_true", help="Sla performancerun over.")
parser.add_argument("--json", action="store_true", help="Toon machineleesbaar JSON.")
return parser.parse_args(argv)
def main(argv: list[str] | None = None) -> int:
args = _parse_args(argv)
report = run_benchmark_report(
Path(args.dataset),
quick=args.quick,
top_n=args.top_n,
jobs=args.jobs,
skip_performance=args.skip_performance,
)
if args.json:
print(json.dumps(report, indent=2, sort_keys=True))
else:
parser = report["parser"]
dedupe = report["dedupe"]
ranking = report["ranking"]
print(
f"Parser dekking: {parser['coverage_ratio']} (onzeker: {parser['unknown_data_ratio']})"
)
print(
f"Dedupe precision: {dedupe.get('precision', 0.0)} recall: {dedupe.get('recall', 0.0)}"
)
print(f"Top-{report['top_n']} churn: {ranking.get('churn_ratio', 0.0)}")
print(f"NFR-007: {report['nfr_007']}")
print(f"Overall status: {report['status']}")
return 0 if report["status"] == "passed" else 1
if __name__ == "__main__":
raise SystemExit(main())