feat: release regional radar and mailbox integrations
deploy / deploy (push) Canceled after 0s

This commit is contained in:
Jens
2026-07-22 05:12:07 +02:00
parent 598d3ec18a
commit 551d0f46c2
131 changed files with 6209 additions and 336 deletions
+6
View File
@@ -8,6 +8,9 @@ from .ats import (
from .base import ExtractedJob, ExtractionResult
from .generic_html import GenericHtmlAdapter
from .jsonld import JsonLdJobPostingAdapter
from .kempen import KempenEmployerAdapter
from .mol_region import MolRegionEmployerAdapter
from .regional import LocalEmployerListingAdapter
from .rss import RssAdapter
__all__ = [
@@ -16,7 +19,10 @@ __all__ = [
"GenericHtmlAdapter",
"GreenhouseAdapter",
"JsonLdJobPostingAdapter",
"KempenEmployerAdapter",
"LeverAdapter",
"LocalEmployerListingAdapter",
"MolRegionEmployerAdapter",
"RecruiteeAdapter",
"RssAdapter",
"SmartRecruitersAdapter",
+45 -6
View File
@@ -169,16 +169,20 @@ class _AtsAdapter(ABC):
def _extract_location(self, record: dict[str, object]) -> tuple[str, str, str, str]:
location_raw = _first_text(
record,
"location",
"location.city",
"location.name",
"location.address",
"location.raw",
"categories.location",
"city",
"cityName",
"place",
"office",
"location.address",
"address",
"data.location",
"officeLocation",
"locationName",
"location",
)
if not location_raw:
location_raw = _first_text(record, "data.city", "data.location")
@@ -223,6 +227,13 @@ class _AtsAdapter(ABC):
"remote_type",
"job_type",
).lower()
if not workplace_raw:
if record.get("hybrid") is True:
return "hybrid"
if record.get("remote") is True:
return "remote"
if record.get("on_site") is True:
return "on_site"
if "remote" in workplace_raw:
return "remote" if "hybrid" not in workplace_raw else "hybrid"
if "hybrid" in workplace_raw:
@@ -256,18 +267,21 @@ class _AtsAdapter(ABC):
"url",
"jobUrl",
"job_url",
"hostedUrl",
"applyUrl",
"link",
"absoluteUrl",
"careers_url",
"data.url",
)
employer_name = _first_text(
record,
"company_name",
"company.name",
"company",
"employer",
"organization",
"organizationName",
"company.name",
"department",
"hiringOrganization",
)
@@ -290,6 +304,7 @@ class _AtsAdapter(ABC):
"published_at",
"created_at",
"jobCreated",
"releasedDate",
)
valid_through = _first_text(
record,
@@ -308,6 +323,9 @@ class _AtsAdapter(ABC):
"jobType",
"type",
"data.employmentType",
"employment_type_code",
"categories.commitment",
"typeOfEmployment.label",
)
)
workplace_type = self._extract_workplace(record)
@@ -428,13 +446,20 @@ class GreenhouseAdapter(_AtsAdapter):
class LeverAdapter(_AtsAdapter):
parser_key = "ats-lever"
source_hosts = ("jobs.lever.co",)
source_hosts = (
"jobs.lever.co",
"jobs.eu.lever.co",
"api.lever.co",
"api.eu.lever.co",
)
support_markers = ("lever", "requisition", "posting")
listing_paths = (("data",), ("jobs",), ("results",))
detail_paths = (("data",), ("job",), ("position",), ("result",))
closed_statuses = CLOSED_STATUSES | {"archived", "deleted"}
def _extract_records(self, payload):
if isinstance(payload, list):
return [item for item in payload if isinstance(item, dict)]
for path in self.listing_paths:
value = payload
for key in path:
@@ -466,7 +491,7 @@ class RecruiteeAdapter(_AtsAdapter):
parser_key = "ats-recruitee"
source_hosts = ("recruitee.com",)
support_markers = ("recruitee", "career", "vacancy")
listing_paths = (("jobs",), ("data", "jobs"), ("vacancies",))
listing_paths = (("jobs",), ("offers",), ("data", "jobs"), ("vacancies",))
detail_paths = (("job",), ("data", "job"), ("vacancy",), ("result",))
closed_statuses = CLOSED_STATUSES | {"hidden", "paused"}
@@ -494,12 +519,26 @@ class RecruiteeAdapter(_AtsAdapter):
class SmartRecruitersAdapter(_AtsAdapter):
parser_key = "ats-smartrecruiters"
parser_version = "1.1.0"
source_hosts = ("smartrecruiters.com",)
support_markers = ("smartrecruiters", "smart recruiter")
listing_paths = (("jobs",), ("data", "jobs"), ("results",))
listing_paths = (("jobs",), ("data", "jobs"), ("results",), ("content",))
detail_paths = (("job",), ("data", "job"), ("posting",), ("result",))
closed_statuses = CLOSED_STATUSES | {"unpublished", "expired"}
def _to_job(self, record: dict[str, object], base_url: str) -> ExtractedJob:
prepared = dict(record)
if not _first_text(prepared, "url", "jobUrl", "hostedUrl", "applyUrl", "link"):
company_identifier = _first_text(prepared, "company.identifier")
posting_id = _first_text(prepared, "id", "uuid")
if company_identifier and posting_id:
prepared["url"] = (
f"https://jobs.smartrecruiters.com/{company_identifier}/{posting_id}"
)
if _find_nested(prepared, "location.remote") is True:
prepared["remote"] = True
return super()._to_job(prepared, base_url)
def _extract_records(self, payload):
for path in self.listing_paths:
value = payload
+59 -3
View File
@@ -9,17 +9,58 @@ from urllib.parse import urljoin, urlsplit
from bs4 import BeautifulSoup
from apps.sources.platforms import PLATFORM_ALERTS
from apps.sources.services.canonicalize import canonicalize_url
from .base import ExtractedJob, ExtractionResult, FieldEvidence
URL_RE = re.compile(r"https?://[^\s<>\"']+", re.I)
SKIP_TEXT = re.compile(r"unsubscribe|afmelden|uitschrijven|privacy|view in browser", re.I)
SKIP_TEXT = re.compile(
r"unsubscribe|afmelden|uitschrijven|privacy|view in browser|bekijk online|"
r"account|aanmelden|inloggen|login|voorkeuren|preferences|voorwaarden|terms|"
r"contact|help|hulp|over ons|about us",
re.I,
)
SKIP_PATH = re.compile(
r"/(?:unsubscribe|uitschrijven|afmelden|privacy|account|login|signin|preferences|"
r"settings|terms|legal|help|contact)(?:/|$)",
re.I,
)
def _matches_domain(hostname: str, domains: tuple[str, ...]) -> bool:
return any(hostname == domain or hostname.endswith(f".{domain}") for domain in domains)
class EmailAlertAdapter:
parser_key = "email-alert"
parser_version = "1.0.0"
parser_version = "1.1.0"
@staticmethod
def _provider_for_candidates(candidates: list[tuple[str, str]]) -> str:
hostnames = {
(urlsplit(href).hostname or "").lower()
for _, href in candidates
if href.lower().startswith(("http://", "https://"))
}
for provider, alert in PLATFORM_ALERTS.items():
if any(_matches_domain(hostname, alert.domains) for hostname in hostnames):
return provider
return "other"
@staticmethod
def _is_expected_platform_link(*, label: str, href: str, provider: str) -> bool:
alert = PLATFORM_ALERTS.get(provider)
if alert is None or len(label.strip()) < 4:
return False
parsed = urlsplit(href)
hostname = (parsed.hostname or "").lower()
return (
parsed.scheme.lower() == "https"
and _matches_domain(hostname, alert.domains)
and not SKIP_TEXT.search(label)
and not SKIP_PATH.search(parsed.path)
)
@staticmethod
def _decode_parts(message: Message) -> tuple[str, str]:
@@ -41,7 +82,9 @@ class EmailAlertAdapter:
html_parts.append(str(payload))
return "\n".join(plain_parts), "\n".join(html_parts)
def extract_message(self, raw_message: bytes) -> ExtractionResult:
def extract_message(
self, raw_message: bytes, *, expected_provider: str | None = None
) -> ExtractionResult:
message = BytesParser(policy=policy.default).parsebytes(raw_message)
plain, html_body = self._decode_parts(message)
candidates: list[tuple[str, str]] = []
@@ -61,6 +104,18 @@ class EmailAlertAdapter:
continue
candidates.append(("", clean_href))
alert_provider = expected_provider or self._provider_for_candidates(candidates)
if expected_provider:
candidates = [
(label, href)
for label, href in candidates
if self._is_expected_platform_link(
label=label,
href=href,
provider=expected_provider,
)
]
jobs: list[ExtractedJob] = []
seen: set[str] = set()
subject = str(message.get("subject") or "Vacature uit e-mail").strip()
@@ -84,6 +139,7 @@ class EmailAlertAdapter:
"email_subject": subject,
"email_sender": str(message.get("from") or ""),
"target_domain": hostname,
"alert_provider": alert_provider,
},
evidence=[FieldEvidence("url", "email-anchor", 0.75, label[:240])],
)
+290
View File
@@ -0,0 +1,290 @@
from __future__ import annotations
import re
from urllib.parse import parse_qs, urljoin, urlsplit
from bs4 import BeautifulSoup, Tag
from .base import ExtractedJob, ExtractionResult, FieldEvidence
class KempenEmployerAdapter:
"""Extract public employer lists around Mol from exact reviewed routes."""
parser_key = "regional-kempen-employers"
parser_version = "1.0.0"
supported_routes = {
"ziekenhuisgeel.careersite.be": {"/nl/vacatures"},
"geel.hro.be": {"/"},
"jobs.turnhout.be": {"/"},
"jobs.renotec.be": {"/nl/alle-jobs"},
"ravago.softgarden.io": {"/en/vacancies"},
"jobs.sanofi.com": {"/en/belgium"},
"www.daf.com": {"/nl-nl/werken-bij-daf/vacatures"},
}
def _route(self, url: str) -> tuple[str, str] | None:
parts = urlsplit(url)
host = (parts.hostname or "").lower()
path = parts.path.rstrip("/") or "/"
if parts.scheme != "https" or path not in self.supported_routes.get(host, set()):
return None
return host, path
def _supports_url(self, url: str) -> bool:
return self._route(url) is not None
@staticmethod
def _same_host_url(base_url: str, href: str) -> str:
job_url = urljoin(base_url, href)
parts = urlsplit(job_url)
if parts.scheme != "https":
return ""
if (parts.hostname or "").lower() != (urlsplit(base_url).hostname or "").lower():
return ""
return job_url
@staticmethod
def _job(
*,
url: str,
title: str,
employer: str,
location: str,
postal_code: str,
description: str,
external_id: str,
evidence_source: str,
) -> ExtractedJob:
return ExtractedJob(
url=url,
title=title,
employer_name=employer,
external_id=external_id,
location_text=location,
postal_code=postal_code,
region="Antwerpen",
country="BE",
description_text=description,
raw={"regional_listing": evidence_source, "region_center": "2400 Mol"},
evidence=[
FieldEvidence("title", evidence_source, 0.94, title[:240]),
FieldEvidence("employer_name", "reviewed-source", 0.97, employer),
FieldEvidence("location_text", "kempen-location-marker", 0.9, location),
],
)
def _ziekenhuis_geel(self, soup: BeautifulSoup, url: str) -> list[ExtractedJob]:
jobs = []
for card in soup.select(".vacature-tegel"):
link = card.select_one("a.vacature-tegel__link[href]")
title_node = card.select_one(".vacature-tegel__titel")
if not isinstance(link, Tag) or not isinstance(title_node, Tag):
continue
job_url = self._same_host_url(url, str(link.get("href") or ""))
match = re.fullmatch(r"/nl/vacature/(\d+)/[^/]+", urlsplit(job_url).path)
title = title_node.get_text(" ", strip=True)
if not title or not match:
continue
jobs.append(
self._job(
url=job_url,
title=title,
employer="Ziekenhuis Geel",
location="Geel, Antwerpen",
postal_code="2440",
description=card.get_text(" ", strip=True),
external_id=match.group(1),
evidence_source="ziekenhuis-geel-card",
)
)
return jobs
def _stad_geel(self, soup: BeautifulSoup, url: str) -> list[ExtractedJob]:
jobs = []
for card in soup.select(".vacatureKader[data-href]"):
href = str(card.get("data-href") or "")
match = re.fullmatch(r"vacature\.php\?id=(\d+)", href)
title_node = card.find("h4")
title = title_node.get_text(" ", strip=True) if title_node else ""
if not title or not match:
continue
jobs.append(
self._job(
url=self._same_host_url(url, href),
title=title,
employer="Lokaal bestuur Geel",
location="Geel, Antwerpen",
postal_code="2440",
description=card.get_text(" ", strip=True),
external_id=match.group(1),
evidence_source="stad-geel-hro-card",
)
)
return jobs
def _stad_turnhout(self, soup: BeautifulSoup, url: str) -> list[ExtractedJob]:
jobs = []
for link in soup.find_all("a", href=True):
href = str(link.get("href") or "")
match = re.fullmatch(r"vacature\.php\?id=(\d+)", href)
title_node = link.select_one(".block-update__body__title__inner")
title = title_node.get_text(" ", strip=True) if title_node else ""
if not title or not match:
continue
jobs.append(
self._job(
url=self._same_host_url(url, href),
title=title,
employer="Stad Turnhout",
location="Turnhout, Antwerpen",
postal_code="2300",
description=link.get_text(" ", strip=True),
external_id=match.group(1),
evidence_source="stad-turnhout-hro-card",
)
)
return jobs
def _renotec(self, soup: BeautifulSoup, url: str) -> list[ExtractedJob]:
jobs = []
for card in soup.select(".s-tile.s-card"):
card_text = card.get_text(" ", strip=True)
if "Geel" not in card_text:
continue
title_node = card.find("h3")
link = card.find("a", href=True)
title = title_node.get_text(" ", strip=True) if title_node else ""
job_url = self._same_host_url(url, str(link.get("href") or "")) if link else ""
parts = urlsplit(job_url)
query = parse_qs(parts.query)
external_id = (query.get("id") or [""])[0]
if parts.path != "/nl/detail/" or not external_id.isdigit() or not title:
continue
jobs.append(
self._job(
url=job_url,
title=title,
employer="Group Renotec",
location="Geel, Antwerpen (mogelijk meerdere werfregio's)",
postal_code="2440",
description=card_text,
external_id=external_id,
evidence_source="renotec-geel-card",
)
)
return jobs
def _ravago(self, soup: BeautifulSoup, url: str) -> list[ExtractedJob]:
jobs = []
for card in soup.select(".matchElement"):
locations = {
node.get_text(" ", strip=True) for node in card.select(".location-view-item")
}
local_places = locations.intersection({"Arendonk", "Olen"})
if not local_places:
continue
link = card.find("a", href=True)
title = link.get_text(" ", strip=True) if link else ""
job_url = self._same_host_url(url, str(link.get("href") or "")) if link else ""
match = re.fullmatch(r"/job/(\d+)/[^/]+/?", urlsplit(job_url).path)
if not title or not match:
continue
place = "Arendonk" if "Arendonk" in local_places else "Olen"
jobs.append(
self._job(
url=job_url,
title=title,
employer="Ravago",
location=f"{place}, Antwerpen",
postal_code="2370" if place == "Arendonk" else "2250",
description=card.get_text(" ", strip=True),
external_id=match.group(1),
evidence_source="ravago-softgarden-card",
)
)
return jobs
def _sanofi(self, soup: BeautifulSoup, url: str) -> list[ExtractedJob]:
jobs = []
for link in soup.select(".job-list a[data-job-id][href]"):
location_node = link.select_one(".job-location")
title_node = link.select_one(".job-title")
location = location_node.get_text(" ", strip=True) if location_node else ""
title = title_node.get_text(" ", strip=True) if title_node else ""
external_id = str(link.get("data-job-id") or "")
job_url = self._same_host_url(url, str(link.get("href") or ""))
if location != "Geel, Belgium" or not title or not external_id.isdigit():
continue
if not re.fullmatch(r"/en/job/geel/[^/]+/\d+/\d+", urlsplit(job_url).path):
continue
jobs.append(
self._job(
url=job_url,
title=title,
employer="Sanofi",
location="Geel, Antwerpen",
postal_code="2440",
description=link.get_text(" ", strip=True),
external_id=external_id,
evidence_source="sanofi-belgium-geel-card",
)
)
return jobs
def _daf(self, soup: BeautifulSoup, url: str) -> list[ExtractedJob]:
jobs = []
for card in soup.select("li.itemlist__item"):
location_node = card.select_one(".js-vac-metalocation")
title_link = card.select_one("a.js-vac-title[href]")
location = location_node.get_text(" ", strip=True) if location_node else ""
title = title_link.get_text(" ", strip=True) if title_link else ""
job_url = (
self._same_host_url(url, str(title_link.get("href") or "")) if title_link else ""
)
path = urlsplit(job_url).path
prefix = "/nl-nl/werken-bij-daf/vacatures/"
if location != "Westerlo" or not title or not path.startswith(prefix):
continue
slug = path.removeprefix(prefix).strip("/")
if not slug or "/" in slug:
continue
jobs.append(
self._job(
url=job_url,
title=title,
employer="DAF Trucks",
location="Westerlo, Antwerpen",
postal_code="2260",
description=card.get_text(" ", strip=True),
external_id=slug,
evidence_source="daf-westerlo-card",
)
)
return jobs
def extract(self, content: str, *, url: str) -> ExtractionResult:
route = self._route(url)
if route is None:
return ExtractionResult(
[], self.parser_key, self.parser_version, 0.0, ["Onherkende Kempen-bron"]
)
host, _ = route
extractors = {
"ziekenhuisgeel.careersite.be": self._ziekenhuis_geel,
"geel.hro.be": self._stad_geel,
"jobs.turnhout.be": self._stad_turnhout,
"jobs.renotec.be": self._renotec,
"ravago.softgarden.io": self._ravago,
"jobs.sanofi.com": self._sanofi,
"www.daf.com": self._daf,
}
jobs = extractors[host](BeautifulSoup(content, "lxml"), url)
unique_jobs = list({job.url: job for job in jobs if job.url}.values())
return ExtractionResult(
unique_jobs,
self.parser_key,
self.parser_version,
0.92 if unique_jobs else 0.0,
[] if unique_jobs else ["Geen actuele regionale vacatures gevonden"],
)
+253
View File
@@ -0,0 +1,253 @@
from __future__ import annotations
import re
from urllib.parse import parse_qs, urljoin, urlsplit
from bs4 import BeautifulSoup
from .base import ExtractedJob, ExtractionResult, FieldEvidence
class MolRegionEmployerAdapter:
"""Extract reviewed employer listings around postcode 2400 without detail fetches."""
parser_key = "regional-mol-employers"
parser_version = "1.0.0"
thomas_more_company_guid = "eab9ca13-ee10-4504-8a87-a785d0b037ef"
supported_routes = {
"www.sckcen.be": {"/nl/carriere/vacatures"},
"cipalschaubroeck.teamtailor.com": {"/jobs"},
"www.vanroey.be": {"/en/job-overview"},
"netropolix.recruitee.com": {"/"},
"jobpage.cvwarehouse.com": {"/"},
}
def _route(self, url: str) -> tuple[str, str] | None:
parts = urlsplit(url)
host = (parts.hostname or "").lower()
path = parts.path.rstrip("/") or "/"
if parts.scheme != "https" or path not in self.supported_routes.get(host, set()):
return None
if host == "jobpage.cvwarehouse.com":
query = parse_qs(parts.query)
if query.get("companyGuid") != [self.thomas_more_company_guid] or "job" in query:
return None
return host, path
def _supports_url(self, url: str) -> bool:
return self._route(url) is not None
@staticmethod
def _same_host_url(base_url: str, href: str) -> str:
job_url = urljoin(base_url, href)
if urlsplit(job_url).scheme != "https":
return ""
if (urlsplit(job_url).hostname or "").lower() != (
urlsplit(base_url).hostname or ""
).lower():
return ""
return job_url
@staticmethod
def _job(
*,
url: str,
title: str,
employer: str,
location: str,
postal_code: str,
description: str,
external_id: str,
evidence_source: str,
) -> ExtractedJob:
return ExtractedJob(
url=url,
title=title,
employer_name=employer,
external_id=external_id,
location_text=location,
postal_code=postal_code,
region="Antwerpen",
country="BE",
description_text=description,
raw={"regional_listing": evidence_source, "region_center": "2400 Mol"},
evidence=[
FieldEvidence("title", evidence_source, 0.94, title[:240]),
FieldEvidence("employer_name", "reviewed-source", 0.96, employer),
FieldEvidence("location_text", "mol-region-review", 0.82, location),
],
)
def _sck_cen(self, soup: BeautifulSoup, url: str) -> list[ExtractedJob]:
jobs = []
for link in soup.select("article a[href]"):
job_url = self._same_host_url(url, str(link.get("href") or ""))
path = urlsplit(job_url).path if job_url else ""
if not path.startswith("/nl/carriere/vacatures/"):
continue
title = link.get_text(" ", strip=True)
if not title:
continue
card = link.find_parent("article")
description = card.get_text(" ", strip=True) if card else title
jobs.append(
self._job(
url=job_url,
title=title,
employer="SCK CEN",
location="Mol, Antwerpen",
postal_code="2400",
description=description,
external_id=path.rstrip("/").rsplit("/", 1)[-1],
evidence_source="sck-cen-vacancy-card",
)
)
return jobs
def _cipal(self, soup: BeautifulSoup, url: str) -> list[ExtractedJob]:
jobs = []
for card in soup.select("li"):
card_text = card.get_text(" ", strip=True)
if "Westerlo" not in card_text and "Geel" not in card_text:
continue
link = card.find("a", href=True)
if not link:
continue
title = link.get_text(" ", strip=True)
job_url = self._same_host_url(url, str(link.get("href") or ""))
path = urlsplit(job_url).path if job_url else ""
match = re.fullmatch(r"/jobs/(\d+)-[^/]+", path.rstrip("/"))
if not title or not match:
continue
location = "Geel, Antwerpen" if "Geel" in card_text else "Westerlo, Antwerpen"
jobs.append(
self._job(
url=job_url,
title=title,
employer="Cipal Schaubroeck",
location=location,
postal_code="2440" if "Geel" in card_text else "2260",
description=card_text,
external_id=match.group(1),
evidence_source="cipal-teamtailor-card",
)
)
return jobs
def _vanroey(self, soup: BeautifulSoup, url: str) -> list[ExtractedJob]:
jobs = []
for heading in soup.select("h3.elementor-heading-title"):
link = heading.find("a", href=True)
if not link:
continue
title = link.get_text(" ", strip=True)
job_url = self._same_host_url(url, str(link.get("href") or ""))
path = urlsplit(job_url).path if job_url else ""
if not title or not re.fullmatch(r"/en/job/[^/]+/", path):
continue
slug = path.rstrip("/").rsplit("/", 1)[-1]
if "oost-vlaanderen" in slug:
continue
card = heading.find_parent("section") or heading.parent
description = card.get_text(" ", strip=True) if card else title
jobs.append(
self._job(
url=job_url,
title=title,
employer="VanRoey",
location="Turnhout/Geel (hybride; controleer vacaturedetail)",
postal_code="",
description=description,
external_id=slug,
evidence_source="vanroey-job-card",
)
)
return jobs
def _netropolix(self, soup: BeautifulSoup, url: str) -> list[ExtractedJob]:
jobs = []
for link in soup.find_all("a", href=True):
title = link.get_text(" ", strip=True)
if "Geel" not in title:
continue
job_url = self._same_host_url(url, str(link.get("href") or ""))
path = urlsplit(job_url).path if job_url else ""
if not re.fullmatch(r"/o/[^/]+", path.rstrip("/")):
continue
card = link.find_parent("div")
description = card.get_text(" ", strip=True) if card else title
jobs.append(
self._job(
url=job_url,
title=title,
employer="NTX (Netropolix)",
location="Geel, Antwerpen",
postal_code="2440",
description=description,
external_id=path.rstrip("/").rsplit("/", 1)[-1],
evidence_source="netropolix-recruitee-card",
)
)
return jobs
def _thomas_more(self, soup: BeautifulSoup, url: str) -> list[ExtractedJob]:
jobs = []
for link in soup.select("a.jobLink[data-item='readmore'][data-jobid][href]"):
title = link.get_text(" ", strip=True)
title_casefold = title.casefold()
if not any(place in title_casefold for place in ("geel", "turnhout", "vorselaar")):
continue
external_id = str(link.get("data-jobid") or "")
if not external_id.isdigit():
continue
job_url = self._same_host_url(url, str(link.get("href") or ""))
query = parse_qs(urlsplit(job_url).query) if job_url else {}
if query.get("companyGuid") != [self.thomas_more_company_guid] or query.get("job") != [
external_id
]:
continue
if "turnhout" in title_casefold:
location, postal_code = "Turnhout, Antwerpen", "2300"
elif "vorselaar" in title_casefold:
location, postal_code = "Vorselaar, Antwerpen", "2290"
else:
location, postal_code = "Geel, Antwerpen", "2440"
jobs.append(
self._job(
url=job_url,
title=title,
employer="Thomas More",
location=location,
postal_code=postal_code,
description=f"Publieke regionale vacature bij Thomas More: {title}.",
external_id=external_id,
evidence_source="thomas-more-cvwarehouse-card",
)
)
return jobs
def extract(self, content: str, *, url: str) -> ExtractionResult:
route = self._route(url)
if route is None:
return ExtractionResult(
[], self.parser_key, self.parser_version, 0.0, ["Onherkende Mol-regiobron"]
)
host, _ = route
soup = BeautifulSoup(content, "lxml")
extractors = {
"www.sckcen.be": self._sck_cen,
"cipalschaubroeck.teamtailor.com": self._cipal,
"www.vanroey.be": self._vanroey,
"netropolix.recruitee.com": self._netropolix,
"jobpage.cvwarehouse.com": self._thomas_more,
}
jobs = extractors[host](soup, url)
unique_jobs = list({job.url: job for job in jobs}.values())
warnings = [] if unique_jobs else ["Geen actuele vacatures rond Mol gevonden"]
return ExtractionResult(
unique_jobs,
self.parser_key,
self.parser_version,
0.9 if unique_jobs else 0.0,
warnings,
)
+312
View File
@@ -0,0 +1,312 @@
from __future__ import annotations
from urllib.parse import urljoin, urlsplit
from bs4 import BeautifulSoup
from .base import ExtractedJob, ExtractionResult, FieldEvidence
class CordaCampusAdapter:
"""Extract the public job cards intentionally published by Corda Campus."""
parser_key = "regional-corda-campus"
parser_version = "1.0.0"
source_hosts = ("cordacampus.com",)
def _supports_url(self, url: str) -> bool:
parts = urlsplit(url)
host = (parts.hostname or "").lower()
return host.endswith(self.source_hosts) and parts.path.rstrip("/") == "/jobs"
def extract(self, content: str, *, url: str) -> ExtractionResult:
if not self._supports_url(url):
return ExtractionResult(
[], self.parser_key, self.parser_version, 0.0, ["Onherkenbare regiobron"]
)
source_host = (urlsplit(url).hostname or "").lower()
soup = BeautifulSoup(content, "lxml")
jobs: list[ExtractedJob] = []
seen_urls: set[str] = set()
for card in soup.select('a.event-item[href*="/job/"]'):
job_url = urljoin(url, str(card.get("href") or ""))
job_host = (urlsplit(job_url).hostname or "").lower()
if job_host != source_host or job_url in seen_urls:
continue
title_node = card.select_one(".job-title")
employer_node = card.select_one(".company-title")
title = title_node.get_text(" ", strip=True) if title_node else ""
employer = employer_node.get_text(" ", strip=True) if employer_node else ""
if not title:
continue
date_node = card.select_one(".bottom-info")
date_posted = date_node.get_text(" ", strip=True) if date_node else ""
external_id = urlsplit(job_url).path.rstrip("/").rsplit("/", 1)[-1]
description = f"Vacature van {employer or 'een Corda-werkgever'} via Corda Campus."
jobs.append(
ExtractedJob(
url=job_url,
title=title,
employer_name=employer,
external_id=external_id,
location_text="Hasselt, Limburg",
region="Limburg",
country="BE",
description_text=description,
date_posted=date_posted,
raw={"regional_listing": "corda-campus"},
evidence=[
FieldEvidence("title", "corda-job-card", 0.95, title[:240]),
FieldEvidence("employer_name", "corda-job-card", 0.92, employer[:240]),
FieldEvidence(
"location_text",
"regional-source-scope",
0.72,
"Corda Campus, Hasselt",
),
],
)
)
seen_urls.add(job_url)
warnings = [] if jobs else ["Geen actuele Corda-vacatures gevonden"]
return ExtractionResult(
jobs,
self.parser_key,
self.parser_version,
0.9 if jobs else 0.0,
warnings,
)
class LocalEmployerListingAdapter:
"""Extract reviewed public listing pages of employers in the Hasselt-Genk area."""
parser_key = "regional-local-employers"
parser_version = "1.0.0"
supported_routes = {
"acagroup.be": {"/en/jobs"},
"www.xploregroup.be": {"/en/jobs"},
"www.uhasselt.be": {"/vacatures"},
"ses.pxl.be": {"/"},
"ziekenhuis-oost-limburg.cvw.io": {"/"},
}
def _supports_url(self, url: str) -> bool:
return self._route(url) is not None
def _route(self, url: str) -> tuple[str, str] | None:
parts = urlsplit(url)
host = (parts.hostname or "").lower()
path = parts.path.rstrip("/") or "/"
if parts.scheme != "https" or path not in self.supported_routes.get(host, set()):
return None
return host, path
@staticmethod
def _job(
*,
url: str,
title: str,
employer: str,
location: str,
description: str,
external_id: str,
evidence_source: str,
) -> ExtractedJob:
return ExtractedJob(
url=url,
title=title,
employer_name=employer,
external_id=external_id,
location_text=location,
region="Limburg",
country="BE",
description_text=description,
raw={"regional_listing": evidence_source},
evidence=[
FieldEvidence("title", evidence_source, 0.94, title[:240]),
FieldEvidence("employer_name", "reviewed-source", 0.96, employer),
FieldEvidence("location_text", "regional-source-scope", 0.78, location),
],
)
@staticmethod
def _same_host_url(base_url: str, href: str) -> str:
job_url = urljoin(base_url, href)
if (urlsplit(job_url).hostname or "").lower() != (
urlsplit(base_url).hostname or ""
).lower():
return ""
return job_url
def _aca(self, soup: BeautifulSoup, url: str) -> list[ExtractedJob]:
jobs = []
for link in soup.find_all("a", href=True):
href = str(link.get("href") or "")
if not href.startswith("/en/jobs/"):
continue
title_node = link.find("h3")
title = title_node.get_text(" ", strip=True) if title_node else ""
job_url = self._same_host_url(url, href)
if not title or not job_url:
continue
description_node = link.find("p")
description = (
description_node.get_text(" ", strip=True)
if description_node
else f"Vacature bij ACA Group: {title}."
)
jobs.append(
self._job(
url=job_url,
title=title,
employer="ACA Group",
location="Hasselt (hybride; kantoorselectie per vacature)",
description=description,
external_id=urlsplit(job_url).path.rstrip("/").rsplit("/", 1)[-1],
evidence_source="aca-job-card",
)
)
return jobs
def _xplore(self, soup: BeautifulSoup, url: str) -> list[ExtractedJob]:
jobs = []
for heading in soup.find_all("h3"):
link = heading.find("a", href=True)
if not link or not str(link.get("href") or "").startswith("/en/jobs/"):
continue
card = heading.parent
location_node = next(
(
node
for node in card.find_all("p")
if "Hasselt" in node.get_text(" ", strip=True)
),
None,
)
if not location_node:
continue
title = link.get_text(" ", strip=True)
job_url = self._same_host_url(url, str(link.get("href") or ""))
if not title or not job_url:
continue
location = location_node.get_text(" ", strip=True)
jobs.append(
self._job(
url=job_url,
title=title,
employer="Xplore Group",
location=location,
description=(
f"Vacature bij Xplore Group met Hasselt als mogelijke werklocatie: {title}."
),
external_id=urlsplit(job_url).path.rstrip("/").rsplit("/", 1)[-1],
evidence_source="xplore-job-card",
)
)
return jobs
def _uhasselt(self, soup: BeautifulSoup, url: str) -> list[ExtractedJob]:
jobs = []
for card in soup.select("section.vacancy-item"):
title_node = card.find("h3")
link = card.find("a", href=True)
title = title_node.get_text(" ", strip=True) if title_node else ""
job_url = self._same_host_url(url, str(link.get("href") or "")) if link else ""
if not title or not job_url or "/vacatures/detail/" not in urlsplit(job_url).path:
continue
external_id = urlsplit(job_url).path.split("/detail/", 1)[-1].split("-", 1)[0]
jobs.append(
self._job(
url=job_url,
title=title,
employer="Universiteit Hasselt",
location="Hasselt/Diepenbeek, Limburg",
description=card.get_text(" ", strip=True),
external_id=external_id,
evidence_source="uhasselt-vacancy-card",
)
)
return jobs
def _pxl(self, soup: BeautifulSoup, url: str) -> list[ExtractedJob]:
jobs = []
for card in soup.select(".vacature-card[id^='Vacature_']"):
external_id = str(card.get("id") or "").removeprefix("Vacature_")
title_node = card.select_one(".vacature-card-titel")
title = title_node.get_text(" ", strip=True) if title_node else ""
if not title or not external_id.isdigit():
continue
cells = [
node.get_text(" ", strip=True) for node in card.select("td.vacature-card-td-text")
]
campus = next((value for value in cells if value.startswith("Campus ")), "Hasselt")
job_url = f"{url.rstrip('/')}?vacature_id={external_id}"
jobs.append(
self._job(
url=job_url,
title=title,
employer="Hogeschool PXL",
location=f"{campus}, Limburg",
description=" · ".join(cells),
external_id=external_id,
evidence_source="pxl-vacancy-card",
)
)
return jobs
def _zol(self, soup: BeautifulSoup, url: str) -> list[ExtractedJob]:
jobs = []
for link in soup.select("a[data-item='readmore'][data-jobid][href]"):
external_id = str(link.get("data-jobid") or "")
title_node = link.select_one(".job-title")
title = title_node.get_text(" ", strip=True) if title_node else ""
job_url = self._same_host_url(url, str(link.get("href") or ""))
if not title or not external_id.isdigit() or not job_url:
continue
jobs.append(
self._job(
url=job_url,
title=title,
employer="Ziekenhuis Oost-Limburg",
location="Genk/Lanaken/Maaseik, Limburg",
description=f"Publieke vacature van Ziekenhuis Oost-Limburg: {title}.",
external_id=external_id,
evidence_source="zol-cvwarehouse-card",
)
)
return jobs
def extract(self, content: str, *, url: str) -> ExtractionResult:
route = self._route(url)
if route is None:
return ExtractionResult(
[],
self.parser_key,
self.parser_version,
0.0,
["Onherkenbare lokale werkgeversbron"],
)
host, _ = route
soup = BeautifulSoup(content, "lxml")
extractors = {
"acagroup.be": self._aca,
"www.xploregroup.be": self._xplore,
"www.uhasselt.be": self._uhasselt,
"ses.pxl.be": self._pxl,
"ziekenhuis-oost-limburg.cvw.io": self._zol,
}
jobs = extractors[host](soup, url)
unique_jobs = list({job.url: job for job in jobs}.values())
warnings = [] if unique_jobs else ["Geen actuele lokale vacatures gevonden"]
return ExtractionResult(
unique_jobs,
self.parser_key,
self.parser_version,
0.9 if unique_jobs else 0.0,
warnings,
)
+11
View File
@@ -12,6 +12,9 @@ from .ats import (
from .base import ExtractionResult
from .generic_html import GenericHtmlAdapter
from .jsonld import JsonLdJobPostingAdapter
from .kempen import KempenEmployerAdapter
from .mol_region import MolRegionEmployerAdapter
from .regional import CordaCampusAdapter, LocalEmployerListingAdapter
from .rss import RssAdapter
@@ -22,10 +25,18 @@ class AdapterRegistry:
self.recruitee = RecruiteeAdapter()
self.smartrecruiters = SmartRecruitersAdapter()
self.workable = WorkableAdapter()
self.corda_campus = CordaCampusAdapter()
self.local_employers = LocalEmployerListingAdapter()
self.mol_region_employers = MolRegionEmployerAdapter()
self.kempen_employers = KempenEmployerAdapter()
self.jsonld = JsonLdJobPostingAdapter()
self.generic = GenericHtmlAdapter()
self.rss = RssAdapter()
self.providers = [
self.corda_campus,
self.local_employers,
self.mol_region_employers,
self.kempen_employers,
self.greenhouse,
self.lever,
self.recruitee,