feat: release regional radar and mailbox integrations
deploy / deploy (push) Canceled after 0s

This commit is contained in:
Jens
2026-07-22 05:12:07 +02:00
parent 598d3ec18a
commit 551d0f46c2
131 changed files with 6209 additions and 336 deletions
+59 -3
View File
@@ -9,17 +9,58 @@ from urllib.parse import urljoin, urlsplit
from bs4 import BeautifulSoup
from apps.sources.platforms import PLATFORM_ALERTS
from apps.sources.services.canonicalize import canonicalize_url
from .base import ExtractedJob, ExtractionResult, FieldEvidence
URL_RE = re.compile(r"https?://[^\s<>\"']+", re.I)
SKIP_TEXT = re.compile(r"unsubscribe|afmelden|uitschrijven|privacy|view in browser", re.I)
SKIP_TEXT = re.compile(
r"unsubscribe|afmelden|uitschrijven|privacy|view in browser|bekijk online|"
r"account|aanmelden|inloggen|login|voorkeuren|preferences|voorwaarden|terms|"
r"contact|help|hulp|over ons|about us",
re.I,
)
SKIP_PATH = re.compile(
r"/(?:unsubscribe|uitschrijven|afmelden|privacy|account|login|signin|preferences|"
r"settings|terms|legal|help|contact)(?:/|$)",
re.I,
)
def _matches_domain(hostname: str, domains: tuple[str, ...]) -> bool:
return any(hostname == domain or hostname.endswith(f".{domain}") for domain in domains)
class EmailAlertAdapter:
parser_key = "email-alert"
parser_version = "1.0.0"
parser_version = "1.1.0"
@staticmethod
def _provider_for_candidates(candidates: list[tuple[str, str]]) -> str:
hostnames = {
(urlsplit(href).hostname or "").lower()
for _, href in candidates
if href.lower().startswith(("http://", "https://"))
}
for provider, alert in PLATFORM_ALERTS.items():
if any(_matches_domain(hostname, alert.domains) for hostname in hostnames):
return provider
return "other"
@staticmethod
def _is_expected_platform_link(*, label: str, href: str, provider: str) -> bool:
alert = PLATFORM_ALERTS.get(provider)
if alert is None or len(label.strip()) < 4:
return False
parsed = urlsplit(href)
hostname = (parsed.hostname or "").lower()
return (
parsed.scheme.lower() == "https"
and _matches_domain(hostname, alert.domains)
and not SKIP_TEXT.search(label)
and not SKIP_PATH.search(parsed.path)
)
@staticmethod
def _decode_parts(message: Message) -> tuple[str, str]:
@@ -41,7 +82,9 @@ class EmailAlertAdapter:
html_parts.append(str(payload))
return "\n".join(plain_parts), "\n".join(html_parts)
def extract_message(self, raw_message: bytes) -> ExtractionResult:
def extract_message(
self, raw_message: bytes, *, expected_provider: str | None = None
) -> ExtractionResult:
message = BytesParser(policy=policy.default).parsebytes(raw_message)
plain, html_body = self._decode_parts(message)
candidates: list[tuple[str, str]] = []
@@ -61,6 +104,18 @@ class EmailAlertAdapter:
continue
candidates.append(("", clean_href))
alert_provider = expected_provider or self._provider_for_candidates(candidates)
if expected_provider:
candidates = [
(label, href)
for label, href in candidates
if self._is_expected_platform_link(
label=label,
href=href,
provider=expected_provider,
)
]
jobs: list[ExtractedJob] = []
seen: set[str] = set()
subject = str(message.get("subject") or "Vacature uit e-mail").strip()
@@ -84,6 +139,7 @@ class EmailAlertAdapter:
"email_subject": subject,
"email_sender": str(message.get("from") or ""),
"target_domain": hostname,
"alert_provider": alert_provider,
},
evidence=[FieldEvidence("url", "email-anchor", 0.75, label[:240])],
)