This commit is contained in:
@@ -9,17 +9,58 @@ from urllib.parse import urljoin, urlsplit
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
from apps.sources.platforms import PLATFORM_ALERTS
|
||||
from apps.sources.services.canonicalize import canonicalize_url
|
||||
|
||||
from .base import ExtractedJob, ExtractionResult, FieldEvidence
|
||||
|
||||
URL_RE = re.compile(r"https?://[^\s<>\"']+", re.I)
|
||||
SKIP_TEXT = re.compile(r"unsubscribe|afmelden|uitschrijven|privacy|view in browser", re.I)
|
||||
SKIP_TEXT = re.compile(
|
||||
r"unsubscribe|afmelden|uitschrijven|privacy|view in browser|bekijk online|"
|
||||
r"account|aanmelden|inloggen|login|voorkeuren|preferences|voorwaarden|terms|"
|
||||
r"contact|help|hulp|over ons|about us",
|
||||
re.I,
|
||||
)
|
||||
SKIP_PATH = re.compile(
|
||||
r"/(?:unsubscribe|uitschrijven|afmelden|privacy|account|login|signin|preferences|"
|
||||
r"settings|terms|legal|help|contact)(?:/|$)",
|
||||
re.I,
|
||||
)
|
||||
|
||||
|
||||
def _matches_domain(hostname: str, domains: tuple[str, ...]) -> bool:
|
||||
return any(hostname == domain or hostname.endswith(f".{domain}") for domain in domains)
|
||||
|
||||
|
||||
class EmailAlertAdapter:
|
||||
parser_key = "email-alert"
|
||||
parser_version = "1.0.0"
|
||||
parser_version = "1.1.0"
|
||||
|
||||
@staticmethod
|
||||
def _provider_for_candidates(candidates: list[tuple[str, str]]) -> str:
|
||||
hostnames = {
|
||||
(urlsplit(href).hostname or "").lower()
|
||||
for _, href in candidates
|
||||
if href.lower().startswith(("http://", "https://"))
|
||||
}
|
||||
for provider, alert in PLATFORM_ALERTS.items():
|
||||
if any(_matches_domain(hostname, alert.domains) for hostname in hostnames):
|
||||
return provider
|
||||
return "other"
|
||||
|
||||
@staticmethod
|
||||
def _is_expected_platform_link(*, label: str, href: str, provider: str) -> bool:
|
||||
alert = PLATFORM_ALERTS.get(provider)
|
||||
if alert is None or len(label.strip()) < 4:
|
||||
return False
|
||||
parsed = urlsplit(href)
|
||||
hostname = (parsed.hostname or "").lower()
|
||||
return (
|
||||
parsed.scheme.lower() == "https"
|
||||
and _matches_domain(hostname, alert.domains)
|
||||
and not SKIP_TEXT.search(label)
|
||||
and not SKIP_PATH.search(parsed.path)
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _decode_parts(message: Message) -> tuple[str, str]:
|
||||
@@ -41,7 +82,9 @@ class EmailAlertAdapter:
|
||||
html_parts.append(str(payload))
|
||||
return "\n".join(plain_parts), "\n".join(html_parts)
|
||||
|
||||
def extract_message(self, raw_message: bytes) -> ExtractionResult:
|
||||
def extract_message(
|
||||
self, raw_message: bytes, *, expected_provider: str | None = None
|
||||
) -> ExtractionResult:
|
||||
message = BytesParser(policy=policy.default).parsebytes(raw_message)
|
||||
plain, html_body = self._decode_parts(message)
|
||||
candidates: list[tuple[str, str]] = []
|
||||
@@ -61,6 +104,18 @@ class EmailAlertAdapter:
|
||||
continue
|
||||
candidates.append(("", clean_href))
|
||||
|
||||
alert_provider = expected_provider or self._provider_for_candidates(candidates)
|
||||
if expected_provider:
|
||||
candidates = [
|
||||
(label, href)
|
||||
for label, href in candidates
|
||||
if self._is_expected_platform_link(
|
||||
label=label,
|
||||
href=href,
|
||||
provider=expected_provider,
|
||||
)
|
||||
]
|
||||
|
||||
jobs: list[ExtractedJob] = []
|
||||
seen: set[str] = set()
|
||||
subject = str(message.get("subject") or "Vacature uit e-mail").strip()
|
||||
@@ -84,6 +139,7 @@ class EmailAlertAdapter:
|
||||
"email_subject": subject,
|
||||
"email_sender": str(message.get("from") or ""),
|
||||
"target_domain": hostname,
|
||||
"alert_provider": alert_provider,
|
||||
},
|
||||
evidence=[FieldEvidence("url", "email-anchor", 0.75, label[:240])],
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user