@@ -0,0 +1,100 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
import re
|
||||
from email import policy
|
||||
from email.message import Message
|
||||
from email.parser import BytesParser
|
||||
from urllib.parse import urljoin, urlsplit
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
from apps.sources.services.canonicalize import canonicalize_url
|
||||
|
||||
from .base import ExtractedJob, ExtractionResult, FieldEvidence
|
||||
|
||||
URL_RE = re.compile(r"https?://[^\s<>\"']+", re.I)
|
||||
SKIP_TEXT = re.compile(r"unsubscribe|afmelden|uitschrijven|privacy|view in browser", re.I)
|
||||
|
||||
|
||||
class EmailAlertAdapter:
|
||||
parser_key = "email-alert"
|
||||
parser_version = "1.0.0"
|
||||
|
||||
@staticmethod
|
||||
def _decode_parts(message: Message) -> tuple[str, str]:
|
||||
plain_parts: list[str] = []
|
||||
html_parts: list[str] = []
|
||||
for part in message.walk() if message.is_multipart() else [message]:
|
||||
disposition = str(part.get("Content-Disposition") or "")
|
||||
if "attachment" in disposition.lower():
|
||||
continue
|
||||
content_type = part.get_content_type()
|
||||
try:
|
||||
payload = part.get_content()
|
||||
except Exception:
|
||||
raw = part.get_payload(decode=True) or b""
|
||||
payload = raw.decode(part.get_content_charset() or "utf-8", errors="replace")
|
||||
if content_type == "text/plain":
|
||||
plain_parts.append(str(payload))
|
||||
elif content_type == "text/html":
|
||||
html_parts.append(str(payload))
|
||||
return "\n".join(plain_parts), "\n".join(html_parts)
|
||||
|
||||
def extract_message(self, raw_message: bytes) -> ExtractionResult:
|
||||
message = BytesParser(policy=policy.default).parsebytes(raw_message)
|
||||
plain, html_body = self._decode_parts(message)
|
||||
candidates: list[tuple[str, str]] = []
|
||||
if html_body:
|
||||
soup = BeautifulSoup(html_body, "lxml")
|
||||
for anchor in soup.find_all("a", href=True):
|
||||
label = " ".join(anchor.get_text(" ", strip=True).split())
|
||||
href = html.unescape(str(anchor["href"]).strip())
|
||||
if not href.lower().startswith(("http://", "https://")):
|
||||
continue
|
||||
if SKIP_TEXT.search(label) or SKIP_TEXT.search(href):
|
||||
continue
|
||||
candidates.append((label, href))
|
||||
for href in URL_RE.findall(plain):
|
||||
clean_href = href.rstrip(".,);]")
|
||||
if SKIP_TEXT.search(clean_href):
|
||||
continue
|
||||
candidates.append(("", clean_href))
|
||||
|
||||
jobs: list[ExtractedJob] = []
|
||||
seen: set[str] = set()
|
||||
subject = str(message.get("subject") or "Vacature uit e-mail").strip()
|
||||
for label, href in candidates:
|
||||
canonical = canonicalize_url(urljoin("https://invalid.local/", href))
|
||||
if not canonical or canonical in seen:
|
||||
continue
|
||||
seen.add(canonical)
|
||||
hostname = urlsplit(canonical).hostname or ""
|
||||
title = label if len(label) >= 4 else subject
|
||||
if len(title) > 300:
|
||||
title = title[:300]
|
||||
jobs.append(
|
||||
ExtractedJob(
|
||||
url=canonical,
|
||||
title=title,
|
||||
employer_name="",
|
||||
description_text=plain[:5000]
|
||||
or BeautifulSoup(html_body, "lxml").get_text("\n", strip=True)[:5000],
|
||||
raw={
|
||||
"email_subject": subject,
|
||||
"email_sender": str(message.get("from") or ""),
|
||||
"target_domain": hostname,
|
||||
},
|
||||
evidence=[FieldEvidence("url", "email-anchor", 0.75, label[:240])],
|
||||
)
|
||||
)
|
||||
return ExtractionResult(
|
||||
jobs,
|
||||
self.parser_key,
|
||||
self.parser_version,
|
||||
0.65 if jobs else 0.0,
|
||||
[] if jobs else ["Geen vacaturelinks in e-mail gevonden"],
|
||||
)
|
||||
|
||||
def extract(self, content: str, *, url: str = "") -> ExtractionResult:
|
||||
return self.extract_message(content.encode("utf-8", errors="replace"))
|
||||
Reference in New Issue
Block a user