294 lines
9.5 KiB
Python
294 lines
9.5 KiB
Python
from __future__ import annotations
|
|
|
|
from collections.abc import Iterable
|
|
from dataclasses import dataclass
|
|
from datetime import datetime, timedelta
|
|
from urllib.parse import urljoin, urlsplit, urlunsplit
|
|
|
|
import httpx
|
|
from django.conf import settings
|
|
from django.utils import timezone
|
|
|
|
from apps.sources.models import Source, SourceRobotsCache
|
|
|
|
from .tls import trusted_tls_context
|
|
from .url_security import UnsafeUrlError, validate_public_url
|
|
|
|
ALLOW = "allow"
|
|
DISALLOW = "disallow"
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class RobotsDecision:
|
|
allowed: bool
|
|
reason: str
|
|
|
|
|
|
def _origin_for(url: str) -> str:
|
|
parts = urlsplit(url)
|
|
if not parts.scheme:
|
|
raise ValueError("Ongeldige URL voor robotscontrole.")
|
|
host = (parts.hostname or "").lower().rstrip(".")
|
|
if not host:
|
|
raise ValueError("Host ontbreekt voor robotscontrole.")
|
|
port = parts.port
|
|
if (
|
|
(parts.scheme == "http" and port == 80)
|
|
or (parts.scheme == "https" and port == 443)
|
|
or not port
|
|
):
|
|
netloc = host
|
|
else:
|
|
netloc = f"{host}:{port}"
|
|
return f"{parts.scheme}://{netloc}"
|
|
|
|
|
|
def _path_for(url: str) -> str:
|
|
path = urlsplit(url).path or "/"
|
|
return path if path.startswith("/") else f"/{path}"
|
|
|
|
|
|
def _robots_url(origin: str) -> str:
|
|
parts = urlsplit(origin)
|
|
return urlunsplit((parts.scheme, parts.netloc, "/robots.txt", "", ""))
|
|
|
|
|
|
def _rules_from_text(text: str) -> dict[str, dict[str, list[str]]]:
|
|
bucket: dict[str, dict[str, list[str]]] = {}
|
|
active_agents: set[str] = set()
|
|
|
|
for raw_line in text.splitlines():
|
|
line = raw_line.split("#", 1)[0].strip()
|
|
if ":" not in line:
|
|
continue
|
|
key, value = (part.strip() for part in line.split(":", 1))
|
|
if not key:
|
|
continue
|
|
key_lower = key.lower()
|
|
if key_lower == "user-agent":
|
|
token = value.lower()
|
|
active_agents = {token} if token else set()
|
|
continue
|
|
if key_lower not in {ALLOW, DISALLOW}:
|
|
continue
|
|
if not active_agents:
|
|
continue
|
|
for agent in active_agents:
|
|
section = bucket.setdefault(agent, {ALLOW: [], DISALLOW: []})
|
|
section[key_lower].append(value.strip() or "/")
|
|
|
|
for entry in bucket.values():
|
|
entry[ALLOW] = list(dict.fromkeys(entry[ALLOW]))
|
|
entry[DISALLOW] = list(dict.fromkeys(entry[DISALLOW]))
|
|
return bucket
|
|
|
|
|
|
def _pick_rules(rules: dict[str, dict[str, list[str]]], user_agent: str) -> dict[str, list[str]]:
|
|
normalized = user_agent.lower()
|
|
selected = {ALLOW: [], DISALLOW: []}
|
|
|
|
for agent, values in rules.items():
|
|
if agent == "*" or (agent and agent in normalized):
|
|
selected[ALLOW].extend(values[ALLOW])
|
|
selected[DISALLOW].extend(values[DISALLOW])
|
|
|
|
selected[ALLOW] = list(dict.fromkeys(selected[ALLOW]))
|
|
selected[DISALLOW] = list(dict.fromkeys(selected[DISALLOW]))
|
|
return selected
|
|
|
|
|
|
def _longest_prefix(path: str, rules: Iterable[str]) -> int:
|
|
return max(
|
|
(len(rule.rstrip("/")) for rule in rules if rule and path.startswith(rule)), default=0
|
|
)
|
|
|
|
|
|
def _evaluate_path(
|
|
path: str, user_agent: str, rules: dict[str, dict[str, list[str]]]
|
|
) -> RobotsDecision:
|
|
selected = _pick_rules(rules, user_agent=user_agent)
|
|
allow_len = _longest_prefix(path, selected[ALLOW])
|
|
disallow_len = _longest_prefix(path, selected[DISALLOW])
|
|
if disallow_len > allow_len:
|
|
return RobotsDecision(False, "Toegang geblokkeerd door robotsregels")
|
|
return RobotsDecision(True, "Robotsregels staan toegang toe")
|
|
|
|
|
|
def _cache_ttl_seconds() -> float:
|
|
return float(getattr(settings, "ROBOTS_CACHE_TTL_SECONDS", 3600))
|
|
|
|
|
|
def _max_bytes() -> int:
|
|
return int(getattr(settings, "ROBOTS_MAX_BYTES", 131072))
|
|
|
|
|
|
def _fetch_robots(origin: str, *, client: httpx.Client | None = None) -> tuple[str, int, str, str]:
|
|
robots_url = _robots_url(origin)
|
|
own_client = client is None
|
|
http_client = client or httpx.Client(
|
|
timeout=httpx.Timeout(getattr(settings, "ROBOTS_FETCH_TIMEOUT_SECONDS", 5)),
|
|
follow_redirects=False,
|
|
verify=trusted_tls_context(),
|
|
)
|
|
try:
|
|
current_url = robots_url
|
|
for _ in range(getattr(settings, "FETCHER_MAX_REDIRECTS", 5) + 1):
|
|
validate_public_url(
|
|
current_url,
|
|
allow_nonstandard_ports=getattr(settings, "FETCHER_ALLOW_NONSTANDARD_PORTS", False),
|
|
)
|
|
response = http_client.get(
|
|
current_url,
|
|
headers={"User-Agent": getattr(settings, "FETCHER_USER_AGENT", "VacatureRadar")},
|
|
)
|
|
if response.status_code not in {301, 302, 303, 307, 308}:
|
|
break
|
|
location = response.headers.get("location")
|
|
if not location:
|
|
raise httpx.HTTPStatusError(
|
|
"Robotsredirect zonder Location-header",
|
|
request=response.request,
|
|
response=response,
|
|
)
|
|
current_url = urljoin(current_url, location)
|
|
else:
|
|
raise httpx.TooManyRedirects(
|
|
"Te veel redirects bij robotscontrole", request=response.request
|
|
)
|
|
finally:
|
|
if own_client:
|
|
http_client.close()
|
|
return (
|
|
response.text,
|
|
response.status_code,
|
|
response.headers.get("etag", ""),
|
|
response.headers.get("last-modified", ""),
|
|
)
|
|
|
|
|
|
def _load_cached(origin: str, now: datetime) -> SourceRobotsCache | None:
|
|
try:
|
|
cache = SourceRobotsCache.objects.get(origin=origin)
|
|
except SourceRobotsCache.DoesNotExist:
|
|
return None
|
|
if cache.expires_at <= now:
|
|
return None
|
|
return cache
|
|
|
|
|
|
def _load_stale_cache(origin: str) -> SourceRobotsCache | None:
|
|
try:
|
|
return SourceRobotsCache.objects.get(origin=origin)
|
|
except SourceRobotsCache.DoesNotExist:
|
|
return None
|
|
|
|
|
|
def _persist_cache(
|
|
origin: str,
|
|
*,
|
|
status_code: int,
|
|
content: str,
|
|
etag: str = "",
|
|
last_modified: str = "",
|
|
now: datetime,
|
|
) -> SourceRobotsCache:
|
|
max_bytes = _max_bytes()
|
|
byte_length = len(content.encode("utf-8"))
|
|
if byte_length > max_bytes:
|
|
return SourceRobotsCache.objects.update_or_create(
|
|
origin=origin,
|
|
defaults={
|
|
"expires_at": now + timedelta(seconds=_cache_ttl_seconds()),
|
|
"allow_rules": {},
|
|
"disallow_rules": {},
|
|
"error": f"robots.txt te groot ({byte_length} bytes)",
|
|
"etag": etag,
|
|
"last_modified": last_modified,
|
|
"byte_length": byte_length,
|
|
},
|
|
)[0]
|
|
|
|
rules = {} if status_code in {404, 410} or status_code >= 400 else _rules_from_text(content)
|
|
|
|
return SourceRobotsCache.objects.update_or_create(
|
|
origin=origin,
|
|
defaults={
|
|
"expires_at": now + timedelta(seconds=_cache_ttl_seconds()),
|
|
"allow_rules": {agent: value[ALLOW] for agent, value in rules.items()},
|
|
"disallow_rules": {agent: value[DISALLOW] for agent, value in rules.items()},
|
|
"error": "",
|
|
"etag": etag,
|
|
"last_modified": last_modified,
|
|
"byte_length": byte_length,
|
|
},
|
|
)[0]
|
|
|
|
|
|
def _build_ruleset(cache: SourceRobotsCache) -> dict[str, dict[str, list[str]]]:
|
|
rules = {"allow": {}, "disallow": {}}
|
|
for agent, entries in cache.allow_rules.items():
|
|
rules["allow"][agent] = list(entries)
|
|
for agent, entries in cache.disallow_rules.items():
|
|
rules["disallow"][agent] = list(entries)
|
|
|
|
normalized_rules: dict[str, dict[str, list[str]]] = {}
|
|
all_agents = set(rules["allow"].keys()) | set(rules["disallow"].keys())
|
|
for agent in all_agents:
|
|
normalized_rules[agent] = {
|
|
ALLOW: rules["allow"].get(agent, []),
|
|
DISALLOW: rules["disallow"].get(agent, []),
|
|
}
|
|
return normalized_rules
|
|
|
|
|
|
def assess_robots(
|
|
url: str,
|
|
*,
|
|
source: Source | None = None,
|
|
user_agent: str | None = None,
|
|
client: httpx.Client | None = None,
|
|
now: datetime | None = None,
|
|
) -> RobotsDecision:
|
|
if source is None or not source.honor_robots:
|
|
return RobotsDecision(True, "Robotscontrole niet vereist")
|
|
|
|
now = now or timezone.now()
|
|
try:
|
|
origin = _origin_for(url)
|
|
path = _path_for(url)
|
|
except ValueError as exc:
|
|
return RobotsDecision(False, str(exc))
|
|
|
|
cache = _load_cached(origin, now=now)
|
|
stale = _load_stale_cache(origin)
|
|
if cache is None:
|
|
try:
|
|
content, status_code, etag, last_modified = _fetch_robots(origin, client=client)
|
|
cache = _persist_cache(
|
|
origin,
|
|
status_code=status_code,
|
|
content=content,
|
|
etag=etag,
|
|
last_modified=last_modified,
|
|
now=now,
|
|
)
|
|
except UnsafeUrlError as exc:
|
|
return RobotsDecision(False, f"Robotscontrole mislukt: {exc}")
|
|
except httpx.HTTPError as exc:
|
|
if stale is not None:
|
|
return _evaluate_path(
|
|
path,
|
|
user_agent=user_agent or getattr(settings, "FETCHER_USER_AGENT", ""),
|
|
rules=_build_ruleset(stale),
|
|
)
|
|
return RobotsDecision(True, f"Robotscontrole tijdelijk niet beschikbaar: {exc}")
|
|
|
|
if cache.error:
|
|
return RobotsDecision(True, cache.error)
|
|
|
|
rules = _build_ruleset(cache)
|
|
decision = _evaluate_path(
|
|
path, user_agent=user_agent or getattr(settings, "FETCHER_USER_AGENT", ""), rules=rules
|
|
)
|
|
return decision
|