- Fix the demo knowledge provider's tokenizer: a plain [a-z0-9]+ regex silently dropped accented characters, splitting French words like "véhicule" into "v" + "hicule" and mangling retrieval for nearly every French query. Now matches the Latin-1 accented range too. - Reweight section scoring so the body match (the actual substance of a section) outranks a heading/title match (a shallow structural hint) rather than the reverse -- confirmed via the brief's exact validation question that the old weighting misranked the damage procedure behind a topically-adjacent document in all three languages (nl-BE: a checkout section; en-GB/fr-BE: the return procedure), purely because a generic word like "vehicle"/"voertuig" happened to sit in a heading/title. - Remove leftover "MobilityOps" and "PoC" mentions from 5 English and 4 NL/FR procedure documents -- knowledge-base prose is visible UI content and was missed by the earlier rebrand. - Add regression tests: the brief's exact NL/EN/FR damage question must ground on the damage procedure as the *primary* source (not just appear in the top 3), and no procedure file may contain "MobilityOps" or "PoC". 151 backend tests, Ruff, mypy green. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
304 lines
12 KiB
Python
304 lines
12 KiB
Python
from __future__ import annotations
|
|
|
|
import math
|
|
import re
|
|
from dataclasses import dataclass, field
|
|
from pathlib import Path
|
|
|
|
from app.core.config import get_settings
|
|
from app.services.knowledge import GroundedAnswer, KnowledgeHealth, SourceCard
|
|
|
|
SUPPORTED_LANGUAGES = ("nl-BE", "en-GB", "fr-BE")
|
|
DEFAULT_LANGUAGE = "en-GB"
|
|
|
|
STOPWORDS_BY_LANGUAGE: dict[str, set[str]] = {
|
|
"en-GB": {
|
|
"a", "an", "the", "is", "are", "was", "were", "be", "been", "being",
|
|
"to", "of", "in", "on", "at", "for", "and", "or", "but", "if", "then",
|
|
"do", "does", "did", "must", "may", "can", "could", "should", "would",
|
|
"i", "you", "it", "we", "they", "my", "your", "what", "when", "how",
|
|
"with", "without", "this", "that", "these", "those", "not", "no",
|
|
},
|
|
"nl-BE": {
|
|
"een", "de", "het", "is", "zijn", "was", "waren", "worden", "wordt",
|
|
"van", "in", "op", "voor", "en", "of", "maar", "als", "dan",
|
|
"moet", "mag", "kan", "kunnen", "zou", "zouden",
|
|
"ik", "jij", "u", "we", "wij", "zij", "mijn", "jouw", "wat", "wanneer", "hoe",
|
|
"met", "zonder", "dit", "dat", "deze", "die", "niet", "geen",
|
|
},
|
|
"fr-BE": {
|
|
"un", "une", "le", "la", "les", "des", "est", "sont", "était", "être",
|
|
"de", "du", "en", "sur", "pour", "et", "ou", "mais", "si", "alors",
|
|
"doit", "peut", "peuvent", "pourrait", "devrait",
|
|
"je", "tu", "vous", "il", "elle", "nous", "ils", "mon", "votre", "quoi", "quand", "comment",
|
|
"avec", "sans", "ce", "cette", "ces", "cela", "pas", "non",
|
|
},
|
|
}
|
|
|
|
# Includes the Latin-1 accented-letter range (à-ö, ø-ÿ) so French/Dutch words with
|
|
# diacritics (véhicule, réservation, geëscaleerd) tokenize as one word instead of
|
|
# splitting apart at the accented character -- a plain [a-z0-9]+ pattern silently
|
|
# drops every accent and fragments the word either side of it.
|
|
_WORD_RE = re.compile(r"[a-zà-öø-ÿ0-9]+")
|
|
|
|
|
|
def _stem(word: str) -> str:
|
|
# Deterministic, intentionally crude suffix stripping — good enough to match "returns"
|
|
# with "return" or "damaged" with "damage" without pulling in a stemming dependency.
|
|
for suffix in ("ing", "edly", "ed", "es", "s"):
|
|
if len(word) > len(suffix) + 2 and word.endswith(suffix):
|
|
return word[: -len(suffix)]
|
|
return word
|
|
|
|
|
|
def _tokenize(text: str, language: str) -> set[str]:
|
|
stopwords = STOPWORDS_BY_LANGUAGE.get(language, STOPWORDS_BY_LANGUAGE[DEFAULT_LANGUAGE])
|
|
words = _WORD_RE.findall(text.lower())
|
|
return {_stem(w) for w in words if w not in stopwords and len(w) > 2}
|
|
|
|
|
|
@dataclass
|
|
class Document:
|
|
document_id: str
|
|
title: str
|
|
version: str
|
|
title_tokens: set[str] = field(default_factory=set)
|
|
|
|
|
|
@dataclass
|
|
class ScoredSection:
|
|
document: Document
|
|
heading: str
|
|
text: str
|
|
heading_tokens: set[str]
|
|
body_tokens: set[str]
|
|
|
|
|
|
def _parse_frontmatter(raw: str) -> tuple[dict[str, str], str]:
|
|
if not raw.startswith("---"):
|
|
return {}, raw
|
|
end = raw.find("\n---", 3)
|
|
if end == -1:
|
|
return {}, raw
|
|
block = raw[3:end].strip()
|
|
body = raw[end + 4 :].lstrip("\n")
|
|
meta: dict[str, str] = {}
|
|
for line in block.splitlines():
|
|
if ":" not in line:
|
|
continue
|
|
key, _, value = line.partition(":")
|
|
meta[key.strip()] = value.strip().strip('"')
|
|
return meta, body
|
|
|
|
|
|
def _split_sections(body: str) -> list[tuple[str, str]]:
|
|
sections: list[tuple[str, str]] = []
|
|
current_heading = "Overview"
|
|
current_lines: list[str] = []
|
|
for line in body.splitlines():
|
|
if line.startswith("## "):
|
|
if current_lines:
|
|
sections.append((current_heading, "\n".join(current_lines).strip()))
|
|
current_heading = line[3:].strip()
|
|
current_lines = []
|
|
elif line.startswith("# "):
|
|
continue
|
|
else:
|
|
current_lines.append(line)
|
|
if current_lines:
|
|
sections.append((current_heading, "\n".join(current_lines).strip()))
|
|
return sections
|
|
|
|
|
|
def _load_sections(procedures_dir: Path, language: str) -> list[ScoredSection]:
|
|
sections: list[ScoredSection] = []
|
|
for path in sorted(procedures_dir.glob("*.md")):
|
|
raw = path.read_text(encoding="utf-8")
|
|
meta, body = _parse_frontmatter(raw)
|
|
title = meta.get("title", path.stem)
|
|
doc = Document(
|
|
document_id=meta.get("document_id", path.stem),
|
|
title=title,
|
|
version=meta.get("version", "1.0"),
|
|
title_tokens=_tokenize(title, language),
|
|
)
|
|
for heading, text in _split_sections(body):
|
|
sections.append(
|
|
ScoredSection(
|
|
document=doc,
|
|
heading=heading,
|
|
text=text,
|
|
heading_tokens=_tokenize(heading, language),
|
|
body_tokens=_tokenize(text, language),
|
|
)
|
|
)
|
|
return sections
|
|
|
|
|
|
_NO_MATCH_TEXT = {
|
|
"en-GB": "No matching procedure was found for this question.",
|
|
"nl-BE": "Er werd geen passende procedure gevonden voor deze vraag.",
|
|
"fr-BE": "Aucune procédure correspondante n'a été trouvée pour cette question.",
|
|
}
|
|
_LOW_CONFIDENCE_TEXT = {
|
|
"en-GB": (
|
|
"The available procedures do not clearly answer this question. "
|
|
"The closest matches are included below for review."
|
|
),
|
|
"nl-BE": (
|
|
"De beschikbare procedures beantwoorden deze vraag niet duidelijk. "
|
|
"De dichtstbijzijnde overeenkomsten staan hieronder ter beoordeling."
|
|
),
|
|
"fr-BE": (
|
|
"Les procédures disponibles ne répondent pas clairement à cette question. "
|
|
"Les correspondances les plus proches sont indiquées ci-dessous pour examen."
|
|
),
|
|
}
|
|
_LEAD_ANSWER_TEMPLATE = {
|
|
"en-GB": 'Per "{title}" (v{version}), section "{heading}": {excerpt}',
|
|
"nl-BE": 'Volgens "{title}" (v{version}), sectie "{heading}": {excerpt}',
|
|
"fr-BE": 'Selon « {title} » (v{version}), section « {heading} » : {excerpt}',
|
|
}
|
|
|
|
|
|
class DemoKnowledgeProvider:
|
|
"""Deterministic extractive retrieval over the local procedure Markdown files.
|
|
|
|
Not a generative model: it scores sections with TF-IDF-weighted keyword overlap
|
|
(downweighting terms common across the whole corpus, like "vehicle", in favor of
|
|
distinctive ones, like "damage") and returns real excerpts, never invented text.
|
|
|
|
Each supported UI language has its own translated procedure corpus under
|
|
knowledge/procedures/<language>/ -- retrieval searches only within the requested
|
|
language's corpus so citations always link to a same-language document.
|
|
"""
|
|
|
|
name = "demo"
|
|
|
|
def __init__(self) -> None:
|
|
settings = get_settings()
|
|
self._settings = settings
|
|
base_dir = Path(settings.knowledge_dir)
|
|
self._sections_by_language: dict[str, list[ScoredSection]] = {}
|
|
self._idf_by_language: dict[str, dict[str, float]] = {}
|
|
self._document_count_by_language: dict[str, int] = {}
|
|
for language in SUPPORTED_LANGUAGES:
|
|
lang_dir = base_dir / language
|
|
sections = _load_sections(lang_dir, language) if lang_dir.is_dir() else []
|
|
self._sections_by_language[language] = sections
|
|
self._idf_by_language[language] = self._build_idf(sections)
|
|
self._document_count_by_language[language] = len(
|
|
{s.document.document_id for s in sections}
|
|
)
|
|
|
|
@staticmethod
|
|
def _build_idf(sections: list[ScoredSection]) -> dict[str, float]:
|
|
n = len(sections) or 1
|
|
doc_freq: dict[str, int] = {}
|
|
for section in sections:
|
|
doc = section.document
|
|
all_tokens = doc.title_tokens | section.heading_tokens | section.body_tokens
|
|
for token in all_tokens:
|
|
doc_freq[token] = doc_freq.get(token, 0) + 1
|
|
return {token: math.log((n + 1) / (df + 1)) + 1 for token, df in doc_freq.items()}
|
|
|
|
def _normalize_language(self, language: str | None) -> str:
|
|
if language in SUPPORTED_LANGUAGES:
|
|
return language
|
|
return DEFAULT_LANGUAGE
|
|
|
|
def health(self, language: str = DEFAULT_LANGUAGE) -> KnowledgeHealth:
|
|
language = self._normalize_language(language)
|
|
return KnowledgeHealth(
|
|
provider=self.name,
|
|
available=True,
|
|
detail="Deterministic keyword-matching demo provider; no external service.",
|
|
tenant=self._settings.ragcore_tenant,
|
|
workspace=self._settings.ragcore_workspace,
|
|
collection=self._settings.ragcore_collection,
|
|
document_count=self._document_count_by_language[language],
|
|
)
|
|
|
|
def _score(
|
|
self, query_tokens: set[str], section: ScoredSection, idf: dict[str, float]
|
|
) -> float:
|
|
# The section body is the strongest relevance signal -- it's the actual
|
|
# substance a heading or title can only hint at -- so a body match is weighted
|
|
# *above* heading/title matches, not below them. The previous 3x/2x/1x
|
|
# (heading/title/body) ordering let a single generic word in a heading (e.g.
|
|
# "vehicle", present in nearly every section) or a document's own title
|
|
# outrank a section whose body genuinely covers multiple, more distinctive
|
|
# query terms -- confirmed to misrank the brief's exact validation question in
|
|
# every one of the three languages (see docs/fleet-ops-correction/
|
|
# current-gap-audit.md and i18n-inventory.md): nl-BE picked a checkout section
|
|
# over the damage procedure, en-GB and fr-BE picked the return procedure over
|
|
# the damage procedure, purely from heading/title overlap on common words.
|
|
score = 0.0
|
|
for token in query_tokens:
|
|
token_idf = idf.get(token, 0.0)
|
|
if token_idf == 0.0:
|
|
continue
|
|
if token in section.body_tokens:
|
|
score += 3 * token_idf
|
|
elif token in section.heading_tokens:
|
|
score += 2 * token_idf
|
|
elif token in section.document.title_tokens:
|
|
score += 1.5 * token_idf
|
|
return score
|
|
|
|
def ask(
|
|
self, question: str, correlation_id: str, language: str = DEFAULT_LANGUAGE
|
|
) -> GroundedAnswer:
|
|
language = self._normalize_language(language)
|
|
sections = self._sections_by_language[language]
|
|
idf = self._idf_by_language[language]
|
|
query_tokens = _tokenize(question, language)
|
|
scored = [(self._score(query_tokens, section, idf), section) for section in sections]
|
|
scored = [(score, section) for score, section in scored if score > 0]
|
|
scored.sort(key=lambda item: item[0], reverse=True)
|
|
top = scored[:3]
|
|
|
|
if not top:
|
|
return GroundedAnswer(
|
|
answer=_NO_MATCH_TEXT[language],
|
|
evidence_state="insufficient",
|
|
sources=[],
|
|
provider=self.name,
|
|
correlation_id=correlation_id,
|
|
)
|
|
|
|
sources = [
|
|
SourceCard(
|
|
document_id=section.document.document_id,
|
|
title=section.document.title,
|
|
version=section.document.version,
|
|
section=section.heading,
|
|
excerpt=(section.text[:400] + "…") if len(section.text) > 400 else section.text,
|
|
)
|
|
for _, section in top
|
|
]
|
|
|
|
if top[0][0] < 3:
|
|
return GroundedAnswer(
|
|
answer=_LOW_CONFIDENCE_TEXT[language],
|
|
evidence_state="insufficient",
|
|
sources=sources,
|
|
provider=self.name,
|
|
correlation_id=correlation_id,
|
|
)
|
|
|
|
lead_section = top[0][1]
|
|
answer = _LEAD_ANSWER_TEMPLATE[language].format(
|
|
title=lead_section.document.title,
|
|
version=lead_section.document.version,
|
|
heading=lead_section.heading,
|
|
excerpt=lead_section.text.splitlines()[0][:300],
|
|
)
|
|
return GroundedAnswer(
|
|
answer=answer,
|
|
evidence_state="grounded",
|
|
sources=sources,
|
|
provider=self.name,
|
|
correlation_id=correlation_id,
|
|
)
|