Coverage for src/lilbee/retrieval/query/tokenize.py: 100%
21 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
1"""Token utilities for the RAG query pipeline."""
3from __future__ import annotations
5import math
6import re
8_MIN_TOKEN_LEN = 1
9_TOKEN_SPLIT_RE = re.compile(r"\W+")
11# A single-character match counts at most this much: enough to prefer the
12# chunk naming the subject (C, R, W-2), too little for a stray contraction
13# splinter to outrank a distinctive longer term.
14_SINGLE_CHAR_MAX_WEIGHT = 1.0
17def _tokenize(text: str) -> list[str]:
18 """Lowercase alphanumeric tokens, split on any non-alnum run."""
19 return [word for word in _TOKEN_SPLIT_RE.split(text.lower()) if len(word) >= _MIN_TOKEN_LEN]
22def _idf_weights(
23 question_terms: set[str],
24 chunk_tokens: list[set[str]],
25) -> dict[str, float]:
26 """Inverse Document Frequency weight per query term over the candidate chunks.
28 Classical IDF per Spärck Jones (1972), "A Statistical Interpretation
29 of Term Specificity and Its Application in Retrieval", Journal of
30 Documentation 28:11-21. Terms that appear in every chunk collapse to
31 zero weight, so corpus-specific stopwords are filtered automatically.
32 Single-character terms are capped, since a rare one-letter fragment
33 would otherwise outrank a distinctive longer term.
34 """
35 n = len(chunk_tokens)
36 df: dict[str, int] = {}
37 for tokens in chunk_tokens:
38 for term in tokens & question_terms:
39 df[term] = df.get(term, 0) + 1
40 weights: dict[str, float] = {}
41 for term in question_terms:
42 weight = max(0.0, math.log(n / (1 + df.get(term, 0))))
43 if len(term) <= _MIN_TOKEN_LEN:
44 weight = min(weight, _SINGLE_CHAR_MAX_WEIGHT)
45 weights[term] = weight
46 return weights