Coverage for src/lilbee/retrieval/query/tokenize.py: 100%

21 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""Token utilities for the RAG query pipeline.""" 

2 

3from __future__ import annotations 

4 

5import math 

6import re 

7 

8_MIN_TOKEN_LEN = 1 

9_TOKEN_SPLIT_RE = re.compile(r"\W+") 

10 

11# A single-character match counts at most this much: enough to prefer the 

12# chunk naming the subject (C, R, W-2), too little for a stray contraction 

13# splinter to outrank a distinctive longer term. 

14_SINGLE_CHAR_MAX_WEIGHT = 1.0 

15 

16 

17def _tokenize(text: str) -> list[str]: 

18 """Lowercase alphanumeric tokens, split on any non-alnum run.""" 

19 return [word for word in _TOKEN_SPLIT_RE.split(text.lower()) if len(word) >= _MIN_TOKEN_LEN] 

20 

21 

22def _idf_weights( 

23 question_terms: set[str], 

24 chunk_tokens: list[set[str]], 

25) -> dict[str, float]: 

26 """Inverse Document Frequency weight per query term over the candidate chunks. 

27 

28 Classical IDF per Spärck Jones (1972), "A Statistical Interpretation 

29 of Term Specificity and Its Application in Retrieval", Journal of 

30 Documentation 28:11-21. Terms that appear in every chunk collapse to 

31 zero weight, so corpus-specific stopwords are filtered automatically. 

32 Single-character terms are capped, since a rare one-letter fragment 

33 would otherwise outrank a distinctive longer term. 

34 """ 

35 n = len(chunk_tokens) 

36 df: dict[str, int] = {} 

37 for tokens in chunk_tokens: 

38 for term in tokens & question_terms: 

39 df[term] = df.get(term, 0) + 1 

40 weights: dict[str, float] = {} 

41 for term in question_terms: 

42 weight = max(0.0, math.log(n / (1 + df.get(term, 0)))) 

43 if len(term) <= _MIN_TOKEN_LEN: 

44 weight = min(weight, _SINGLE_CHAR_MAX_WEIGHT) 

45 weights[term] = weight 

46 return weights