Coverage for src/lilbee/retrieval/query/intent.py: 100%

135 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""Query intent detection: known-item lookups and corpus aggregates. 

2 

3Top-k similarity retrieval answers topical questions. Two other question 

4shapes reach the same pipe and fail structurally: 

5 

6- A known-item lookup ("summarize survey_214.pdf") names the thing it 

7 wants; the answer is a document, not a ranking. 

8- An aggregate ("how many documents mention the observatory") is a property 

9 of the whole corpus; the top 20 of 500k chunks cannot count anything. 

10 

11Detection here is deterministic and deliberately conservative: a missed 

12route degrades to topical retrieval, which handles the query the way it 

13always has, while a false positive would hijack a topical question. Every 

14pattern therefore requires an explicit structural cue. 

15 

16Language-specific patterns live in ``retrieval.language`` packs; the logic 

17here consumes the active pack, so another language is an added pack, not an 

18edited parser. Only language-neutral shapes (filenames, quoting, token 

19splitting) are defined in this module. 

20""" 

21 

22from __future__ import annotations 

23 

24import re 

25from dataclasses import dataclass 

26from enum import Enum 

27from pathlib import Path 

28 

29from lilbee.core.llm_json import first_json_object 

30from lilbee.retrieval.language import QueryLanguage, query_language 

31 

32 

33class AggregateKind(Enum): 

34 """What a count-shaped question is asking to count.""" 

35 

36 TOTAL_SOURCES = "total_sources" 

37 TERM_MENTIONS = "term_mentions" 

38 DISTINCT_TYPE = "distinct_type" 

39 TYPE_ASSOCIATION = "type_association" 

40 UNSUPPORTED = "unsupported" 

41 

42 

43@dataclass(frozen=True) 

44class AggregateQuery: 

45 """A parsed aggregate question. 

46 

47 ``noun`` carries the thing being counted for the typed kinds; 

48 ``group_noun`` the per-group dimension of an association question. Both 

49 are question words, resolved against the extraction schema by the caller 

50 (the parser stays schema-free and purely syntactic). 

51 """ 

52 

53 kind: AggregateKind 

54 term: str = "" 

55 noun: str = "" 

56 group_noun: str = "" 

57 

58 

59# Filename-shaped tokens: a path-ish word with a known document extension. 

60# No spaces: a name containing them arrives quoted and the quote pattern 

61# catches it; allowing spaces here would swallow leading sentence words. 

62# Language-neutral: filenames look the same in every language. 

63_FILENAME_RE = re.compile( 

64 r"[\w.-][\w./-]*\.(?:pdf|md|txt|docx?|rst|html?|epub|csv|py|rs|js|ts|java|go)\b", 

65 re.IGNORECASE, 

66) 

67 

68# Quoted names: 'harbor survey 2002' / "harbor survey 2002". A quote only 

69# delimits when the pair matches and neither end touches a word from the 

70# outside, so contractions and possessives ("what's", "Alice's") never pair 

71# into a phantom name; double-quoted names may contain apostrophes. 

72_QUOTED_RE = re.compile(r"(?<!\w)\"([^\"]{2,80})\"(?!\w)|(?<!\w)'([^']{2,80})'(?!\w)") 

73 

74 

75def document_references(question: str, lang: QueryLanguage | None = None) -> list[str]: 

76 """Candidate document identifiers named in *question*, best-first. 

77 

78 Filenames beat quoted names beat "document N" references; all are 

79 resolved against real source metadata by the caller, so a wrong 

80 candidate costs one lookup, not a wrong route. 

81 """ 

82 lang = lang or query_language() 

83 candidates: list[str] = [] 

84 for m in _FILENAME_RE.finditer(question): 

85 candidates.append(m.group(0).strip()) 

86 for m in _QUOTED_RE.finditer(question): 

87 quoted = (m.group(1) or m.group(2)).strip() 

88 if quoted: 

89 candidates.append(quoted) 

90 for m in lang.doc_ref_pattern.finditer(question): 

91 ref = m.group(1).strip() 

92 if ref.lower() not in lang.ref_stopwords: 

93 candidates.append(ref) 

94 seen: set[str] = set() 

95 unique = [] 

96 for c in candidates: 

97 key = c.lower() 

98 if key not in seen: 

99 seen.add(key) 

100 unique.append(c) 

101 return unique 

102 

103 

104_TOKEN_SPLIT_RE = re.compile(r"[^0-9A-Za-z]+") 

105 

106 

107def matches_reference(ref: str, filename: str) -> bool: 

108 """Whether *filename* names the document *ref* refers to, token-exactly. 

109 

110 Substring search cannot resolve a bare number against zero-padded ids: 

111 "482" is a substring of both "...00000482" and "...00010482". Tokens 

112 split on non-alphanumerics are compared whole; numeric tokens compare by 

113 value so leading zeros don't hide the match, and a longer number sharing 

114 a suffix stays a non-match. 

115 """ 

116 ref_token = ref.strip().lower() 

117 if ref_token in (filename.lower(), Path(filename).name.lower()): 

118 return True 

119 stem = Path(filename).stem.lower() 

120 for token in _TOKEN_SPLIT_RE.split(stem): 

121 if not token: 

122 continue 

123 if token == ref_token: 

124 return True 

125 if _same_number(token, ref_token): 

126 return True 

127 return False 

128 

129 

130def _same_number(token: str, ref_token: str) -> bool: 

131 """Whether two tokens are the same number ignoring leading zeros. 

132 

133 Compares zero-stripped decimal strings rather than calling ``int``: 

134 ``str.isdigit()`` is True for Unicode digits like the superscript two, 

135 which ``int`` rejects, and the reference pattern matches those. 

136 """ 

137 if not (token.isdecimal() and ref_token.isdecimal()): 

138 return False 

139 return token.lstrip("0") == ref_token.lstrip("0") 

140 

141 

142# Shortest reference the loose arm accepts, counted over its alphanumerics. 

143# Below this a reference is a common word as often as a document name. 

144def _tokens(text: str) -> list[str]: 

145 """Lowercased alphanumeric tokens of *text*.""" 

146 return [t for t in _TOKEN_SPLIT_RE.split(text.lower()) if t] 

147 

148 

149def contains_reference(ref: str, filename: str) -> bool: 

150 """Whether *filename*'s stem carries *ref* as a run of whole tokens. 

151 

152 A quoted title never token-matches a hyphenated filename, so "harbor survey 

153 2010" still names harbor-survey-2010.pdf. Whole tokens keep "we" out of 

154 "Web". 

155 """ 

156 ref_tokens = _tokens(ref) 

157 if not ref_tokens: 

158 return False 

159 stem_tokens = _tokens(Path(filename).stem) 

160 span = len(ref_tokens) 

161 return any(stem_tokens[i : i + span] == ref_tokens for i in range(len(stem_tokens) - span + 1)) 

162 

163 

164def title_candidates(question: str, lang: QueryLanguage | None = None) -> list[str]: 

165 """Document-title candidates from known-item question shapes. 

166 

167 "summarize Frankenstein" yields "Frankenstein"; a question with no 

168 known-item shape yields nothing, so topical questions that merely 

169 mention a title word never reach title resolution. 

170 """ 

171 lang = lang or query_language() 

172 candidates = [] 

173 for pattern in lang.known_item_patterns: 

174 m = pattern.match(question) 

175 if m: 

176 title = m.group(1).strip().strip("\"'") 

177 if title: 

178 candidates.append(title) 

179 return candidates 

180 

181 

182def _title_tokens(text: str, lang: QueryLanguage) -> list[str]: 

183 """Lowercased comparison tokens with the leading article stripped.""" 

184 return _tokens(lang.leading_article_pattern.sub("", text.strip())) 

185 

186 

187def matches_title(title: str, filename: str, lang: QueryLanguage | None = None) -> bool: 

188 """Whether *filename*'s stem is the document *title* names, token-exactly. 

189 

190 Leading articles are stripped from both sides so "the prince" matches 

191 "The Prince.txt" and "Prince.txt" alike; every remaining token must 

192 match, so "the report" never resolves "Annual Report 2020.txt". 

193 """ 

194 lang = lang or query_language() 

195 title_tokens = _title_tokens(title, lang) 

196 return bool(title_tokens) and title_tokens == _title_tokens(Path(filename).stem, lang) 

197 

198 

199def matches_stored_title(title: str, stored: str | None, lang: QueryLanguage | None = None) -> bool: 

200 """Whether the stored document title is what *title* names, token-exactly. 

201 

202 Covers documents whose ingested title (markdown H1, extraction metadata) 

203 differs from their filename, so "summarize Frankenstein Analysis" routes 

204 to notes-2024.md. 

205 """ 

206 if not stored: 

207 return False 

208 lang = lang or query_language() 

209 title_tokens = _title_tokens(title, lang) 

210 return bool(title_tokens) and title_tokens == _title_tokens(stored, lang) 

211 

212 

213def parse_aggregate(question: str, lang: QueryLanguage | None = None) -> AggregateQuery | None: 

214 """Parse a count-shaped question, or ``None`` for anything else. 

215 

216 Only "how many ..." questions qualify; of those, term-mention and 

217 corpus-total forms are answerable against today's schema. The rest 

218 (counts over typed records the store does not hold) come back as 

219 ``UNSUPPORTED`` so the caller can decline precisely instead of feeding 

220 the question to top-k retrieval that structurally cannot count. 

221 """ 

222 lang = lang or query_language() 

223 if not lang.how_many_pattern.search(question): 

224 return None 

225 m = lang.association_pattern.search(question) or lang.per_pattern.search(question) 

226 if m: 

227 return AggregateQuery( 

228 AggregateKind.TYPE_ASSOCIATION, 

229 noun=m.group(1).strip(), 

230 group_noun=m.group(2).strip(), 

231 ) 

232 m = lang.distinct_pattern.search(question) 

233 if m: 

234 return AggregateQuery(AggregateKind.DISTINCT_TYPE, noun=m.group(1).strip()) 

235 m = lang.term_mention_pattern.search(question) 

236 if m: 

237 term = m.group(1).strip().strip("\"'") 

238 # Strip a leading article so 'mention the observatory' counts 'observatory'. 

239 term = lang.leading_article_pattern.sub("", term) 

240 if term: 

241 return AggregateQuery(AggregateKind.TERM_MENTIONS, term=term) 

242 if lang.total_pattern.search(question): 

243 return AggregateQuery(AggregateKind.TOTAL_SOURCES) 

244 return AggregateQuery(AggregateKind.UNSUPPORTED) 

245 

246 

247# --- LLM-backed classification (config-gated; see Searcher.route_direct_answer) --- 

248 

249# Answer budget for the classification call: one small JSON object. 

250INTENT_CLASSIFY_MAX_TOKENS = 96 

251 

252# The classifier prompt is intentionally language-agnostic about the QUESTION 

253# (the model reads any language); only the label vocabulary is fixed. 

254INTENT_CLASSIFY_PROMPT = """Classify this question for a document-search engine. 

255Respond with ONLY a JSON object, no other text: 

256{{"kind": "...", "term": "", "noun": "", "group_noun": ""}} 

257 

258kind must be exactly one of: 

259- "topical": an ordinary question answered by reading passages (the default) 

260- "total_sources": asks how many documents/files the collection holds 

261- "term_mentions": asks how many documents mention or contain a specific \ 

262term; put that term in "term" 

263- "distinct_type": asks how many distinct entities of some type exist; put \ 

264the type noun in "noun" 

265- "type_association": asks how many X each Y has; put X in "noun" and Y in \ 

266"group_noun" 

267 

268When unsure, use "topical". 

269 

270Question: {question} 

271""" 

272 

273 

274_LLM_KINDS = { 

275 "total_sources": AggregateKind.TOTAL_SOURCES, 

276 "term_mentions": AggregateKind.TERM_MENTIONS, 

277 "distinct_type": AggregateKind.DISTINCT_TYPE, 

278 "type_association": AggregateKind.TYPE_ASSOCIATION, 

279} 

280 

281 

282def parse_llm_aggregate(text: str) -> AggregateQuery | None: 

283 """Map the classifier's reply to a route, or ``None`` for no route. 

284 

285 Conservative on every axis: anything malformed, unknown, "topical", or 

286 missing a required field yields ``None``, which sends the question to 

287 ordinary retrieval -- the same harmless degrade as a deterministic miss. 

288 ``UNSUPPORTED`` is never produced here; declining is reserved for the 

289 deterministic layer, whose patterns prove the question is count-shaped. 

290 """ 

291 data = first_json_object(text) 

292 if data is None: 

293 return None 

294 raw_kind = data.get("kind", "") 

295 # A non-string kind (list, dict) is malformed, not a crash: an unhashable 

296 # value would raise TypeError inside dict.get. 

297 kind = _LLM_KINDS.get(raw_kind) if isinstance(raw_kind, str) else None 

298 if kind is None: 

299 return None 

300 term = str(data.get("term", "") or "").strip() 

301 noun = str(data.get("noun", "") or "").strip() 

302 group_noun = str(data.get("group_noun", "") or "").strip() 

303 required_ok = { 

304 AggregateKind.TOTAL_SOURCES: True, 

305 AggregateKind.TERM_MENTIONS: bool(term), 

306 AggregateKind.DISTINCT_TYPE: bool(noun), 

307 AggregateKind.TYPE_ASSOCIATION: bool(noun and group_noun), 

308 }[kind] 

309 if not required_ok: 

310 return None 

311 return AggregateQuery(kind, term=term, noun=noun, group_noun=group_noun) 

312 

313 

314# Fewer tokens than this cannot stand alone ("Why?", "how much?"). A script the 

315# ASCII splitter yields no tokens for lands here too and keeps the rewrite. 

316_STANDALONE_MIN_TOKENS = 3 

317 

318 

319def refers_to_history(question: str, lang: QueryLanguage | None = None) -> bool: 

320 """Whether *question* needs earlier turns to be searchable on its own.""" 

321 lang = lang or query_language() 

322 tokens = [t for t in _TOKEN_SPLIT_RE.split(question) if t] 

323 if len(tokens) < _STANDALONE_MIN_TOKENS: 

324 return True 

325 return lang.follow_up_pattern.search(question) is not None