Coverage for src/lilbee/retrieval/query/intent.py: 100%
135 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
1"""Query intent detection: known-item lookups and corpus aggregates.
3Top-k similarity retrieval answers topical questions. Two other question
4shapes reach the same pipe and fail structurally:
6- A known-item lookup ("summarize survey_214.pdf") names the thing it
7 wants; the answer is a document, not a ranking.
8- An aggregate ("how many documents mention the observatory") is a property
9 of the whole corpus; the top 20 of 500k chunks cannot count anything.
11Detection here is deterministic and deliberately conservative: a missed
12route degrades to topical retrieval, which handles the query the way it
13always has, while a false positive would hijack a topical question. Every
14pattern therefore requires an explicit structural cue.
16Language-specific patterns live in ``retrieval.language`` packs; the logic
17here consumes the active pack, so another language is an added pack, not an
18edited parser. Only language-neutral shapes (filenames, quoting, token
19splitting) are defined in this module.
20"""
22from __future__ import annotations
24import re
25from dataclasses import dataclass
26from enum import Enum
27from pathlib import Path
29from lilbee.core.llm_json import first_json_object
30from lilbee.retrieval.language import QueryLanguage, query_language
33class AggregateKind(Enum):
34 """What a count-shaped question is asking to count."""
36 TOTAL_SOURCES = "total_sources"
37 TERM_MENTIONS = "term_mentions"
38 DISTINCT_TYPE = "distinct_type"
39 TYPE_ASSOCIATION = "type_association"
40 UNSUPPORTED = "unsupported"
43@dataclass(frozen=True)
44class AggregateQuery:
45 """A parsed aggregate question.
47 ``noun`` carries the thing being counted for the typed kinds;
48 ``group_noun`` the per-group dimension of an association question. Both
49 are question words, resolved against the extraction schema by the caller
50 (the parser stays schema-free and purely syntactic).
51 """
53 kind: AggregateKind
54 term: str = ""
55 noun: str = ""
56 group_noun: str = ""
59# Filename-shaped tokens: a path-ish word with a known document extension.
60# No spaces: a name containing them arrives quoted and the quote pattern
61# catches it; allowing spaces here would swallow leading sentence words.
62# Language-neutral: filenames look the same in every language.
63_FILENAME_RE = re.compile(
64 r"[\w.-][\w./-]*\.(?:pdf|md|txt|docx?|rst|html?|epub|csv|py|rs|js|ts|java|go)\b",
65 re.IGNORECASE,
66)
68# Quoted names: 'harbor survey 2002' / "harbor survey 2002". A quote only
69# delimits when the pair matches and neither end touches a word from the
70# outside, so contractions and possessives ("what's", "Alice's") never pair
71# into a phantom name; double-quoted names may contain apostrophes.
72_QUOTED_RE = re.compile(r"(?<!\w)\"([^\"]{2,80})\"(?!\w)|(?<!\w)'([^']{2,80})'(?!\w)")
75def document_references(question: str, lang: QueryLanguage | None = None) -> list[str]:
76 """Candidate document identifiers named in *question*, best-first.
78 Filenames beat quoted names beat "document N" references; all are
79 resolved against real source metadata by the caller, so a wrong
80 candidate costs one lookup, not a wrong route.
81 """
82 lang = lang or query_language()
83 candidates: list[str] = []
84 for m in _FILENAME_RE.finditer(question):
85 candidates.append(m.group(0).strip())
86 for m in _QUOTED_RE.finditer(question):
87 quoted = (m.group(1) or m.group(2)).strip()
88 if quoted:
89 candidates.append(quoted)
90 for m in lang.doc_ref_pattern.finditer(question):
91 ref = m.group(1).strip()
92 if ref.lower() not in lang.ref_stopwords:
93 candidates.append(ref)
94 seen: set[str] = set()
95 unique = []
96 for c in candidates:
97 key = c.lower()
98 if key not in seen:
99 seen.add(key)
100 unique.append(c)
101 return unique
104_TOKEN_SPLIT_RE = re.compile(r"[^0-9A-Za-z]+")
107def matches_reference(ref: str, filename: str) -> bool:
108 """Whether *filename* names the document *ref* refers to, token-exactly.
110 Substring search cannot resolve a bare number against zero-padded ids:
111 "482" is a substring of both "...00000482" and "...00010482". Tokens
112 split on non-alphanumerics are compared whole; numeric tokens compare by
113 value so leading zeros don't hide the match, and a longer number sharing
114 a suffix stays a non-match.
115 """
116 ref_token = ref.strip().lower()
117 if ref_token in (filename.lower(), Path(filename).name.lower()):
118 return True
119 stem = Path(filename).stem.lower()
120 for token in _TOKEN_SPLIT_RE.split(stem):
121 if not token:
122 continue
123 if token == ref_token:
124 return True
125 if _same_number(token, ref_token):
126 return True
127 return False
130def _same_number(token: str, ref_token: str) -> bool:
131 """Whether two tokens are the same number ignoring leading zeros.
133 Compares zero-stripped decimal strings rather than calling ``int``:
134 ``str.isdigit()`` is True for Unicode digits like the superscript two,
135 which ``int`` rejects, and the reference pattern matches those.
136 """
137 if not (token.isdecimal() and ref_token.isdecimal()):
138 return False
139 return token.lstrip("0") == ref_token.lstrip("0")
142# Shortest reference the loose arm accepts, counted over its alphanumerics.
143# Below this a reference is a common word as often as a document name.
144def _tokens(text: str) -> list[str]:
145 """Lowercased alphanumeric tokens of *text*."""
146 return [t for t in _TOKEN_SPLIT_RE.split(text.lower()) if t]
149def contains_reference(ref: str, filename: str) -> bool:
150 """Whether *filename*'s stem carries *ref* as a run of whole tokens.
152 A quoted title never token-matches a hyphenated filename, so "harbor survey
153 2010" still names harbor-survey-2010.pdf. Whole tokens keep "we" out of
154 "Web".
155 """
156 ref_tokens = _tokens(ref)
157 if not ref_tokens:
158 return False
159 stem_tokens = _tokens(Path(filename).stem)
160 span = len(ref_tokens)
161 return any(stem_tokens[i : i + span] == ref_tokens for i in range(len(stem_tokens) - span + 1))
164def title_candidates(question: str, lang: QueryLanguage | None = None) -> list[str]:
165 """Document-title candidates from known-item question shapes.
167 "summarize Frankenstein" yields "Frankenstein"; a question with no
168 known-item shape yields nothing, so topical questions that merely
169 mention a title word never reach title resolution.
170 """
171 lang = lang or query_language()
172 candidates = []
173 for pattern in lang.known_item_patterns:
174 m = pattern.match(question)
175 if m:
176 title = m.group(1).strip().strip("\"'")
177 if title:
178 candidates.append(title)
179 return candidates
182def _title_tokens(text: str, lang: QueryLanguage) -> list[str]:
183 """Lowercased comparison tokens with the leading article stripped."""
184 return _tokens(lang.leading_article_pattern.sub("", text.strip()))
187def matches_title(title: str, filename: str, lang: QueryLanguage | None = None) -> bool:
188 """Whether *filename*'s stem is the document *title* names, token-exactly.
190 Leading articles are stripped from both sides so "the prince" matches
191 "The Prince.txt" and "Prince.txt" alike; every remaining token must
192 match, so "the report" never resolves "Annual Report 2020.txt".
193 """
194 lang = lang or query_language()
195 title_tokens = _title_tokens(title, lang)
196 return bool(title_tokens) and title_tokens == _title_tokens(Path(filename).stem, lang)
199def matches_stored_title(title: str, stored: str | None, lang: QueryLanguage | None = None) -> bool:
200 """Whether the stored document title is what *title* names, token-exactly.
202 Covers documents whose ingested title (markdown H1, extraction metadata)
203 differs from their filename, so "summarize Frankenstein Analysis" routes
204 to notes-2024.md.
205 """
206 if not stored:
207 return False
208 lang = lang or query_language()
209 title_tokens = _title_tokens(title, lang)
210 return bool(title_tokens) and title_tokens == _title_tokens(stored, lang)
213def parse_aggregate(question: str, lang: QueryLanguage | None = None) -> AggregateQuery | None:
214 """Parse a count-shaped question, or ``None`` for anything else.
216 Only "how many ..." questions qualify; of those, term-mention and
217 corpus-total forms are answerable against today's schema. The rest
218 (counts over typed records the store does not hold) come back as
219 ``UNSUPPORTED`` so the caller can decline precisely instead of feeding
220 the question to top-k retrieval that structurally cannot count.
221 """
222 lang = lang or query_language()
223 if not lang.how_many_pattern.search(question):
224 return None
225 m = lang.association_pattern.search(question) or lang.per_pattern.search(question)
226 if m:
227 return AggregateQuery(
228 AggregateKind.TYPE_ASSOCIATION,
229 noun=m.group(1).strip(),
230 group_noun=m.group(2).strip(),
231 )
232 m = lang.distinct_pattern.search(question)
233 if m:
234 return AggregateQuery(AggregateKind.DISTINCT_TYPE, noun=m.group(1).strip())
235 m = lang.term_mention_pattern.search(question)
236 if m:
237 term = m.group(1).strip().strip("\"'")
238 # Strip a leading article so 'mention the observatory' counts 'observatory'.
239 term = lang.leading_article_pattern.sub("", term)
240 if term:
241 return AggregateQuery(AggregateKind.TERM_MENTIONS, term=term)
242 if lang.total_pattern.search(question):
243 return AggregateQuery(AggregateKind.TOTAL_SOURCES)
244 return AggregateQuery(AggregateKind.UNSUPPORTED)
247# --- LLM-backed classification (config-gated; see Searcher.route_direct_answer) ---
249# Answer budget for the classification call: one small JSON object.
250INTENT_CLASSIFY_MAX_TOKENS = 96
252# The classifier prompt is intentionally language-agnostic about the QUESTION
253# (the model reads any language); only the label vocabulary is fixed.
254INTENT_CLASSIFY_PROMPT = """Classify this question for a document-search engine.
255Respond with ONLY a JSON object, no other text:
256{{"kind": "...", "term": "", "noun": "", "group_noun": ""}}
258kind must be exactly one of:
259- "topical": an ordinary question answered by reading passages (the default)
260- "total_sources": asks how many documents/files the collection holds
261- "term_mentions": asks how many documents mention or contain a specific \
262term; put that term in "term"
263- "distinct_type": asks how many distinct entities of some type exist; put \
264the type noun in "noun"
265- "type_association": asks how many X each Y has; put X in "noun" and Y in \
266"group_noun"
268When unsure, use "topical".
270Question: {question}
271"""
274_LLM_KINDS = {
275 "total_sources": AggregateKind.TOTAL_SOURCES,
276 "term_mentions": AggregateKind.TERM_MENTIONS,
277 "distinct_type": AggregateKind.DISTINCT_TYPE,
278 "type_association": AggregateKind.TYPE_ASSOCIATION,
279}
282def parse_llm_aggregate(text: str) -> AggregateQuery | None:
283 """Map the classifier's reply to a route, or ``None`` for no route.
285 Conservative on every axis: anything malformed, unknown, "topical", or
286 missing a required field yields ``None``, which sends the question to
287 ordinary retrieval -- the same harmless degrade as a deterministic miss.
288 ``UNSUPPORTED`` is never produced here; declining is reserved for the
289 deterministic layer, whose patterns prove the question is count-shaped.
290 """
291 data = first_json_object(text)
292 if data is None:
293 return None
294 raw_kind = data.get("kind", "")
295 # A non-string kind (list, dict) is malformed, not a crash: an unhashable
296 # value would raise TypeError inside dict.get.
297 kind = _LLM_KINDS.get(raw_kind) if isinstance(raw_kind, str) else None
298 if kind is None:
299 return None
300 term = str(data.get("term", "") or "").strip()
301 noun = str(data.get("noun", "") or "").strip()
302 group_noun = str(data.get("group_noun", "") or "").strip()
303 required_ok = {
304 AggregateKind.TOTAL_SOURCES: True,
305 AggregateKind.TERM_MENTIONS: bool(term),
306 AggregateKind.DISTINCT_TYPE: bool(noun),
307 AggregateKind.TYPE_ASSOCIATION: bool(noun and group_noun),
308 }[kind]
309 if not required_ok:
310 return None
311 return AggregateQuery(kind, term=term, noun=noun, group_noun=group_noun)
314# Fewer tokens than this cannot stand alone ("Why?", "how much?"). A script the
315# ASCII splitter yields no tokens for lands here too and keeps the rewrite.
316_STANDALONE_MIN_TOKENS = 3
319def refers_to_history(question: str, lang: QueryLanguage | None = None) -> bool:
320 """Whether *question* needs earlier turns to be searchable on its own."""
321 lang = lang or query_language()
322 tokens = [t for t in _TOKEN_SPLIT_RE.split(question) if t]
323 if len(tokens) < _STANDALONE_MIN_TOKENS:
324 return True
325 return lang.follow_up_pattern.search(question) is not None