Coverage for src/lilbee/data/store/types.py: 100%

188 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""Public dataclasses, TypedDicts, enums, and constants for the store package.""" 

2 

3from __future__ import annotations 

4 

5from dataclasses import dataclass 

6from datetime import timedelta 

7from enum import StrEnum 

8from typing import NamedTuple, NotRequired, TypedDict 

9 

10from pydantic import BaseModel, ConfigDict, Field, field_validator 

11 

12# How often readers re-check the manifest for new versions from other processes. 

13# Zero means strong consistency (every read checks); higher values reduce disk I/O 

14# on slow media (HDD) at the cost of serving slightly stale data. 

15READ_CONSISTENCY_INTERVAL = timedelta(seconds=5) 

16 

17 

18@dataclass 

19class ConceptRecords: 

20 """Rows for the three concept tables, built from one or more files' chunks.""" 

21 

22 nodes: list[dict] 

23 edges: list[dict] 

24 chunk_concepts: list[dict] 

25 

26 @classmethod 

27 def merged(cls, batches: list[ConceptRecords]) -> ConceptRecords: 

28 """Concatenate several record sets into one batched write unit.""" 

29 return cls( 

30 nodes=[row for batch in batches for row in batch.nodes], 

31 edges=[row for batch in batches for row in batch.edges], 

32 chunk_concepts=[row for batch in batches for row in batch.chunk_concepts], 

33 ) 

34 

35 

36class SourceType(StrEnum): 

37 """Values for the ``_sources.source_type`` column. 

38 

39 ``DOCUMENT`` mirrors a file under ``documents/`` and is managed by the 

40 file-driven sync. ``IMPORTED`` is detached: it came from ``lilbee import`` 

41 and has no backing file, so sync must not treat it as a missing document. 

42 """ 

43 

44 DOCUMENT = "document" 

45 IMPORTED = "imported" 

46 

47 

48class SourceMeta(NamedTuple): 

49 """Document-level metadata captured at extraction time. 

50 

51 ``title`` is always derivable (extraction metadata or the cleaned filename 

52 stem); ``authors`` and ``created_at`` are only present when the extractor 

53 reports them. Empty strings persist as NULL so old and new rows read alike. 

54 """ 

55 

56 title: str = "" 

57 authors: str = "" 

58 created_at: str = "" 

59 

60 

61class ChunkWrite(NamedTuple): 

62 """One document's chunks plus its source-table update, for a batched write. 

63 

64 ``Store.write_chunks_batch`` folds many of these into a single locked 

65 transaction so bulk ingest doesn't pay a write-lock acquisition per document. 

66 ``page_texts`` rows land in the same transaction, after the cleanup delete 

67 and before the source row. ``source_type`` lets the detached import path 

68 reuse the same atomic write while still tagging its rows ``IMPORTED``. 

69 """ 

70 

71 source: str 

72 file_hash: str 

73 records: list[dict] 

74 needs_cleanup: bool 

75 stat: SourceStat | None = None 

76 page_texts: list[dict] | None = None 

77 source_type: SourceType = SourceType.DOCUMENT 

78 meta: SourceMeta | None = None 

79 

80 

81class ChunkType(StrEnum): 

82 """Values for the ``chunk_type`` column. 

83 

84 Documents ingest as ``RAW``, extracted tables as ``TABLE`` (when table 

85 extraction is on), and wiki pages written by the wiki producer as 

86 ``WIKI``. Callers filter with ``Store.search(chunk_type=...)``; a ``RAW`` 

87 filter also covers table chunks, since both are document content. 

88 """ 

89 

90 RAW = "raw" 

91 TABLE = "table" 

92 WIKI = "wiki" 

93 

94 

95# ``schema_version`` is an integer for forward-compat. Bump only if we ever need to 

96# add or rename a meta column without forcing every store to drop_all. 

97# 2: nomic-embed document prefixes are stamped at ingest (see 

98# lilbee.retrieval.embedding_profiles doc_prefix_since). 

99META_SCHEMA_VERSION = 2 

100 

101# Always-true predicate used to clear the single-row ``_meta`` table before re-insert. 

102# Lance's ``Table.delete`` requires a SQL where clause; this matches every row without 

103# coupling the deletion to any specific column's value domain. 

104META_DELETE_ALL_PREDICATE = "schema_version IS NOT NULL" 

105 

106# Same, for the single-row ``_entity_schema`` table. 

107ENTITY_SCHEMA_DELETE_ALL_PREDICATE = "updated_at IS NOT NULL" 

108 

109 

110class EntitySchemaState(TypedDict): 

111 """Single-row state of the induced entity schema. 

112 

113 ``applied`` records whether a full extraction pass completed under this 

114 schema; an interrupted pass leaves it False so the next sync redoes the 

115 (idempotent) pass. ``source_count`` is how many documents the index held 

116 when the schema was induced, which is what the next sync compares against 

117 to decide the corpus has drifted far enough to re-induce. 

118 """ 

119 

120 schema_json: str 

121 applied: bool 

122 source_count: int 

123 updated_at: str 

124 

125 

126class SearchScope(StrEnum): 

127 """What the user wants to search over. 

128 

129 Values are used as-is on CLI flags, MCP params, and HTTP query strings. 

130 ``BOTH`` resolves to a ``None`` ``chunk_type`` (no filter); the two 

131 others map 1:1 to the chunks-table values. 

132 """ 

133 

134 RAW = ChunkType.RAW 

135 WIKI = ChunkType.WIKI 

136 BOTH = "both" 

137 

138 

139def scope_to_chunk_type(scope: SearchScope | str | None) -> ChunkType | None: 

140 """Translate a user-facing scope into a ``Store.search`` ``chunk_type`` arg. 

141 

142 ``None``/``"both"`` → no filter. ``"raw"`` / ``"wiki"`` → the matching 

143 ``ChunkType``. Raises ``ValueError`` on any other string. 

144 """ 

145 if scope is None: 

146 return None 

147 normalized = SearchScope(scope) 

148 if normalized is SearchScope.BOTH: 

149 return None 

150 return ChunkType(normalized.value) 

151 

152 

153class SearchChunk(BaseModel): 

154 """A search result from LanceDB. 

155 Every store search path sets ``score``: canonical [0, 1] relevance, 

156 higher = better. Ranking, filtering, and selection compare only this 

157 field; the arm-specific fields below it are provenance. 

158 Vector-arm rows carry ``distance``; FTS-arm rows carry ``bm25_score``; 

159 reranked rows additionally carry ``rerank_score`` (higher = better). 

160 Memory pseudo-rows carry ``memory_id`` and a ``memory:<id>`` source. 

161 """ 

162 

163 model_config = ConfigDict(populate_by_name=True) 

164 

165 source: str 

166 content_type: str 

167 chunk_type: ChunkType = ChunkType.RAW 

168 

169 @field_validator("chunk_type", mode="before") 

170 @classmethod 

171 def _coerce_none_chunk_type(cls, v: str | None) -> str: 

172 """LanceDB rows from before the chunk_type column was added return None.""" 

173 return v if v is not None else ChunkType.RAW 

174 

175 # Document title at ingest time; None on rows written before the column 

176 # existed (or by writers that carry no title, e.g. wiki pages). 

177 title: str | None = None 

178 

179 page_start: int 

180 page_end: int 

181 line_start: int 

182 line_end: int 

183 chunk: str 

184 chunk_index: int 

185 vector: list[float] = Field(repr=False) 

186 distance: float | None = Field(None, alias="_distance") 

187 # Legacy ``_relevance_score`` passthrough. No store path populates it and 

188 # no ranking code reads it; it survives only as a display-compatible field 

189 # for rows produced by external LanceDB rerankers. 

190 relevance_score: float | None = Field(None, validation_alias="_relevance_score") 

191 # FTS/BM25-only rows carry a raw, unbounded ``_score``. It lives in its own 

192 # field so it never contaminates the canonical ``score``; the 

193 # confidence-based expansion-skip reads it (squashed to [0, 1]), and the 

194 # relevance filter treats its presence as lexical support. 

195 bm25_score: float | None = Field(None, validation_alias="_score") 

196 rerank_score: float | None = None 

197 # Canonical relevance in [0, 1], set by the store on every search path: 

198 # normalized reciprocal-rank fusion on the hybrid path, clamped cosine 

199 # similarity on vector-only, list-normalized BM25 on FTS-only probes. 

200 score: float | None = None 

201 # Set only on memory pseudo-rows built from a recalled MemoryRow (never by 

202 # the store): the memory id, so consumers can tell a memory source apart 

203 # from a retrieved passage and link it back to /api/memories. 

204 memory_id: str | None = None 

205 

206 

207class SourceRecord(TypedDict): 

208 """A tracked source document record. 

209 

210 The stat columns are absent on rows read from stores created before they 

211 existed; ``source_stat`` is the accessor that folds absence and the 

212 ``SOURCE_STAT_UNKNOWN`` sentinel into ``None``. 

213 """ 

214 

215 filename: str 

216 file_hash: str 

217 ingested_at: str 

218 chunk_count: int 

219 source_type: str 

220 size_bytes: NotRequired[int] 

221 mtime_ns: NotRequired[int] 

222 stat_captured_ns: NotRequired[int] 

223 # Extraction-time document metadata; absent or None on rows written 

224 # before the columns existed, and None when the extractor reported none. 

225 title: NotRequired[str | None] 

226 authors: NotRequired[str | None] 

227 created_at: NotRequired[str | None] 

228 

229 

230# Sentinel for the stat columns on rows written before they existed (or for 

231# detached imports with no backing file). Planning treats it as "unknown: re-hash". 

232SOURCE_STAT_UNKNOWN = -1 

233 

234 

235class SourceStat(NamedTuple): 

236 """File size and mtime captured when a source was hashed, plus the capture time. 

237 

238 ``captured_ns`` is the wall-clock time the stat was taken; the sync planner 

239 hashes a file whose mtime is not strictly older than it (racily clean). 

240 """ 

241 

242 size_bytes: int 

243 mtime_ns: int 

244 captured_ns: int = SOURCE_STAT_UNKNOWN 

245 

246 

247def source_stat(record: SourceRecord) -> SourceStat | None: 

248 """Stored stat for a source row, or None when unknown. 

249 

250 The stat columns are nullable ``int64``, so a row can carry an explicit 

251 ``None`` (an import, or a write before the columns existed) as well as a 

252 missing key or the ``SOURCE_STAT_UNKNOWN`` sentinel. All three mean "no 

253 usable stat": return None so the caller re-hashes instead of crashing on 

254 ``int(None)``. 

255 """ 

256 size = record.get("size_bytes") 

257 mtime = record.get("mtime_ns") 

258 captured = record.get("stat_captured_ns") 

259 if size is None or mtime is None or size == SOURCE_STAT_UNKNOWN or mtime == SOURCE_STAT_UNKNOWN: 

260 return None 

261 captured_ns = SOURCE_STAT_UNKNOWN if captured is None else int(captured) 

262 return SourceStat(int(size), int(mtime), captured_ns) 

263 

264 

265class SourceStatBackfill(NamedTuple): 

266 """An already-tracked source row paired with its freshly verified stat.""" 

267 

268 record: SourceRecord 

269 stat: SourceStat 

270 

271 

272class PageTextRecord(TypedDict): 

273 """One row of the per-page text dataset, matching ``_page_texts``. 

274 

275 The export dataset additionally carries the source's extraction metadata 

276 (denormalized onto every page row) so an export/import cycle preserves it; 

277 these are absent on rows read from the ``_page_texts`` table and on datasets 

278 exported before the columns existed. 

279 """ 

280 

281 source: str 

282 page: int 

283 text: str 

284 content_type: str 

285 title: NotRequired[str | None] 

286 authors: NotRequired[str | None] 

287 created_at: NotRequired[str | None] 

288 

289 

290class CitationRecord(TypedDict): 

291 """A citation linking a wiki chunk to a specific source location.""" 

292 

293 wiki_source: str 

294 wiki_chunk_index: int 

295 citation_key: str 

296 claim_type: str 

297 source_filename: str 

298 source_hash: str 

299 page_start: int 

300 page_end: int 

301 line_start: int 

302 line_end: int 

303 excerpt: str 

304 created_at: str 

305 

306 

307class MemoryKind(StrEnum): 

308 """Whether a memory is an always-injected preference or a similarity-recalled fact.""" 

309 

310 PREFERENCE = "preference" 

311 FACT = "fact" 

312 

313 

314class MemorySource(StrEnum): 

315 """Provenance of a memory: user-typed, LLM-extracted, or agent-written.""" 

316 

317 MANUAL = "manual" 

318 EXTRACTED = "extracted" 

319 AGENT = "agent" 

320 

321 

322# Memory owner namespaces. ``"local"`` is the single human (TUI/CLI/REST); agents own 

323# ``"agent:<id>"`` namespaces. The prefix lives only here so it is never hand-spliced. 

324LOCAL_OWNER = "local" 

325AGENT_OWNER_PREFIX = "agent:" 

326 

327# Memory pseudo-source values on SearchChunk: ``"memory:<id>"`` with a 

328# ``"memory"`` content type. Built by retrieval, never stored in the chunks 

329# table, so recalled facts show in sources without a backing document. 

330MEMORY_SOURCE_PREFIX = "memory:" 

331MEMORY_CONTENT_TYPE = "memory" 

332 

333 

334def agent_owner(agent_id: str) -> str: 

335 """Owner string for an agent identity (``"opencode"`` -> ``"agent:opencode"``).""" 

336 return f"{AGENT_OWNER_PREFIX}{agent_id}" 

337 

338 

339def is_agent_owner(owner: str) -> bool: 

340 """True when *owner* is an agent namespace rather than the local human.""" 

341 return owner.startswith(AGENT_OWNER_PREFIX) 

342 

343 

344def memory_source(memory_id: str) -> str: 

345 """Source value for a recalled memory (``"abc"`` -> ``"memory:abc"``).""" 

346 return f"{MEMORY_SOURCE_PREFIX}{memory_id}" 

347 

348 

349def is_memory_source(source: str) -> bool: 

350 """True when *source* names a recalled memory rather than a document.""" 

351 return source.startswith(MEMORY_SOURCE_PREFIX) 

352 

353 

354class MemoryRow(BaseModel): 

355 """A long-term memory entry in the per-library ``_memories`` table. 

356 

357 Built from a LanceDB row via ``MemoryRow(**row)`` (which coerces the ``kind`` 

358 and ``source`` strings to enums) and written back via ``model_dump(mode="json")``. 

359 Extra keys like a search ``_distance`` are ignored on construction. 

360 """ 

361 

362 model_config = ConfigDict(extra="ignore") 

363 

364 id: str 

365 owner: str 

366 shared: bool 

367 kind: MemoryKind 

368 source: MemorySource 

369 text: str 

370 vector: list[float] = Field(repr=False) 

371 created_at: str 

372 updated_at: str 

373 

374 

375class StoreMeta(TypedDict): 

376 """Single-row store metadata recording the embedding model used to build the store. 

377 

378 Compatibility is checked before every read and write. When ``cfg.embedding_model`` 

379 or ``cfg.embedding_dim`` drifts from the persisted row, the store refuses to serve 

380 until ``lilbee rebuild`` (CLI) or ``POST /api/sync {"force_rebuild": true}`` (HTTP) 

381 rewrites the chunks under the new model. 

382 

383 ``updated_at`` is an ISO 8601 UTC timestamp produced by ``datetime.isoformat()``; 

384 kept as ``str`` to match the LanceDB ``utf8`` schema column. 

385 """ 

386 

387 embedding_model: str 

388 embedding_dim: int 

389 schema_version: int 

390 updated_at: str 

391 

392 

393class IndexMismatch(BaseModel): 

394 """The embedder drift between a persisted index and the configured model. 

395 

396 ``adoptable`` is true when the dimensions agree, so switching the embedding 

397 model back to ``persisted_model`` makes the index searchable without a rebuild. 

398 """ 

399 

400 persisted_model: str 

401 persisted_dim: int 

402 current_model: str 

403 current_dim: int 

404 adoptable: bool 

405 message: str 

406 

407 

408class EmbeddingModelMismatchError(RuntimeError): 

409 """Raised when stored vectors were built with a different embedder than ``cfg``. 

410 

411 Carries the persisted and configured refs and dims so each surface renders its 

412 own recovery affordance (TUI prompt, CLI command, REST body) from the facts. 

413 """ 

414 

415 def __init__( 

416 self, 

417 *, 

418 persisted_model: str, 

419 persisted_dim: int, 

420 current_model: str, 

421 current_dim: int, 

422 ) -> None: 

423 self.persisted_model = persisted_model 

424 self.persisted_dim = persisted_dim 

425 self.current_model = current_model 

426 self.current_dim = current_dim 

427 super().__init__(self._build_message()) 

428 

429 @property 

430 def dims_match(self) -> bool: 

431 """True when the index is adoptable by switching embedder alone (same dim).""" 

432 return self.persisted_dim == self.current_dim 

433 

434 def describe(self) -> IndexMismatch: 

435 """The drift as a serializable record for sync results and status payloads.""" 

436 return IndexMismatch( 

437 persisted_model=self.persisted_model, 

438 persisted_dim=self.persisted_dim, 

439 current_model=self.current_model, 

440 current_dim=self.current_dim, 

441 adoptable=self.dims_match, 

442 message=str(self), 

443 ) 

444 

445 def _build_message(self) -> str: 

446 if self.dims_match: 

447 return ( 

448 f"This index was built with embedding model '{self.persisted_model}', " 

449 f"but lilbee is configured to use '{self.current_model}'. Configure lilbee " 

450 f"to use '{self.persisted_model}' to search this index, or rebuild it under " 

451 f"'{self.current_model}'." 

452 ) 

453 return ( 

454 f"This index was built with embedding model '{self.persisted_model}' " 

455 f"(dim {self.persisted_dim}), which differs from the current " 

456 f"'{self.current_model}' (dim {self.current_dim}). The dimensions differ, " 

457 f"so rebuild the index under '{self.current_model}' to use it." 

458 ) 

459 

460 

461@dataclass 

462class RemoveResult: 

463 """Result of a remove: *removed* can also name held-out files and root labels.""" 

464 

465 removed: list[str] 

466 not_found: list[str]