Coverage for src/lilbee/data/store/types.py: 100%
188 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
1"""Public dataclasses, TypedDicts, enums, and constants for the store package."""
3from __future__ import annotations
5from dataclasses import dataclass
6from datetime import timedelta
7from enum import StrEnum
8from typing import NamedTuple, NotRequired, TypedDict
10from pydantic import BaseModel, ConfigDict, Field, field_validator
12# How often readers re-check the manifest for new versions from other processes.
13# Zero means strong consistency (every read checks); higher values reduce disk I/O
14# on slow media (HDD) at the cost of serving slightly stale data.
15READ_CONSISTENCY_INTERVAL = timedelta(seconds=5)
18@dataclass
19class ConceptRecords:
20 """Rows for the three concept tables, built from one or more files' chunks."""
22 nodes: list[dict]
23 edges: list[dict]
24 chunk_concepts: list[dict]
26 @classmethod
27 def merged(cls, batches: list[ConceptRecords]) -> ConceptRecords:
28 """Concatenate several record sets into one batched write unit."""
29 return cls(
30 nodes=[row for batch in batches for row in batch.nodes],
31 edges=[row for batch in batches for row in batch.edges],
32 chunk_concepts=[row for batch in batches for row in batch.chunk_concepts],
33 )
36class SourceType(StrEnum):
37 """Values for the ``_sources.source_type`` column.
39 ``DOCUMENT`` mirrors a file under ``documents/`` and is managed by the
40 file-driven sync. ``IMPORTED`` is detached: it came from ``lilbee import``
41 and has no backing file, so sync must not treat it as a missing document.
42 """
44 DOCUMENT = "document"
45 IMPORTED = "imported"
48class SourceMeta(NamedTuple):
49 """Document-level metadata captured at extraction time.
51 ``title`` is always derivable (extraction metadata or the cleaned filename
52 stem); ``authors`` and ``created_at`` are only present when the extractor
53 reports them. Empty strings persist as NULL so old and new rows read alike.
54 """
56 title: str = ""
57 authors: str = ""
58 created_at: str = ""
61class ChunkWrite(NamedTuple):
62 """One document's chunks plus its source-table update, for a batched write.
64 ``Store.write_chunks_batch`` folds many of these into a single locked
65 transaction so bulk ingest doesn't pay a write-lock acquisition per document.
66 ``page_texts`` rows land in the same transaction, after the cleanup delete
67 and before the source row. ``source_type`` lets the detached import path
68 reuse the same atomic write while still tagging its rows ``IMPORTED``.
69 """
71 source: str
72 file_hash: str
73 records: list[dict]
74 needs_cleanup: bool
75 stat: SourceStat | None = None
76 page_texts: list[dict] | None = None
77 source_type: SourceType = SourceType.DOCUMENT
78 meta: SourceMeta | None = None
81class ChunkType(StrEnum):
82 """Values for the ``chunk_type`` column.
84 Documents ingest as ``RAW``, extracted tables as ``TABLE`` (when table
85 extraction is on), and wiki pages written by the wiki producer as
86 ``WIKI``. Callers filter with ``Store.search(chunk_type=...)``; a ``RAW``
87 filter also covers table chunks, since both are document content.
88 """
90 RAW = "raw"
91 TABLE = "table"
92 WIKI = "wiki"
95# ``schema_version`` is an integer for forward-compat. Bump only if we ever need to
96# add or rename a meta column without forcing every store to drop_all.
97# 2: nomic-embed document prefixes are stamped at ingest (see
98# lilbee.retrieval.embedding_profiles doc_prefix_since).
99META_SCHEMA_VERSION = 2
101# Always-true predicate used to clear the single-row ``_meta`` table before re-insert.
102# Lance's ``Table.delete`` requires a SQL where clause; this matches every row without
103# coupling the deletion to any specific column's value domain.
104META_DELETE_ALL_PREDICATE = "schema_version IS NOT NULL"
106# Same, for the single-row ``_entity_schema`` table.
107ENTITY_SCHEMA_DELETE_ALL_PREDICATE = "updated_at IS NOT NULL"
110class EntitySchemaState(TypedDict):
111 """Single-row state of the induced entity schema.
113 ``applied`` records whether a full extraction pass completed under this
114 schema; an interrupted pass leaves it False so the next sync redoes the
115 (idempotent) pass. ``source_count`` is how many documents the index held
116 when the schema was induced, which is what the next sync compares against
117 to decide the corpus has drifted far enough to re-induce.
118 """
120 schema_json: str
121 applied: bool
122 source_count: int
123 updated_at: str
126class SearchScope(StrEnum):
127 """What the user wants to search over.
129 Values are used as-is on CLI flags, MCP params, and HTTP query strings.
130 ``BOTH`` resolves to a ``None`` ``chunk_type`` (no filter); the two
131 others map 1:1 to the chunks-table values.
132 """
134 RAW = ChunkType.RAW
135 WIKI = ChunkType.WIKI
136 BOTH = "both"
139def scope_to_chunk_type(scope: SearchScope | str | None) -> ChunkType | None:
140 """Translate a user-facing scope into a ``Store.search`` ``chunk_type`` arg.
142 ``None``/``"both"`` → no filter. ``"raw"`` / ``"wiki"`` → the matching
143 ``ChunkType``. Raises ``ValueError`` on any other string.
144 """
145 if scope is None:
146 return None
147 normalized = SearchScope(scope)
148 if normalized is SearchScope.BOTH:
149 return None
150 return ChunkType(normalized.value)
153class SearchChunk(BaseModel):
154 """A search result from LanceDB.
155 Every store search path sets ``score``: canonical [0, 1] relevance,
156 higher = better. Ranking, filtering, and selection compare only this
157 field; the arm-specific fields below it are provenance.
158 Vector-arm rows carry ``distance``; FTS-arm rows carry ``bm25_score``;
159 reranked rows additionally carry ``rerank_score`` (higher = better).
160 Memory pseudo-rows carry ``memory_id`` and a ``memory:<id>`` source.
161 """
163 model_config = ConfigDict(populate_by_name=True)
165 source: str
166 content_type: str
167 chunk_type: ChunkType = ChunkType.RAW
169 @field_validator("chunk_type", mode="before")
170 @classmethod
171 def _coerce_none_chunk_type(cls, v: str | None) -> str:
172 """LanceDB rows from before the chunk_type column was added return None."""
173 return v if v is not None else ChunkType.RAW
175 # Document title at ingest time; None on rows written before the column
176 # existed (or by writers that carry no title, e.g. wiki pages).
177 title: str | None = None
179 page_start: int
180 page_end: int
181 line_start: int
182 line_end: int
183 chunk: str
184 chunk_index: int
185 vector: list[float] = Field(repr=False)
186 distance: float | None = Field(None, alias="_distance")
187 # Legacy ``_relevance_score`` passthrough. No store path populates it and
188 # no ranking code reads it; it survives only as a display-compatible field
189 # for rows produced by external LanceDB rerankers.
190 relevance_score: float | None = Field(None, validation_alias="_relevance_score")
191 # FTS/BM25-only rows carry a raw, unbounded ``_score``. It lives in its own
192 # field so it never contaminates the canonical ``score``; the
193 # confidence-based expansion-skip reads it (squashed to [0, 1]), and the
194 # relevance filter treats its presence as lexical support.
195 bm25_score: float | None = Field(None, validation_alias="_score")
196 rerank_score: float | None = None
197 # Canonical relevance in [0, 1], set by the store on every search path:
198 # normalized reciprocal-rank fusion on the hybrid path, clamped cosine
199 # similarity on vector-only, list-normalized BM25 on FTS-only probes.
200 score: float | None = None
201 # Set only on memory pseudo-rows built from a recalled MemoryRow (never by
202 # the store): the memory id, so consumers can tell a memory source apart
203 # from a retrieved passage and link it back to /api/memories.
204 memory_id: str | None = None
207class SourceRecord(TypedDict):
208 """A tracked source document record.
210 The stat columns are absent on rows read from stores created before they
211 existed; ``source_stat`` is the accessor that folds absence and the
212 ``SOURCE_STAT_UNKNOWN`` sentinel into ``None``.
213 """
215 filename: str
216 file_hash: str
217 ingested_at: str
218 chunk_count: int
219 source_type: str
220 size_bytes: NotRequired[int]
221 mtime_ns: NotRequired[int]
222 stat_captured_ns: NotRequired[int]
223 # Extraction-time document metadata; absent or None on rows written
224 # before the columns existed, and None when the extractor reported none.
225 title: NotRequired[str | None]
226 authors: NotRequired[str | None]
227 created_at: NotRequired[str | None]
230# Sentinel for the stat columns on rows written before they existed (or for
231# detached imports with no backing file). Planning treats it as "unknown: re-hash".
232SOURCE_STAT_UNKNOWN = -1
235class SourceStat(NamedTuple):
236 """File size and mtime captured when a source was hashed, plus the capture time.
238 ``captured_ns`` is the wall-clock time the stat was taken; the sync planner
239 hashes a file whose mtime is not strictly older than it (racily clean).
240 """
242 size_bytes: int
243 mtime_ns: int
244 captured_ns: int = SOURCE_STAT_UNKNOWN
247def source_stat(record: SourceRecord) -> SourceStat | None:
248 """Stored stat for a source row, or None when unknown.
250 The stat columns are nullable ``int64``, so a row can carry an explicit
251 ``None`` (an import, or a write before the columns existed) as well as a
252 missing key or the ``SOURCE_STAT_UNKNOWN`` sentinel. All three mean "no
253 usable stat": return None so the caller re-hashes instead of crashing on
254 ``int(None)``.
255 """
256 size = record.get("size_bytes")
257 mtime = record.get("mtime_ns")
258 captured = record.get("stat_captured_ns")
259 if size is None or mtime is None or size == SOURCE_STAT_UNKNOWN or mtime == SOURCE_STAT_UNKNOWN:
260 return None
261 captured_ns = SOURCE_STAT_UNKNOWN if captured is None else int(captured)
262 return SourceStat(int(size), int(mtime), captured_ns)
265class SourceStatBackfill(NamedTuple):
266 """An already-tracked source row paired with its freshly verified stat."""
268 record: SourceRecord
269 stat: SourceStat
272class PageTextRecord(TypedDict):
273 """One row of the per-page text dataset, matching ``_page_texts``.
275 The export dataset additionally carries the source's extraction metadata
276 (denormalized onto every page row) so an export/import cycle preserves it;
277 these are absent on rows read from the ``_page_texts`` table and on datasets
278 exported before the columns existed.
279 """
281 source: str
282 page: int
283 text: str
284 content_type: str
285 title: NotRequired[str | None]
286 authors: NotRequired[str | None]
287 created_at: NotRequired[str | None]
290class CitationRecord(TypedDict):
291 """A citation linking a wiki chunk to a specific source location."""
293 wiki_source: str
294 wiki_chunk_index: int
295 citation_key: str
296 claim_type: str
297 source_filename: str
298 source_hash: str
299 page_start: int
300 page_end: int
301 line_start: int
302 line_end: int
303 excerpt: str
304 created_at: str
307class MemoryKind(StrEnum):
308 """Whether a memory is an always-injected preference or a similarity-recalled fact."""
310 PREFERENCE = "preference"
311 FACT = "fact"
314class MemorySource(StrEnum):
315 """Provenance of a memory: user-typed, LLM-extracted, or agent-written."""
317 MANUAL = "manual"
318 EXTRACTED = "extracted"
319 AGENT = "agent"
322# Memory owner namespaces. ``"local"`` is the single human (TUI/CLI/REST); agents own
323# ``"agent:<id>"`` namespaces. The prefix lives only here so it is never hand-spliced.
324LOCAL_OWNER = "local"
325AGENT_OWNER_PREFIX = "agent:"
327# Memory pseudo-source values on SearchChunk: ``"memory:<id>"`` with a
328# ``"memory"`` content type. Built by retrieval, never stored in the chunks
329# table, so recalled facts show in sources without a backing document.
330MEMORY_SOURCE_PREFIX = "memory:"
331MEMORY_CONTENT_TYPE = "memory"
334def agent_owner(agent_id: str) -> str:
335 """Owner string for an agent identity (``"opencode"`` -> ``"agent:opencode"``)."""
336 return f"{AGENT_OWNER_PREFIX}{agent_id}"
339def is_agent_owner(owner: str) -> bool:
340 """True when *owner* is an agent namespace rather than the local human."""
341 return owner.startswith(AGENT_OWNER_PREFIX)
344def memory_source(memory_id: str) -> str:
345 """Source value for a recalled memory (``"abc"`` -> ``"memory:abc"``)."""
346 return f"{MEMORY_SOURCE_PREFIX}{memory_id}"
349def is_memory_source(source: str) -> bool:
350 """True when *source* names a recalled memory rather than a document."""
351 return source.startswith(MEMORY_SOURCE_PREFIX)
354class MemoryRow(BaseModel):
355 """A long-term memory entry in the per-library ``_memories`` table.
357 Built from a LanceDB row via ``MemoryRow(**row)`` (which coerces the ``kind``
358 and ``source`` strings to enums) and written back via ``model_dump(mode="json")``.
359 Extra keys like a search ``_distance`` are ignored on construction.
360 """
362 model_config = ConfigDict(extra="ignore")
364 id: str
365 owner: str
366 shared: bool
367 kind: MemoryKind
368 source: MemorySource
369 text: str
370 vector: list[float] = Field(repr=False)
371 created_at: str
372 updated_at: str
375class StoreMeta(TypedDict):
376 """Single-row store metadata recording the embedding model used to build the store.
378 Compatibility is checked before every read and write. When ``cfg.embedding_model``
379 or ``cfg.embedding_dim`` drifts from the persisted row, the store refuses to serve
380 until ``lilbee rebuild`` (CLI) or ``POST /api/sync {"force_rebuild": true}`` (HTTP)
381 rewrites the chunks under the new model.
383 ``updated_at`` is an ISO 8601 UTC timestamp produced by ``datetime.isoformat()``;
384 kept as ``str`` to match the LanceDB ``utf8`` schema column.
385 """
387 embedding_model: str
388 embedding_dim: int
389 schema_version: int
390 updated_at: str
393class IndexMismatch(BaseModel):
394 """The embedder drift between a persisted index and the configured model.
396 ``adoptable`` is true when the dimensions agree, so switching the embedding
397 model back to ``persisted_model`` makes the index searchable without a rebuild.
398 """
400 persisted_model: str
401 persisted_dim: int
402 current_model: str
403 current_dim: int
404 adoptable: bool
405 message: str
408class EmbeddingModelMismatchError(RuntimeError):
409 """Raised when stored vectors were built with a different embedder than ``cfg``.
411 Carries the persisted and configured refs and dims so each surface renders its
412 own recovery affordance (TUI prompt, CLI command, REST body) from the facts.
413 """
415 def __init__(
416 self,
417 *,
418 persisted_model: str,
419 persisted_dim: int,
420 current_model: str,
421 current_dim: int,
422 ) -> None:
423 self.persisted_model = persisted_model
424 self.persisted_dim = persisted_dim
425 self.current_model = current_model
426 self.current_dim = current_dim
427 super().__init__(self._build_message())
429 @property
430 def dims_match(self) -> bool:
431 """True when the index is adoptable by switching embedder alone (same dim)."""
432 return self.persisted_dim == self.current_dim
434 def describe(self) -> IndexMismatch:
435 """The drift as a serializable record for sync results and status payloads."""
436 return IndexMismatch(
437 persisted_model=self.persisted_model,
438 persisted_dim=self.persisted_dim,
439 current_model=self.current_model,
440 current_dim=self.current_dim,
441 adoptable=self.dims_match,
442 message=str(self),
443 )
445 def _build_message(self) -> str:
446 if self.dims_match:
447 return (
448 f"This index was built with embedding model '{self.persisted_model}', "
449 f"but lilbee is configured to use '{self.current_model}'. Configure lilbee "
450 f"to use '{self.persisted_model}' to search this index, or rebuild it under "
451 f"'{self.current_model}'."
452 )
453 return (
454 f"This index was built with embedding model '{self.persisted_model}' "
455 f"(dim {self.persisted_dim}), which differs from the current "
456 f"'{self.current_model}' (dim {self.current_dim}). The dimensions differ, "
457 f"so rebuild the index under '{self.current_model}' to use it."
458 )
461@dataclass
462class RemoveResult:
463 """Result of a remove: *removed* can also name held-out files and root labels."""
465 removed: list[str]
466 not_found: list[str]