Coverage for src/lilbee/core/config/model.py: 100%
545 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
1"""The :class:`Config` dataclass and the ``cfg`` singleton.
3The settings sources, TOML parser, and the resilient builder that falls
4back to defaults on stale-config validation failures live here too. Every
5``from lilbee.core.config import cfg`` resolves through ``lilbee.core.config.__init__``
6to the same instance defined at module bottom.
7"""
9import logging
10import os
11import re
12from pathlib import Path
13from typing import Any, ClassVar
15from pydantic import Field, ValidationInfo, field_validator, model_validator
16from pydantic_settings import BaseSettings, SettingsConfigDict
18from lilbee.core.system import scaled_chat_ctx_target_default
20from .defaults import (
21 CONFIG_FILE_NAME,
22 DEFAULT_ALLOWED_NER_LABELS,
23 DEFAULT_CORS_ORIGIN_REGEX,
24 DEFAULT_CRAWL_EXCLUDE_PATTERNS,
25 DEFAULT_GENERAL_SYSTEM_PROMPT,
26 DEFAULT_IGNORE_DIRS,
27 DEFAULT_RAG_SYSTEM_PROMPT,
28)
29from .enums import (
30 ChatMode,
31 ClustererBackend,
32 CrawlRenderMode,
33 FtsLanguage,
34 KvCacheType,
35 LlmProvider,
36 OcrPageStrategy,
37 ReasoningMode,
38 RerankerType,
39 TableModel,
40 WikiEntityMode,
41)
42from .parsing import parse_bool
43from .validators import ConfigField
45log = logging.getLogger(__name__)
47# Sentinel for unset Path-typed fields. ``Field(default=Path())`` produces an
48# instance equal to this, so the model_validator can distinguish "user passed
49# the default" from "user explicitly set a value".
50_UNSET_PATH = Path()
52# A Tesseract language code: ISO 639 letters plus script or orientation suffixes
53# (eng, en, chi_sim, jpn_vert). xberg rejects anything else before extracting.
54_TESSERACT_LANGUAGE_CODE = re.compile(r"[a-z]{2,3}(?:_[a-z]+)*")
56# Model roles that can be off. An empty LILBEE_<FIELD> or config.toml value
57# clears one of these; on every other field an empty value counts as unset.
58CLEARABLE_MODEL_FIELDS = frozenset({"vision_model", "reranker_model"})
61def value_is_set(field_name: str, raw: object) -> bool:
62 """Whether an env or config.toml value is set: non-empty, or empty on a clearable model role."""
63 if raw is None:
64 return False
65 return raw != "" or field_name in CLEARABLE_MODEL_FIELDS
68def _as_int(item: Any) -> int:
69 """A force_ocr_pages entry as an int: an int, or a string of digits."""
70 if isinstance(item, str) and item.strip().lstrip("-").isdigit():
71 return int(item)
72 # bool is an int subclass; True is not a page number.
73 if isinstance(item, int) and not isinstance(item, bool):
74 return item
75 raise ValueError(f"force_ocr_pages: {item!r} is not a page number")
78def _split_page_item(item: Any) -> list[Any]:
79 """A string force_ocr_pages item split on commas and newlines; other items as-is."""
80 if isinstance(item, str):
81 return item.replace("\n", ",").split(",")
82 return [item]
85def _page_number(item: Any) -> int:
86 """One force_ocr_pages entry as a 1-indexed page."""
87 page = _as_int(item)
88 if page < 1:
89 raise ValueError(f"force_ocr_pages: page numbers start at 1 (got {page})")
90 return page
93class Config(BaseSettings):
94 """Runtime configuration: one singleton instance, mutated by CLI overrides."""
96 model_config = SettingsConfigDict(
97 env_prefix="LILBEE_",
98 validate_assignment=True,
99 arbitrary_types_allowed=True,
100 extra="ignore",
101 )
103 # Paths: resolved from env/defaults in model_validator(mode='before')
104 data_root: Path = Field(
105 default=Path(),
106 description=(
107 "Root directory for this library. Resolved at start from LILBEE_DATA, "
108 "then a .lilbee/ directory walked up from the working directory, then "
109 "the platform default. Every other path below hangs off it"
110 ),
111 )
112 # Writable so plugin-managed servers can pivot storage to a vault path on
113 # first boot; rebuild the index after migrating.
114 documents_dir: Path = ConfigField(default=Path(), writable=True)
115 # External source roots ``add`` registered, mapping label -> absolute path.
116 # lilbee indexes the files where they live (no copy, no symlink); the label
117 # prefixes their source keys so a root at /data/corpus keys as ``corpus/…``.
118 # Managed by ``add`` / ``remove``, so it is writable (persisted to
119 # config.toml) but not surfaced in the settings UI.
120 linked_roots: dict[str, str] = ConfigField(
121 default_factory=dict,
122 writable=True,
123 public=False,
124 description=(
125 "External source roots that `add` registered, as label -> absolute path. "
126 "`add` and `remove` maintain it; do not edit it by hand"
127 ),
128 )
129 data_dir: Path = Field(
130 default=Path(),
131 description="Directory holding the database. Defaults to data_root/data",
132 )
133 lancedb_dir: Path = Field(
134 default=Path(),
135 description=(
136 "Directory holding the LanceDB vector tables. Defaults to data_root/data/lancedb"
137 ),
138 )
139 models_dir: Path = Field(
140 default=Path(),
141 description=(
142 "Directory holding downloaded model files. Shared across libraries, so a "
143 "model pulled for one is available to all"
144 ),
145 )
146 # Markdown vault root; when set, search results carry a vault-relative
147 # ``vault_path`` so a host UI can deep-link into the vault.
148 vault_base: Path | None = ConfigField(default=None, writable=True)
150 # Human-readable label for the active lilbee. Empty falls back to
151 # "global" for the platform default dir, otherwise the project path
152 # (~-substituted and left-truncated to a hard cap).
153 lilbee_name: str = ConfigField(default="", writable=True)
154 # If True, the status bar pill shows the full absolute path: expands
155 # "global" to the on-disk platform-default path and skips the
156 # ~-substitution / left-truncation for project paths. Toggled by F4.
157 show_lilbee_path: bool = ConfigField(default=False, writable=True)
159 # Whether an agent launcher (opencode, hermes) registers lilbee's MCP search
160 # tool into the agent's config. Per-launch --mcp/--no-mcp overrides it.
161 agent_mcp_enabled: bool = ConfigField(default=True, writable=True)
163 # Empty = not configured, same convention as vision_model. A fresh install
164 # has no models; the catalog assigns these on the first download.
165 chat_model: str = Field(default="")
166 embedding_model: str = Field(default="")
167 # Vision OCR model for scanned PDFs and image-only pages. Empty = disabled;
168 # there is no cross-role fallback onto the chat model even if multimodal.
169 vision_model: str = ConfigField(default="", public=True)
170 embedding_dim: int = Field(
171 default=768,
172 ge=1,
173 description=(
174 "Vector width the index is built with. The embedding model sets it; "
175 "change it only to match a model lilbee cannot introspect"
176 ),
177 )
178 chunk_size: int = ConfigField(default=512, ge=64, writable=True, reindex=True)
179 chunk_overlap: int = ConfigField(default=100, ge=0, writable=True, reindex=True)
180 # A file over this many chunks is skipped before embedding; 0 lifts the ceiling.
181 max_chunks_per_file: int = ConfigField(default=3_000, ge=0, writable=True)
182 # Workers for the parallel discovery/hash planning pass. 0 = auto, sized to
183 # the container-aware CPU budget (see runtime.cpu.available_cpu_count).
184 # `add --max-cpus N` sets this per invocation. Sizes only the planning pass,
185 # not the GPU-fed extract/embed batch.
186 ingest_workers: int = ConfigField(default=0, ge=0, writable=True)
187 # Worker PROCESSES for a bulk ingest (distinct from ingest_workers, which sizes
188 # the planning pass's threads). Each owns a GPU, a private store and its own
189 # slice of the corpus, and the shards are folded into one index at the end.
190 # 0 = auto: one worker per visible card, used once the corpus is big enough to
191 # pay for them. N pins the count; worker i takes card i % card_count, so more
192 # workers than cards share a card's engine rather than double-booking it.
193 ingest_processes: int = ConfigField(default=0, ge=0, writable=True)
194 # Passages packed into one embed request. Larger batches keep a GPU's
195 # continuous-batching slots full: small per-passage requests leave the card
196 # batch-starved (~96% util, low throughput). The engine still re-splits to
197 # its physical batch, so raising this only helps up to the server's --batch.
198 embed_batch_sequences: int = ConfigField(
199 default=64,
200 ge=1,
201 writable=True,
202 description=(
203 "Passages packed into one embed request. Larger batches keep a GPU's "
204 "continuous-batching slots full. The engine re-splits to its physical "
205 "batch, so raising this helps only up to the server's --batch"
206 ),
207 )
208 # Files allowed in their compute phase at once during ingest. 0 = auto: the
209 # ceiling scales with the detected embed fleet (replicas x per-replica
210 # in-flight) so a multi-GPU box is kept fed without a manual cap, falling back
211 # to the CPU quota on a single card. Set a positive value only to override the
212 # auto sizing. Sizes the extract+embed fan-out, not the plan pass.
213 ingest_max_inflight: int = ConfigField(
214 default=0,
215 ge=0,
216 writable=True,
217 description=(
218 "Files allowed in their compute phase at once during ingest. 0 = auto, "
219 "scaled to the detected embed fleet. Sizes the extract and embed fan-out, "
220 "not the planning pass"
221 ),
222 )
223 # Gate for the pre-ask sync; --no-sync overrides per invocation.
224 auto_sync: bool = ConfigField(default=True, writable=True)
225 max_embed_chars: int = Field(
226 default=2000,
227 ge=1,
228 description="Maximum characters sent to the embedding model per chunk. Longer text is cut",
229 )
230 top_k: int = ConfigField(default=12, ge=1, writable=True)
231 max_distance: float = ConfigField(default=0.75, ge=0.0, writable=True)
232 # Abstention floor against the [0, 1] fused relevance score (0.0 = no
233 # filtering). When every retrieved chunk falls below it, ask refuses instead
234 # of feeding noise as context. The fused score normalizes against the
235 # configured weight budget (a constant), so an arm's top hit scores a stable
236 # share of it; useful floors start around 0.4. Tune against your own corpus.
237 min_relevance_score: float = ConfigField(default=0.0, ge=0.0, writable=True)
238 adaptive_threshold: bool = ConfigField(default=False, writable=True)
239 rag_system_prompt: str = ConfigField(
240 default=DEFAULT_RAG_SYSTEM_PROMPT, min_length=1, writable=True
241 )
242 general_system_prompt: str = ConfigField(
243 default=DEFAULT_GENERAL_SYSTEM_PROMPT, min_length=1, writable=True
244 )
245 chat_mode: ChatMode = ConfigField(default=ChatMode.SEARCH, writable=True)
246 ignore_dirs: frozenset[str] = Field(
247 default=DEFAULT_IGNORE_DIRS,
248 description=(
249 "Directory names ingest never walks (.git, node_modules, and similar). "
250 "Use a .lilbeeignore file for per-library patterns"
251 ),
252 )
253 # OCR for scanned PDFs via vision-capable chat model.
254 # None = auto-detect (use OCR if chat model is vision-capable).
255 # True = force OCR regardless of detection.
256 # False = disable OCR entirely.
257 enable_ocr: bool | None = ConfigField(default=None, writable=True)
258 # Per-page timeout in seconds for vision OCR (0 = no limit). Sized so a dense
259 # full-page scan finishes on modest hardware; a raised vision_ocr_max_tokens
260 # needs matching headroom here.
261 ocr_timeout: float = ConfigField(default=300.0, ge=0.0, writable=True)
262 # Outer wall-clock budget for the streamed pool drain: load grace plus
263 # per_page * pages. Tune up for slow hardware (M1 Pro vision is
264 # ~5min/page) or down for fast hardware. ocr_timeout still governs the
265 # per-page expectation that drives the total budget.
266 vision_load_budget_s: float = ConfigField(default=300.0, ge=0.0, writable=True)
267 # Hard cap on tokens generated per OCR page. A real page is well under this;
268 # the vision request's repeat penalty stops a page looping one line, and the cap
269 # bounds any loop that still escapes it. Raising it lengthens per-page
270 # generation on dense scans, so give ocr_timeout matching headroom.
271 vision_ocr_max_tokens: int = ConfigField(default=4096, ge=256, writable=True)
272 # Pages OCR'd concurrently, and the vision server's continuous-batching slots.
273 # A single-page decode underutilizes a modern GPU (~half SM); batching several
274 # pages raises throughput. Each slot adds KV cache, so lower it on small GPUs.
275 vision_ocr_concurrency: int = ConfigField(default=4, ge=1, writable=True)
277 # Tesseract OCR language codes for the scanned-document fallback (used when no
278 # vision model is set), e.g. ["eng"] or ["eng", "deu"]. Set via env as
279 # LILBEE_OCR_LANGUAGE="eng+deu". xberg requires a non-empty list.
280 ocr_language: list[str] = ConfigField(default_factory=lambda: ["eng"], writable=True)
281 # PDF pages xberg OCRs. auto = pages whose native text fails its quality
282 # check; scanned_pages also OCRs every page graded as a scan.
283 ocr_strategy: OcrPageStrategy = ConfigField(default=OcrPageStrategy.AUTO, writable=True)
284 # Scan-grade threshold for scanned_pages. A slide with a full-bleed
285 # background image grades 0.5, so lower this to OCR such slides too.
286 ocr_scan_confidence: float = ConfigField(default=0.7, ge=0.0, le=1.0, writable=True)
287 # 1-indexed pages that lilbee OCRs in every PDF; while set, it replaces the
288 # ocr_strategy page selection. Env form: LILBEE_FORCE_OCR_PAGES="1,3".
289 force_ocr_pages: list[int] = ConfigField(default_factory=list, writable=True)
290 # Typed entity table for exact counting/cross-referencing; corpus-scale pass, off by default.
291 entity_extraction: bool = ConfigField(default=False, writable=True)
292 semantic_chunking: bool = ConfigField(default=False, writable=True)
293 topic_threshold: float = ConfigField(default=0.75, ge=0.0, le=1.0, writable=True)
294 # Size chunks in real tokens via the embedder's tokenizer backend, not the
295 # chars-per-token heuristic. Plain/heading chunkers only; semantic sizes by chars.
296 token_sizing: bool = ConfigField(default=False, writable=True, reindex=True)
297 # Index each recognized table as its own markdown-serialized chunk.
298 table_extraction: bool = ConfigField(default=False, writable=True, reindex=True)
299 # Layout-aware PDF extraction (reading-order sort, header/footer stripping),
300 # run in xberg's AUTO strategy so detection only fires when it helps. Off by
301 # default: enabling it downloads the ONNX layout and table-structure models
302 # and adds per-page inference, which a CPU-only ingest pays for.
303 layout_detection: bool = ConfigField(default=False, writable=True, reindex=True)
304 # Table structure model; only applied when layout_detection is on.
305 table_model: TableModel = ConfigField(
306 default=TableModel.SLANET_AUTO, writable=True, reindex=True
307 )
308 # Wall-clock cap per file for one xberg extraction, seconds. 0 = no cap.
309 # xberg's own default is 600s and applies on its batch path, so leaving this
310 # unset drops a slow file at ten minutes with no lilbee knob to raise it.
311 # Zero matches the uncapped single-file path, and ingest is background work.
312 extraction_timeout: int = ConfigField(default=0, ge=0, writable=True)
313 # Coalesce concurrent extractions into one xberg extract_batch call.
314 batch_extraction: bool = ConfigField(default=False, writable=True)
315 batch_extraction_size: int = ConfigField(default=8, ge=1, writable=True)
316 # xberg's shared thread budget: PDF rendering, OCR and ONNX inference. It
317 # also bounds concurrent Tesseract sessions, which xberg further limits to
318 # what free memory holds. 0 = auto, runtime.cpu.cpu_quota() (half the usable
319 # CPUs). The rayon pool is fixed at the first extraction, so a change takes
320 # full effect after a restart.
321 extraction_threads: int = ConfigField(default=0, ge=0, writable=True)
322 # Size of anyio's thread pool: synchronous handlers (MCP tools, sync routes)
323 # that may run off the event loop at once. The ceiling on agents one daemon
324 # serves before their calls queue.
325 mcp_tool_threads: int = ConfigField(default=40, ge=1, writable=True)
326 # Crawled pages converted to markdown on anyio's thread pool at once. The
327 # conversion is synchronous, so this keeps it off the event loop that serves
328 # requests. 0 converts inline on the loop.
329 crawl_convert_workers: int = ConfigField(default=2, ge=0, writable=True)
330 server_host: str = Field(
331 default="127.0.0.1",
332 description=(
333 "Address `lilbee serve` binds. Loopback by default; set 0.0.0.0 to accept "
334 "connections from the network"
335 ),
336 )
337 server_port: int = Field(
338 default=0,
339 ge=0,
340 le=65535,
341 description="Port `lilbee serve` binds. 0 picks a free port and prints it",
342 )
343 cors_origins: list[str] = Field(
344 default_factory=list,
345 description=(
346 "Extra browser origins the HTTP server accepts, in addition to cors_origin_regex"
347 ),
348 )
349 cors_origin_regex: str = Field(
350 default=DEFAULT_CORS_ORIGIN_REGEX,
351 description="Regular expression matching browser origins the HTTP server accepts",
352 )
353 # Seconds between SSE heartbeat events when the producer queue is idle.
354 # Must stay well below the plugin's STREAM_IDLE_TIMEOUT_MS (120s) so a
355 # single long-running vision OCR page can't starve the client into aborting.
356 sse_heartbeat_interval: float = ConfigField(default=30.0, ge=0.0, writable=True)
357 json_mode: bool = Field(
358 default=False,
359 description=(
360 "Emit structured JSON from CLI commands. The --json flag sets it for one invocation"
361 ),
362 )
363 temperature: float | None = ConfigField(default=0.1, ge=0.0, writable=True)
364 top_p: float | None = ConfigField(default=0.9, ge=0.0, le=1.0, writable=True)
365 top_k_sampling: int | None = ConfigField(default=40, ge=1, writable=True)
366 # 1.1 is llama.cpp's default. Leaving this at None caused n-gram loops
367 # ("tire tire tire...") on some open-weights models.
368 repeat_penalty: float | None = ConfigField(default=1.1, ge=0.0, writable=True)
369 num_ctx: int | None = ConfigField(default=None, ge=1, writable=True)
370 max_tokens: int | None = ConfigField(default=4096, ge=1, writable=True)
371 seed: int | None = ConfigField(default=None, writable=True)
372 llm_provider: LlmProvider = ConfigField(default=LlmProvider.AUTO, writable=True)
373 # Path to a llama-server binary. Empty = use the bundled lilbee-engine
374 # wheel binary, else a llama-server on PATH.
375 llama_server_path: str = ConfigField(default="", writable=True)
376 # Per-server local model-manager URLs. Blank means "use the server's spec
377 # default" (resolved in providers.local_servers.config_urls); the default
378 # URL literal lives only in the spec, which core must not import.
379 ollama_base_url: str = ConfigField(default="", writable=True)
380 lm_studio_base_url: str = ConfigField(default="", writable=True)
381 llm_api_key: str = ConfigField(default="", writable=True, write_only=True)
382 openrouter_api_key: str = ConfigField(default="", writable=True, write_only=True)
383 gemini_api_key: str = ConfigField(default="", writable=True, write_only=True)
384 anthropic_api_key: str = ConfigField(default="", writable=True, write_only=True)
385 openai_api_key: str = ConfigField(default="", writable=True, write_only=True)
386 mistral_api_key: str = ConfigField(default="", writable=True, write_only=True)
387 deepseek_api_key: str = ConfigField(default="", writable=True, write_only=True)
388 hf_token: str = ConfigField(default="", writable=True, write_only=True)
390 # Retrieval quality knobs.
392 # Max chunks per source in top-k; prevents one large file monopolizing results.
393 diversity_max_per_source: int = ConfigField(default=5, ge=1, writable=True)
395 # MMR relevance/diversity tradeoff; 0 = max diversity, 1 = pure relevance
396 # (Carbonell & Goldstein 1998).
397 mmr_lambda: float = ConfigField(default=0.5, ge=0.0, le=1.0, writable=True)
399 # Vector-only search retrieves this many candidates per final result so
400 # MMR reranking has a pool to diversify from. Hybrid search ignores it:
401 # fusion arms stay exactly top_k deep.
402 candidate_multiplier: int = ConfigField(default=3, ge=1, writable=True)
404 # Third lexical arm in hybrid search: BM25 over document titles, fused with
405 # the vector and chunk arms so a query naming a document by title surfaces
406 # its chunks. Off by default until the eval harness measures it.
407 title_search: bool = ConfigField(default=False, writable=True)
409 # Title arm weight relative to a full arm in rank fusion (1.0 = equal voice
410 # with the vector and chunk arms).
411 title_search_weight: float = ConfigField(default=0.5, ge=0.0, le=1.0, writable=True)
413 # Lexical (BM25) arm weight relative to the vector arm in rank fusion.
414 # 1.0 gives the two arms equal voice; lowering it lets
415 # a strong dense embedder dominate on corpora where the lexical arm adds
416 # noise rather than signal. The right value is corpus-dependent and set by
417 # the retrieval benchmark, not guessed here.
418 lexical_fusion_weight: float = ConfigField(default=1.0, ge=0.0, le=1.0, writable=True)
420 # Adaptive fusion: scale the BM25 arm per query by vector-arm confidence
421 # instead of a fixed lexical_fusion_weight (a peaked dense ranking downweights
422 # lexical, a flat one keeps it). OFF by default, pending a benchmark run to
423 # confirm it beats the fixed weight. lexical_fusion_weight is the ceiling the
424 # rule scales down from. Set adaptive_fusion=true to enable it.
425 adaptive_fusion: bool = ConfigField(default=False, writable=True)
427 # Vector-similarity margin at which the lexical arm is fully silenced; smaller
428 # = more aggressive downweighting. 0 disables adaptation entirely (the lexical
429 # arm keeps its full fixed weight).
430 adaptive_fusion_margin: float = ConfigField(default=0.15, ge=0.0, le=2.0, writable=True)
432 # Stemmer/stop-word language for the BM25 (FTS) indexes, a tantivy language
433 # name ("English", "German", "French", ...). Applied when an index is
434 # (re)built, so changing it needs `lilbee rebuild` on an existing store.
435 # A bad name would otherwise fail index creation quietly and hybrid search
436 # would degrade to vector-only.
437 fts_language: FtsLanguage = ConfigField(
438 default=FtsLanguage.ENGLISH, writable=True, reindex=True
439 )
441 @field_validator("fts_language", mode="before")
442 @classmethod
443 def _validate_fts_language(cls, value: Any) -> FtsLanguage:
444 """Accept a language name in any casing, with surrounding whitespace."""
445 try:
446 return FtsLanguage(str(value).strip().title())
447 except ValueError as exc:
448 valid = ", ".join(member.value for member in FtsLanguage)
449 raise ValueError(f"fts_language must be one of: {valid}") from exc
451 # Prefix each chunk's document title to its embedding input (the stored
452 # chunk text is unchanged). Changes the embedding space: toggling it needs
453 # `lilbee rebuild`, so it ships off.
454 embed_titles: bool = ConfigField(default=False, writable=True, reindex=True)
456 # Contextual retrieval: prepend one LLM-written sentence situating each
457 # chunk in its document to the embedding input. One generation per chunk,
458 # so ingest slows substantially; stored text and citations stay verbatim.
459 # Toggling needs `lilbee rebuild`.
460 contextual_enrichment: bool = ConfigField(default=False, writable=True, reindex=True)
462 # Drop tables-of-contents and classification-banner cover/title pages from
463 # search results. OFF by default; validate per corpus, since the cover-page
464 # heuristic can also fire on short banner-carrying body pages. A query-matched
465 # or top-ranked page is never dropped, so removal is limited to structural
466 # chunks the query did not hit.
467 filter_structural_chunks: bool = ConfigField(default=False, writable=True)
469 # Chunk count at/above which sync builds an approximate (ANN) vector index
470 # so search stays fast at millions of vectors. Below this, search uses exact
471 # flat scan (faster and exact for small vaults). 0 disables the ANN index.
472 ann_index_threshold: int = ConfigField(default=50_000, ge=0, writable=True)
474 # Condense a follow-up question into a standalone retrieval query using
475 # the chat history (one LLM call; skipped when there is no history).
476 # Without it, "what about his brother?" is embedded and BM25-matched
477 # with its pronouns.
478 history_rewrite: bool = ConfigField(default=False, writable=True)
480 # Route questions by shape before top-k retrieval: a question naming a
481 # document resolves to that document's chunks; a count-shaped question
482 # runs a full-corpus scan (a count is a corpus property top-k cannot
483 # answer). Unrecognized shapes take the topical path unchanged.
484 intent_routing: bool = ConfigField(default=True, writable=True)
486 # Ask the chat model to classify count questions the deterministic
487 # patterns miss (phrasing variants, other languages). Adds one short LLM
488 # call to every turn the patterns don't already route, so it's opt-in.
489 intent_llm: bool = ConfigField(default=False, writable=True)
491 # LLM-generated alternative queries for expansion. 0 disables.
492 query_expansion_count: int = ConfigField(default=3, ge=0, writable=True)
494 # Skip LLM expansion when tokenized query length ≤ this. The LLM round-trip
495 # dominates latency on small local models; short queries already have strong
496 # BM25/vector signal. Concept-graph expansion still runs. 0 disables the skip.
497 expansion_short_query_tokens: int = ConfigField(default=2, ge=0, writable=True)
499 # Cosine-distance step when adaptive-widening retry kicks in.
500 adaptive_threshold_step: float = ConfigField(default=0.2, gt=0.0, writable=True)
502 # Reject expansion variants below expansion_similarity_threshold.
503 expansion_guardrails: bool = ConfigField(default=True, writable=True)
505 # Min cosine similarity between question and variant embeddings.
506 expansion_similarity_threshold: float = ConfigField(default=0.5, ge=0.0, le=1.0, writable=True)
508 # Saturating BM25 confidence (s / (s + 5)) above which query expansion is
509 # skipped; 0.8 corresponds to a raw BM25 score of 20.
510 expansion_skip_threshold: float = Field(
511 default=0.8,
512 ge=0.0,
513 le=1.0,
514 description=(
515 "Saturating BM25 confidence above which query expansion is skipped. "
516 "0.8 corresponds to a raw BM25 score of 20"
517 ),
518 )
520 # Min relative BM25 top-1 vs top-2 gap ((top - second) / top) to skip expansion.
521 expansion_skip_gap: float = Field(
522 default=0.15,
523 ge=0.0,
524 le=1.0,
525 description=(
526 "Minimum relative BM25 gap between the top two hits that skips query expansion"
527 ),
528 )
530 # Chunks included in LLM context after adaptive selection.
531 max_context_sources: int = ConfigField(default=8, ge=1, writable=True)
533 # Adjacent chunks pulled from the same source on each side of every
534 # selected chunk and merged into one contiguous passage, so a hit that
535 # lands mid-argument regains the text before and after it. 0 disables.
536 # Capped: it is a small chunk radius (useful values are single digits), and
537 # the merged text is token-budget-bounded anyway, so a large value only
538 # inflates per-query fetch cost -- and a misread as a token count (e.g.
539 # 50000) would build a megabyte-long IN-predicate per source.
540 neighbor_expansion: int = ConfigField(default=0, ge=0, le=100, writable=True)
542 # HyDE (Gao et al. 2022): hypothetical-answer embedding search. +~500ms.
543 hyde: bool = ConfigField(default=False, writable=True)
545 # HyDE result weight relative to real-doc search (0.0-1.0).
546 hyde_weight: float = ConfigField(default=0.7, ge=0.0, le=1.0, writable=True)
548 # HyDE prompt template. Must contain {question} placeholder.
549 hyde_prompt: str = Field(
550 default=(
551 "Write a 50-100 word passage that directly answers this question as if "
552 "it were an excerpt from a real document. Do not include any preamble, "
553 "just write the passage.\n\nQuestion: {question}"
554 ),
555 description=(
556 "Prompt template HyDE uses to write the hypothetical answer. Must contain {question}"
557 ),
558 )
560 # Reranker model ref. Empty disables reranking. Native GGUFs run on
561 # llama-server (rank pooling or LLM logprob scoring); hosted refs
562 # (cohere/voyage/jina/together/hf-tei) need the backend extra.
563 reranker_model: str = ConfigField(default="", public=True)
565 # auto detects cross-encoder vs LLM reranker by GGUF arch; override forces one.
566 reranker_type: RerankerType = ConfigField(default=RerankerType.AUTO, writable=True, public=True)
567 # Relevance prompt for LLM rerankers; empty uses the built-in generic template.
568 # A format string with {query} and {document} placeholders.
569 reranker_prompt: str = ConfigField(default="", writable=True, public=True)
571 # Recommend safety-stripped models in Picks and Discover. Off keeps them
572 # in browse and search only; on restores them to the recommendations.
573 include_uncensored: bool = ConfigField(default=False, writable=True)
575 # Long-term chat memory. Off by default (opt-in): when disabled the whole
576 # subsystem is dormant and the write surfaces respond with an enable hint.
577 memory_enabled: bool = ConfigField(default=False, writable=True)
579 # Facts recalled by similarity per turn (preferences are always injected).
580 memory_top_k: int = ConfigField(default=5, ge=0, writable=True)
582 # Cosine-distance ceiling for fact recall; stricter than the document default
583 # because a tiny memory corpus floods at the wider document threshold.
584 memory_max_distance: float = ConfigField(default=0.6, ge=0.0, le=1.0, writable=True)
586 # Char/4 token budget for the injected memory block.
587 memory_token_budget: int = ConfigField(default=512, ge=0, writable=True)
589 # Per-owner soft cap; oldest memories evicted past it (runaway-write guard).
590 memory_max_per_owner: int = ConfigField(default=200, ge=1, writable=True)
592 # Cosine distance below which a new memory is treated as a duplicate of an
593 # existing same-owner memory and updates it in place instead of inserting.
594 memory_dedup_distance: float = ConfigField(default=0.05, ge=0.0, le=1.0, writable=True)
596 # LLM pass that extracts memories from the chat loop. Off by default; extracted
597 # memories are saved directly and recalled like any other memory.
598 memory_auto_extract: bool = ConfigField(default=False, writable=True)
600 # Candidate count sent to the reranker.
601 rerank_candidates: int = ConfigField(default=60, ge=1, writable=True, public=True)
603 # Blend reranker scores with the retrieval fusion signal (position-aware).
604 # Off = the cross-encoder's own ordering stands unblended, which isolates
605 # the reranker's effect when measuring it.
606 rerank_blend: bool = ConfigField(default=True, writable=True, public=True)
608 # Drop candidates whose RAW reranker score falls below this; unset = off.
609 # The scale is provider/model specific (bge logits can be negative, hosted
610 # rerankers use 0..1), so set it against observed scores.
611 rerank_min_score: float | None = ConfigField(default=None, writable=True, public=True)
613 # Date-range filter; only fires when a temporal keyword is detected.
614 temporal_filtering: bool = ConfigField(default=True, writable=True)
616 # If True, emit <think>…</think> content as separate SSE reasoning events;
617 # if False, strip it silently.
618 show_reasoning: bool = ConfigField(default=False, writable=True)
620 # How /v1/chat/completions presents a reasoning model's thinking. ``separate``
621 # reports it in ``reasoning_content`` (OpenAI-compatible); ``inline`` keeps it
622 # in ``content`` as <think> text for clients that never render
623 # ``reasoning_content``; ``off`` asks the model not to think. A request's
624 # ``reasoning`` field overrides this per call.
625 completions_reasoning: ReasoningMode = ConfigField(
626 default=ReasoningMode.SEPARATE, writable=True
627 )
629 # How /v1/messages presents a reasoning model's thinking. ``separate`` reports
630 # it as a ``thinking`` block (Anthropic-compatible); ``inline`` folds it into
631 # the answer text for clients that never render thinking blocks; ``off`` asks
632 # the model not to think and drops any thinking it produces anyway. A
633 # request's ``thinking`` parameter overrides this per call.
634 messages_reasoning: ReasoningMode = ConfigField(default=ReasoningMode.SEPARATE, writable=True)
636 # Maximum reasoning characters before lilbee forces the model to answer.
637 # Per-model overrides apply on top of this default. Approx N/4 tokens.
638 # 0 disables the cap (unlimited reasoning; accept the runaway-loop risk).
639 max_reasoning_chars: int = ConfigField(default=64_000, ge=0, writable=True)
641 # Web crawling.
643 # How crawls fetch pages. ``http`` (default) uses a plain HTTP client with
644 # no browser, the lightweight path for static / server-rendered sites.
645 # ``browser`` launches a tuned Chromium with JavaScript enabled for sites
646 # that render content client-side, at a much higher memory cost.
647 crawl_render_mode: CrawlRenderMode = ConfigField(default=CrawlRenderMode.HTTP, writable=True)
649 # Browser-mode memory levers (only used when crawl_render_mode is browser).
650 # Recycle the Chromium process every N fetched pages to cap RSS growth on a
651 # long recursive crawl; 0 disables recycling. Raise on a roomy machine for
652 # fewer restarts, lower it if memory is tight.
653 crawl_browser_recycle_pages: int = ConfigField(default=50, ge=0, writable=True)
655 # Extra Chromium launch flags for browser-mode crawls. Defaults trim shared
656 # memory and GPU use; override to pass site- or environment-specific flags.
657 crawl_browser_extra_args: list[str] = ConfigField(
658 default_factory=lambda: ["--disable-dev-shm-usage", "--disable-gpu"],
659 writable=True,
660 )
662 # Optional global ceilings. None = no ceiling.
663 crawl_max_depth: int | None = ConfigField(default=None, ge=0, writable=True)
664 crawl_max_pages: int | None = ConfigField(default=None, ge=1, writable=True)
666 # Default page bound for an unbounded crawl (no explicit max_pages /
667 # crawl_max_pages), so a hostile site can't exhaust the disk by default.
668 # An explicit limit overrides it; raise this to crawl larger sites unbounded.
669 crawl_safety_max_pages: int = ConfigField(default=5_000, ge=1, writable=True)
671 # Per-URL fetch timeout, seconds.
672 crawl_timeout: int = ConfigField(default=30, ge=1, writable=True)
674 # 0 = unlimited, default = CPU count.
675 crawl_max_concurrent: int = Field(
676 default=0,
677 ge=0,
678 description="Pages fetched in parallel during a crawl. 0 = unlimited; default = CPU count",
679 )
681 # Seconds between periodic syncs during crawl. 0 = sync only at end.
682 crawl_sync_interval: int = ConfigField(default=30, ge=0, writable=True)
684 # Per-request delay + jitter (defaults chosen to be gentler than crawl4ai's).
685 crawl_mean_delay: float = ConfigField(default=0.5, ge=0.0, writable=True)
686 crawl_max_delay_range: float = ConfigField(default=0.5, ge=0.0, writable=True)
688 # In-flight requests per crawl.
689 crawl_concurrent_requests: int = ConfigField(default=3, ge=1, writable=True)
691 # Per-domain rate-limiter that backs off on HTTP 429/503 and retries.
692 crawl_retry_on_rate_limit: bool = ConfigField(default=True, writable=True)
693 crawl_retry_base_delay_min: float = ConfigField(default=1.0, ge=0.0, writable=True)
694 crawl_retry_base_delay_max: float = ConfigField(default=3.0, ge=0.0, writable=True)
695 crawl_retry_max_backoff: float = ConfigField(default=30.0, ge=0.0, writable=True)
696 crawl_retry_max_attempts: int = ConfigField(default=3, ge=0, writable=True)
698 # Regex patterns dropped at link-discovery time. Defaults block CMS
699 # scaffolding (WordPress admin, archives, tracking params, etc.).
700 crawl_exclude_patterns: list[str] = ConfigField(
701 default_factory=lambda: list(DEFAULT_CRAWL_EXCLUDE_PATTERNS),
702 writable=True,
703 )
705 # Fraction of GPU/unified memory reserved for loaded models.
706 gpu_memory_fraction: float = ConfigField(default=0.75, ge=0.1, le=1.0, writable=True)
708 # Share of a card placement may charge, leaving room for allocator
709 # fragmentation and driver overhead. Tunable because it decides admission: at
710 # the default, a 16 GB machine whose chat model needs 12-13 GB can be refused
711 # chat entirely, and the owner is the one who knows whether that card has the
712 # room. Raising it trades safety margin for the ability to serve at all.
713 usable_vram_fraction: float = ConfigField(default=0.9, ge=0.5, le=1.0, writable=True)
715 # RAM held back for the OS when placing against system memory, in GiB. Capped
716 # at a quarter of total RAM either way, so a small host keeps its proportional
717 # reserve however this is set.
718 system_memory_reserve_gb: float = ConfigField(default=4.0, ge=0.0, le=64.0, writable=True)
720 # Data-parallel replicas of the embed / vision role across GPUs: N independent
721 # servers, round-robined, so large-scale ingest fans the embedding / OCR work
722 # across the whole box. 0 means "auto": one replica per detected GPU, capped by
723 # the VRAM left after the persistent query fleet (chat, one embed, rerank, one
724 # vision) is reserved. A positive value pins the count. The extra replicas are
725 # ingest-only and reclaimed when ingest ends; the persistent query embedder /
726 # vision (replica 0) always exists if its model fits.
727 embed_replicas: int = ConfigField(default=0, ge=0, writable=True)
728 vision_replicas: int = ConfigField(default=0, ge=0, writable=True)
730 # Seconds a model stays loaded after last use. 0 = unload immediately.
731 model_keep_alive: int = ConfigField(default=300, ge=0, writable=True)
733 # Spawn every configured role server at startup instead of on first use.
734 # Trades a slower TUI mount (the role servers cold-start in parallel) for a
735 # responsive first interaction. Roles whose model is unset are skipped, so a
736 # setup with only chat + embed never spawns rerank or vision. Set to false
737 # for headless / scripted use where the first call doesn't need to be fast.
738 worker_pool_eager_start: bool = ConfigField(default=True, writable=True)
740 # Leave the engine fleet running on quit so the next launch adopts it warm.
741 # On keeps the engine process alive across app close so the next launch
742 # binds instantly; its weights still follow engine_idle_ttl_minutes. Off
743 # (default): the engine stops when the last lilbee process exits, leaving
744 # the machine clean.
745 keep_engine_warm: bool = ConfigField(default=False, writable=True)
747 # Hugging Face's high-performance transfer mode: more connections and much
748 # larger in-flight buffers. Off by default, those ceilings suit a server
749 # rather than a laptop also holding a model in memory.
750 fast_model_downloads: bool = ConfigField(default=False, writable=True)
752 # Idle minutes before the engine unloads its weights (llama-swap ttl), in
753 # every mode: even a persistent engine naps when unused. 0 keeps weights
754 # loaded until the engine stops.
755 engine_idle_ttl_minutes: int = ConfigField(default=5, writable=True)
757 # Working n_ctx the dynamic picker aims for. Default scales with
758 # total host RAM (see core.system.chat_ctx_target_for_total_bytes):
759 # <16 GiB -> 8192, 16-32 -> 12288, 32-64 -> 16384, 64-128 -> 24576,
760 # >=128 -> 65536 (an agent-capable window on server-class hosts).
761 # 8192 is the floor; the picker still clamps to training_ctx and
762 # host headroom.
763 chat_n_ctx_target: int = ConfigField(
764 default_factory=scaled_chat_ctx_target_default,
765 ge=512,
766 writable=True,
767 )
769 # Condense turns that outgrow chat_n_ctx_target into carried notes instead
770 # of dropping them. Off: zero model calls; the oldest turns drop and the
771 # context chip shows it. On: each firing blocks on a summarize call
772 # (measured: 1.3-2.5s per 60-turn fold on a datacenter GPU, 0.7-2s on an
773 # 8-core CPU with a 0.6B-4B model).
774 chat_compaction: bool = ConfigField(default=False, writable=True)
776 # Persist conversations and expose the Sessions drawer, tab, and commands.
777 # On by default; turning it off stops chats being written to disk, hides the
778 # ctrl+o binding from the footer, and gates the Sessions view behind a notice.
779 # Governs the human surfaces (TUI, HTTP, CLI); agent sessions have their own
780 # flag below, so the two domains the store already separates stay separate.
781 sessions_enabled: bool = ConfigField(default=True, writable=True)
783 # The agent (MCP) half of the same feature, off by default: agent hosts
784 # generally track their own conversation history, and the seven session
785 # tools cost schema on every request whether or not anything uses them.
786 mcp_sessions_enabled: bool = ConfigField(default=False, writable=True)
788 # Explicit ceiling for the dynamic n_ctx picker. ``None`` (default)
789 # lets the model's training_ctx from GGUF metadata be the ceiling,
790 # so a 128K-context model can reach for it on a host with the RAM
791 # to back it. Set explicitly to cap below the model's training_ctx.
792 num_ctx_max: int | None = ConfigField(default=None, ge=512, writable=True)
794 # Flash attention. None (default) = on, True = force on, False = off
795 # for backends or models where it misbehaves.
796 # Resolves the 'padding V cache to 1024' warning on models with
797 # uneven per-layer V dims (e.g. Gemma3) and saves ~25% KV memory.
798 flash_attention: bool | None = ConfigField(default=None, writable=True)
800 # KV cache element type. q8_0 (default) halves cache memory vs f16
801 # with no measurable quality loss for chat; q4_0 quarters it with a
802 # small quality cost. Both require flash attention to be enabled.
803 kv_cache_type: KvCacheType = ConfigField(default=KvCacheType.Q8_0, writable=True)
805 # Number of model layers to offload to GPU. None (default) = all
806 # layers, 0 = CPU only, positive int = partial offload. Useful when a
807 # discrete GPU has less VRAM than the model needs.
808 n_gpu_layers: int | None = ConfigField(default=None, writable=True)
810 # Keep a MoE model's expert weights in system memory, attention and shared
811 # layers on the GPU. Lets a sparse model run on a card too small to hold it.
812 # No effect on dense models, which have no expert tensors.
813 cpu_moe: bool = ConfigField(default=False, writable=True)
815 # Offload only the first N layers' experts. Takes precedence over cpu_moe;
816 # a smaller N keeps more of the model resident.
817 n_cpu_moe: int | None = ConfigField(default=None, writable=True)
819 # GPU device picker for dual-GPU machines (typical laptop case:
820 # discrete NVIDIA + integrated Intel/AMD). The Vulkan backend
821 # enumerates every adapter the system exposes and may pick the
822 # integrated one first, producing stalls or OOMs that look like
823 # llama.cpp bugs. Setting ``gpu_devices`` constrains visibility
824 # before the servers spawn, pinning inference to the chosen device(s).
825 #
826 # Accepts a comma-separated list of device indexes ("0", "1",
827 # "0,1") and applies it to every backend simultaneously:
828 # ``GGML_VK_VISIBLE_DEVICES`` for Vulkan, ``CUDA_VISIBLE_DEVICES``
829 # for CUDA, ``HIP_VISIBLE_DEVICES`` / ``ROCR_VISIBLE_DEVICES`` for
830 # ROCm. Setting one variable that the active backend ignores is
831 # harmless, so we set all four rather than detecting the build.
832 #
833 # Must be set before the first llama.cpp call; in practice that
834 # means via ``LILBEE_GPU_DEVICES`` or ``config.toml`` (TUI edits
835 # only take effect after a restart). ``None`` (default) hands off
836 # to the autodetect in ``providers/fleet/gpu_select.py``,
837 # which parses ``vulkaninfo --summary`` and pins the discrete
838 # adapter when one is present. The autodetect is silent on failure
839 # (no vulkaninfo, single device, parse error), leaving the
840 # Vulkan-loader's default ordering in place.
841 gpu_devices: str | None = ConfigField(default=None, writable=True)
843 # Primary GPU index passed to ``Llama(main_gpu=...)``. Only matters
844 # when multiple devices remain visible after ``gpu_devices``; with
845 # a single visible device, llama.cpp ignores this. ``None``
846 # (default) lets llama.cpp pick (index 0).
847 main_gpu: int | None = ConfigField(default=None, writable=True)
849 # Manual GPU placement override stored as a JSON scalar (the config.toml store
850 # is flat, and core must not depend on the provider PlacementSpec type). When
851 # set, it fully replaces the automatic placement planner: each active role pins
852 # to the listed device indices, with an optional tensor_split and replica count.
853 # Edited via the placement CLI/MCP/HTTP/TUI surfaces rather than the generic
854 # settings list, so public=False. None hands off to the VRAM-aware auto planner.
855 placement: str | None = ConfigField(
856 default=None,
857 writable=True,
858 public=False,
859 description=(
860 "Manual multi-GPU placement spec. It fully replaces the automatic planner: "
861 "each active role pins to the listed device indices. Edit it with the "
862 "placement commands, not the settings list. Empty uses the VRAM-aware planner"
863 ),
864 )
866 # Allow PUT/DELETE /api/placement to apply or clear placement over HTTP.
867 # Off by default because applying placement restarts the shared fleet's moved roles, which
868 # is unsafe across concurrent HTTP clients. Turn it on (LILBEE_ALLOW_HTTP_PLACEMENT=1)
869 # only for a single-client / owned deployment: the plugin's managed local
870 # server, or a personally-owned pod where one operator runs `lilbee serve`.
871 allow_http_placement: bool = Field(
872 default=False,
873 description=(
874 "Let PUT and DELETE /api/placement change model placement over HTTP. Off by "
875 "default because applying placement restarts the moved roles, which is unsafe "
876 "with concurrent clients. Turn it on only for a deployment you alone use"
877 ),
878 )
880 # True = Markdown widget for chat; False = plain Static (faster).
881 markdown_rendering: bool = Field(
882 default=True,
883 description=(
884 "Render chat replies as Markdown in the TUI. Off draws plain text, which is faster"
885 ),
886 )
888 # TUI theme name; persists the last Ctrl+T pick across sessions.
889 theme: str = ConfigField(default="rose-pine", writable=True)
891 # Per-model generation defaults set via apply_model_defaults().
892 _model_defaults: Any = None
894 # Wiki layer. LLM-maintained synthesis pages with citation provenance.
895 # Off by default; flip to True (or set LILBEE_WIKI=1) to enable. When off,
896 # the Wiki view tab and the chat ModelBar's scope picker are both hidden.
897 wiki: bool = ConfigField(default=False, writable=True)
898 # Whether a sync regenerates touched wiki pages on its own. Off by
899 # default: enabling the wiki never starts generating by itself, the
900 # user wikifies explicitly via `lilbee wiki build` / `wiki update`.
901 wiki_auto_update: bool = ConfigField(default=False, writable=True)
902 # Read-only: changing the directory at runtime strands prior wiki pages
903 # under the old path. Users who want a different location set it via
904 # LILBEE_WIKI_DIR / config.toml before the first wiki_build.
905 wiki_dir: str = "wiki"
906 wiki_prune_raw: bool = ConfigField(default=False, writable=True)
908 # Minimum cosine similarity between a page body and the mean of its
909 # source chunk vectors before a page is published (below → drafts).
910 # Replaces the old LLM-based faithfulness score: mean-of-chunks is a
911 # deterministic, zero-LLM-call signal that routes topic-drifted
912 # pages to drafts without the 0.0 to 1.0 ambiguity of a model-emitted
913 # number. Tuning knob: swap to per-chunk max or top-K-mean if the
914 # default 0.5 produces false drafts.
915 wiki_embedding_faithfulness_threshold: float = ConfigField(
916 default=0.5, ge=0.0, le=1.0, writable=True
917 )
919 # Per-call output token cap for wiki generation. Without this a
920 # reasoning model (Qwen3, DeepSeek-R1) can burn the full context
921 # window emitting <think> tokens before the actual answer, taking
922 # minutes per page. Default leaves headroom for a typical reasoning
923 # budget plus a real response (~1000 output + ~1000 slack).
924 wiki_summary_max_tokens: int = ConfigField(default=2048, ge=256, writable=True)
926 # Wiki generation is a structured-output task: the model must emit the
927 # block separators, the citation footnotes, and verbatim quotes. The
928 # usual chat default (~0.8) is too creative for that. Lowering the
929 # sampling temperature makes the model stick to the template and quote
930 # more faithfully. 0.1 leaves just enough slack to avoid hard loops.
931 wiki_temperature: float = ConfigField(default=0.1, ge=0.0, le=2.0, writable=True)
933 # Fraction of citations that must be stale before a wiki page is flagged.
934 wiki_stale_citation_threshold: float = ConfigField(default=0.5, ge=0.0, le=1.0, writable=True)
936 # Fraction of content changed that triggers human-review drift guard.
937 wiki_drift_threshold: float = ConfigField(default=0.3, ge=0.0, le=1.0, writable=True)
939 # LLM prompt templates for wiki page generation: wiki_synthesis_prompt
940 # for cross-source synthesis pages, wiki_entity_batch_prompt (below)
941 # for the per-source batched call. Writable so advanced users can
942 # override them from /settings, config.toml, or ``LILBEE_WIKI_*_PROMPT``
943 # env vars. Templates must keep the expected ``{placeholders}``. If you
944 # remove one the generator will crash on first use.
945 wiki_synthesis_prompt: str = ConfigField(
946 writable=True,
947 default=(
948 "You are a knowledge compiler. Given source chunks from MULTIPLE documents "
949 "about related concepts, write a synthesis wiki page in markdown that connects "
950 "ideas across sources.\n\n"
951 "Rules:\n"
952 "1. Every factual claim MUST have an inline citation [^src1], [^src2], etc. "
953 "Never cite by chunk label: [Chunk N] labels only organize the "
954 "chunks below and must not appear in the page.\n"
955 "2. Cite the EXACT text from the source that supports each claim by quoting it.\n"
956 "3. For connections, interpretations, or patterns you identify across sources, "
957 "mark with [*inference*].\n"
958 "4. Use blockquotes (>) for directly cited facts.\n"
959 "5. Reference each source by its filename when drawing connections.\n"
960 "6. End with a citation block in this format:\n\n"
961 "---\n"
962 "<!-- citations (auto-generated from _citations table -- do not edit) -->\n"
963 '[^src1]: {{source_name}}, excerpt: "exact quoted text"\n'
964 '[^src2]: {{source_name}}, excerpt: "exact quoted text"\n\n'
965 "Topic: {topic}\n\n"
966 "Sources:\n{source_list}\n\n"
967 "Chunks:\n{chunks_text}\n\n"
968 "Write the synthesis page now. Start with a heading."
969 ),
970 )
972 # Wiki synthesis clusterer backend. CONCEPTS requires the [graph] extra
973 # and falls back to EMBEDDING when unavailable.
974 wiki_clusterer: ClustererBackend = ConfigField(
975 default=ClustererBackend.EMBEDDING, writable=True
976 )
978 # Neighborhood size for the mutual-kNN graph. 0 = auto-scale from corpus size.
979 wiki_clusterer_k: int = ConfigField(default=0, ge=0, writable=True)
981 # LazyGraphRAG-style concept graph. Requires the [graph] extra.
982 concept_graph: bool = ConfigField(default=True, writable=True)
984 # Weight of concept overlap boost relative to vector similarity.
985 concept_boost_weight: float = ConfigField(default=0.3, ge=0.0, le=1.0, writable=True)
987 # Max noun-phrase concepts extracted per chunk.
988 concept_max_per_chunk: int = ConfigField(default=5, ge=1, writable=True)
990 # spaCy NER labels kept by the wiki entity extractor. Anything not
991 # in this set (QUANTITY, CARDINAL, DATE, TIME, MONEY, PERCENT,
992 # ORDINAL, ...) is dropped before aggregation. Override via
993 # LILBEE_CONCEPT_ALLOWED_ENT_TYPES as a comma-separated list.
994 concept_allowed_ent_types: frozenset[str] = Field(
995 default=DEFAULT_ALLOWED_NER_LABELS,
996 description=(
997 "spaCy NER labels the wiki entity extractor keeps. It drops everything else "
998 "(QUANTITY, CARDINAL, DATE, and so on) before aggregation"
999 ),
1000 )
1002 # Strategy used to extract entities for the concept/entity wiki.
1003 # NER_ENTITIES (default) pulls typed NER entities with spaCy; concept
1004 # pages are proposed by the LLM inside the per-source batched call,
1005 # not by the extractor. NER_CONCEPTS_PLUS_LLM_TYPES layers an
1006 # LLM-proposed domain schema on top. LLM_TAGGED asks the LLM to tag
1007 # every chunk (most expensive). Unimplemented modes fall back to
1008 # NER_ENTITIES.
1009 wiki_entity_mode: WikiEntityMode = ConfigField(
1010 default=WikiEntityMode.NER_ENTITIES, writable=True
1011 )
1013 # Minimum distinct chunk mentions before an entity or concept earns
1014 # its own wiki page. Filters one-off noise.
1015 wiki_entity_min_mentions: int = ConfigField(default=3, ge=1, writable=True)
1016 wiki_stub_max_chunk_refs: int = ConfigField(default=50, ge=1, writable=True)
1018 # Auto-update cap: if a single sync touches more than this many
1019 # concept or entity pages, skip the per-slug regeneration and tell
1020 # the user to run `lilbee wiki update` explicitly. Keeps a surprise
1021 # bulk import from firing hundreds of LLM calls.
1022 wiki_ingest_update_cap: int = ConfigField(default=20, ge=1, writable=True)
1024 # Whether the per-source batched call asks the LLM to curate
1025 # concept pages alongside the pre-extracted entity list. False →
1026 # entity sections only, no concept curation (incremental ingest
1027 # path uses this to avoid churning concept slugs per source-touch).
1028 wiki_extract_concepts: bool = ConfigField(default=True, writable=True)
1030 # Minimum chunk count a source must contribute before it is eligible
1031 # for concept curation. Sources below the floor still get a batched
1032 # call when they have entities (the prompt writes entity-only
1033 # sections); sources below the floor with zero entities are skipped
1034 # entirely. Prevents boilerplate / TOC / appendix documents from
1035 # burning an LLM call to invent "concepts".
1036 wiki_batch_min_chunks: int = ConfigField(default=3, ge=1, writable=True)
1038 # Prompt template for the per-source batched call. Placeholders:
1039 # {source}, {entity_list}, {chunks_text}, {concept_instruction}.
1040 # {concept_instruction} is filled with a concept-curation paragraph
1041 # when concepts are requested, or the empty string otherwise.
1042 # Single-entity page written on demand from that entity's chunks across
1043 # every source naming it. The batched prompt above cannot serve this: it
1044 # writes every section for one source in one call.
1045 wiki_entity_page_prompt: str = ConfigField(
1046 writable=True,
1047 default=(
1048 "You are a knowledge compiler. Given source chunks that mention "
1049 "ONE subject, write a wiki page about that subject in markdown.\n\n"
1050 "Rules:\n"
1051 "1. Every factual claim MUST have an inline citation [^src1], [^src2], etc. "
1052 "Never cite by chunk label: [Chunk N] labels only organize the "
1053 "chunks below and must not appear in the page.\n"
1054 "2. Cite the EXACT text from the source that supports each claim by quoting it.\n"
1055 "3. Write only what the chunks support. Mark anything you infer with "
1056 "[*inference*].\n"
1057 "4. Use blockquotes (>) for directly cited facts.\n"
1058 "5. When sources disagree, say so and cite both.\n"
1059 "6. End with a citation block in this format:\n\n"
1060 "---\n"
1061 "<!-- citations (auto-generated from _citations table -- do not edit) -->\n"
1062 '[^src1]: {{source_name}}, excerpt: "exact quoted text"\n'
1063 '[^src2]: {{source_name}}, excerpt: "exact quoted text"\n\n'
1064 "Subject: {topic}\n\n"
1065 "Sources:\n{source_list}\n\n"
1066 "Chunks:\n{chunks_text}\n\n"
1067 "Write the page now. Start with a heading naming the subject."
1068 ),
1069 )
1070 wiki_entity_batch_prompt: str = ConfigField(
1071 writable=True,
1072 default=(
1073 "You are writing wiki sections based on these chunks from {source}.\n\n"
1074 "{concept_instruction}"
1075 "Write a wiki section for each of these NER ENTITIES: {entity_list}\n\n"
1076 "Format each section exactly as:\n"
1077 "## Name\n"
1078 "{{content with [^src1]-style citations}}\n\n"
1079 "Rules:\n"
1080 "1. Every factual claim MUST have an inline citation [^src1], [^src2], etc. "
1081 "Never cite by chunk label: [Chunk N] labels only organize the "
1082 "chunks below and must not appear in the page.\n"
1083 "2. Cite the EXACT text from the source that supports each claim by quoting it.\n"
1084 "3. For interpretations or connections not directly stated, mark with [*inference*].\n"
1085 "4. Use blockquotes (>) for directly cited facts.\n"
1086 "5. End the response with a citation block in this format:\n\n"
1087 "---\n"
1088 "<!-- citations (auto-generated from _citations table -- do not edit) -->\n"
1089 '[^src1]: {{source_name}}, excerpt: "exact quoted text"\n'
1090 '[^src2]: {{source_name}}, excerpt: "exact quoted text"\n\n'
1091 "Source chunks:\n{chunks_text}\n"
1092 ),
1093 )
1095 # Class variable: not a settings field
1096 _toml_cache: ClassVar[dict[str, Any]] = {}
1098 @field_validator("lilbee_name", mode="after")
1099 @classmethod
1100 def _strip_lilbee_name(cls, value: str) -> str:
1101 """Strip whitespace; an empty string signals 'use the path-derived label'."""
1102 return value.strip()
1104 @field_validator(
1105 "temperature",
1106 "top_p",
1107 "repeat_penalty",
1108 "top_k_sampling",
1109 "num_ctx",
1110 "seed",
1111 mode="before",
1112 )
1113 @classmethod
1114 def _empty_string_to_none(cls, v: Any) -> Any:
1115 if isinstance(v, str) and v.strip() == "":
1116 return None
1117 return v
1119 @field_validator("chat_mode", mode="before")
1120 @classmethod
1121 def _normalize_chat_mode(cls, v: Any) -> ChatMode:
1122 """Coerce chat_mode to a ChatMode value; default ChatMode.SEARCH."""
1123 if v is None or v == "":
1124 return ChatMode.SEARCH
1125 candidate = str(v).strip().lower()
1126 try:
1127 return ChatMode(candidate)
1128 except ValueError as exc:
1129 valid = ", ".join(repr(m.value) for m in ChatMode)
1130 raise ValueError(f"chat_mode must be one of {{{valid}}}, got {v!r}") from exc
1132 @field_validator("enable_ocr", mode="before")
1133 @classmethod
1134 def _parse_enable_ocr(cls, v: Any) -> bool | None:
1135 """Parse enable_ocr from env var string or direct value.
1137 Accepts: true/false/1/0/yes/no (case-insensitive), empty string
1138 or None for auto-detect.
1139 """
1140 if v is None:
1141 return None
1142 if isinstance(v, bool):
1143 return v
1144 if isinstance(v, str):
1145 if v.strip().lower() in ("", "auto", "none"):
1146 return None
1147 try:
1148 return parse_bool(v)
1149 except ValueError:
1150 # bool() on a non-empty string is True, so falling through here
1151 # turned an unparseable value into "on". Warn and auto-detect,
1152 # matching the sibling validators.
1153 log.warning("Invalid LILBEE_ENABLE_OCR=%r, using auto", v)
1154 return None
1155 return bool(v)
1157 @field_validator("ocr_language", mode="before")
1158 @classmethod
1159 def _parse_ocr_language(cls, v: Any) -> list[str]:
1160 """Accept a list or a ``+``/comma/newline-separated string; never empty.
1162 Tesseract joins languages with ``+`` (e.g. ``eng+deu``), so that is the
1163 canonical user-facing form. Commas are also accepted. Newlines are
1164 accepted because ``app.settings`` joins list values with ``\\n`` when it
1165 persists them to config.toml; without splitting on it a multi-language
1166 value would reload as one malformed token. Blank input falls back to
1167 English, since xberg errors on an empty list.
1168 """
1169 if isinstance(v, str):
1170 v = v.replace("+", ",").replace("\n", ",").split(",")
1171 items = v or []
1172 langs = [s.strip() for s in items if isinstance(s, str) and s.strip()]
1173 for lang in langs:
1174 if not _TESSERACT_LANGUAGE_CODE.fullmatch(lang):
1175 raise ValueError(
1176 f"ocr_language: {lang!r} is not a Tesseract language code "
1177 "(examples: eng, deu, chi_sim, jpn_vert)"
1178 )
1179 return langs or ["eng"]
1181 @field_validator("force_ocr_pages", mode="before")
1182 @classmethod
1183 def _parse_force_ocr_pages(cls, v: Any) -> list[int]:
1184 """Accept a list, a string or one int; split string items on commas and newlines.
1186 Newlines are accepted because ``app.settings`` joins list values with
1187 ``\\n`` when it persists them to config.toml. Returns sorted unique pages.
1188 """
1189 items = v if isinstance(v, list) else [v]
1190 parts = [p for item in items for p in _split_page_item(item)]
1191 return sorted({_page_number(part) for part in parts if str(part).strip()})
1193 @field_validator("flash_attention", mode="before")
1194 @classmethod
1195 def _parse_flash_attention(cls, v: Any) -> bool | None:
1196 """Auto/on/off tri-state: empty/auto/none -> None, else parse bool."""
1197 if v is None:
1198 return None
1199 if isinstance(v, bool):
1200 return v
1201 if isinstance(v, str):
1202 if v.strip().lower() in ("", "auto", "none"):
1203 return None
1204 try:
1205 return parse_bool(v)
1206 except ValueError:
1207 log.warning("Invalid flash_attention=%r, using auto", v)
1208 return None
1209 return bool(v)
1211 @field_validator("n_gpu_layers", mode="before")
1212 @classmethod
1213 def _parse_n_gpu_layers(cls, v: Any) -> int | None:
1214 """Auto -> None, ``cpu`` alias -> 0, integers parsed verbatim."""
1215 if v is None:
1216 return None
1217 if isinstance(v, str):
1218 label = v.strip().lower()
1219 if label in ("", "auto", "none"):
1220 return None
1221 if label == "cpu":
1222 return 0
1223 try:
1224 return int(label)
1225 except ValueError:
1226 log.warning("Invalid LILBEE_N_GPU_LAYERS=%r, using auto", v)
1227 return None
1228 return int(v)
1230 @field_validator("main_gpu", mode="before")
1231 @classmethod
1232 def _parse_main_gpu(cls, v: Any) -> int | None:
1233 """Empty/auto strings -> None, integers parsed verbatim."""
1234 if v is None:
1235 return None
1236 if isinstance(v, str):
1237 label = v.strip().lower()
1238 if label in ("", "auto", "none"):
1239 return None
1240 try:
1241 return int(label)
1242 except ValueError:
1243 log.warning("Invalid LILBEE_MAIN_GPU=%r, using auto", v)
1244 return None
1245 return int(v)
1247 @field_validator("gpu_devices", mode="before")
1248 @classmethod
1249 def _parse_gpu_devices(cls, v: Any) -> str | None:
1250 """Normalize device list: strip whitespace, drop empties, keep order."""
1251 if v is None:
1252 return None
1253 if isinstance(v, str):
1254 label = v.strip().lower()
1255 if label in ("", "auto", "all", "none"):
1256 return None
1257 parts = [p.strip() for p in v.split(",") if p.strip()]
1258 if not parts:
1259 return None
1260 for part in parts:
1261 if not part.lstrip("-").isdigit():
1262 log.warning("Invalid LILBEE_GPU_DEVICES=%r, ignoring", v)
1263 return None
1264 return ",".join(parts)
1265 return str(v)
1267 @field_validator("placement", mode="before")
1268 @classmethod
1269 def _parse_placement(cls, v: Any) -> str | None:
1270 """Blank/None -> None; validate a JSON string or PlacementSpec; store JSON."""
1271 from lilbee.providers.fleet.placement_spec import PlacementError, PlacementSpec
1273 if v is None:
1274 return None
1275 if isinstance(v, PlacementSpec):
1276 json_str = v.to_json()
1277 PlacementSpec.from_json(json_str) # re-validate a directly-built spec
1278 return json_str
1279 if isinstance(v, str):
1280 if v.strip() == "":
1281 return None
1282 PlacementSpec.from_json(v)
1283 return v
1284 raise PlacementError("placement must be a JSON string or PlacementSpec")
1286 @field_validator("semantic_chunking", mode="before")
1287 @classmethod
1288 def _parse_semantic_chunking(cls, v: Any) -> bool:
1289 """Parse from env string; invalid values warn and fall back to False."""
1290 if isinstance(v, bool):
1291 return v
1292 if isinstance(v, str):
1293 try:
1294 return parse_bool(v)
1295 except ValueError:
1296 log.warning("Invalid LILBEE_SEMANTIC_CHUNKING=%r, using default False", v)
1297 return False
1298 return bool(v)
1300 @field_validator(
1301 "chat_model", "embedding_model", "vision_model", "reranker_model", mode="after"
1302 )
1303 @classmethod
1304 def _normalize_model_tag(cls, v: str, info: ValidationInfo) -> str:
1305 """Validate and canonicalize a model ref; blank means the role is unconfigured."""
1306 if not v or not v.strip():
1307 return ""
1308 from lilbee.providers.model_ref import parse_model_ref
1310 return parse_model_ref(v).for_openai_prefix()
1312 @field_validator("ollama_base_url", "lm_studio_base_url", mode="after")
1313 @classmethod
1314 def _strip_trailing_slash(cls, v: str) -> str:
1315 """Canonicalize a local-server URL once at the write boundary."""
1316 return v.rstrip("/")
1318 @field_validator("cors_origins", mode="before")
1319 @classmethod
1320 def _split_cors_origins(cls, v: Any) -> Any:
1321 if isinstance(v, str):
1322 return [o.strip() for o in v.split(",") if o.strip()]
1323 return v
1325 @field_validator("crawl_browser_extra_args", mode="before")
1326 @classmethod
1327 def _split_crawl_browser_extra_args(cls, v: Any) -> Any:
1328 """Accept a newline-separated string, matching how the field is persisted.
1330 ``app.settings`` joins list values with newlines before writing them to
1331 ``config.toml`` as a scalar string. Without this inverse, reload cannot
1332 coerce that string to ``list[str]`` and the whole config.toml is dropped.
1333 TOML lists and JSON arrays pass through unchanged.
1334 """
1335 if isinstance(v, str):
1336 return [a.strip() for a in v.splitlines() if a.strip()]
1337 return v
1339 @field_validator("crawl_exclude_patterns", mode="before")
1340 @classmethod
1341 def _split_crawl_exclude_patterns(cls, v: Any) -> Any:
1342 """Accept newline-separated strings from env vars / plain-text config.
1344 Regex commonly uses commas (e.g. `{2,4}`) and pipes (alternation), so
1345 newline is the only separator safe to use for this field. TOML lists
1346 and JSON arrays pass through unchanged.
1347 """
1348 if isinstance(v, str):
1349 return [p.strip() for p in v.splitlines() if p.strip()]
1350 return v
1352 @field_validator("crawl_exclude_patterns", mode="after")
1353 @classmethod
1354 def _validate_crawl_exclude_patterns(cls, v: list[str]) -> list[str]:
1355 """Reject any entry that isn't a valid Python regex.
1357 These patterns are compiled at crawl time. An invalid pattern there
1358 surfaces as an opaque mid-crawl error; catching it at PATCH time gives
1359 the user a 400 with a pointer to the bad entry.
1360 """
1361 import re
1363 bad: list[str] = []
1364 for i, pattern in enumerate(v):
1365 try:
1366 re.compile(pattern)
1367 except re.error as exc:
1368 bad.append(f"[{i}] {pattern!r}: {exc}")
1369 if bad:
1370 raise ValueError("invalid regex in crawl_exclude_patterns:\n " + "\n ".join(bad))
1371 return v
1373 @field_validator("ignore_dirs", mode="before")
1374 @classmethod
1375 def _merge_ignore_dirs(cls, v: Any) -> frozenset[str]:
1376 if isinstance(v, str):
1377 extra = frozenset(name.strip() for name in v.split(",") if name.strip())
1378 return DEFAULT_IGNORE_DIRS | extra
1379 if isinstance(v, (set, frozenset, list)):
1380 return DEFAULT_IGNORE_DIRS | frozenset(v)
1381 return DEFAULT_IGNORE_DIRS
1383 @field_validator("concept_allowed_ent_types", mode="before")
1384 @classmethod
1385 def _parse_ent_types(cls, v: Any) -> frozenset[str]:
1386 """Replace-semantics override: a narrowed set is used as-is,
1387 not unioned with defaults. A user asking for ``PERSON,ORG``
1388 wants exactly those kinds. Accepts comma-separated strings
1389 from env and list / set / frozenset from code. Empty input
1390 falls back to :data:`DEFAULT_ALLOWED_NER_LABELS` so an empty
1391 env var does not silently disable the gate.
1392 """
1393 if isinstance(v, str):
1394 parts = frozenset(name.strip().upper() for name in v.split(",") if name.strip())
1395 return parts or DEFAULT_ALLOWED_NER_LABELS
1396 if isinstance(v, (set, frozenset, list)):
1397 parts = frozenset(str(x).upper() for x in v)
1398 return parts or DEFAULT_ALLOWED_NER_LABELS
1399 return DEFAULT_ALLOWED_NER_LABELS
1401 @model_validator(mode="before")
1402 @classmethod
1403 def _resolve_defaults(cls, data: Any) -> Any:
1404 from lilbee.core.system import (
1405 canonical_data_root,
1406 canonical_models_dir,
1407 default_data_dir,
1408 find_local_root,
1409 )
1411 if not isinstance(data, dict):
1412 return data
1414 # An empty LILBEE_DATA_ROOT (delivered as "") must fall through to default
1415 # resolution like an unset one, not become Path(".") = the process cwd.
1416 if isinstance(data.get("data_root"), str) and not data["data_root"].strip():
1417 data["data_root"] = None
1418 if data.get("data_root") in (None, _UNSET_PATH):
1419 data_env = os.environ.get("LILBEE_DATA", "").strip()
1420 if data_env:
1421 data["data_root"] = Path(data_env)
1422 else:
1423 local = find_local_root()
1424 data["data_root"] = local if local is not None else default_data_dir()
1425 # Every child path below derives from this, and the server lock keys on
1426 # those, so canonicalizing here is what makes one directory key one lock.
1427 # Also coerces a raw string (LILBEE_DATA_ROOT) to Path.
1428 root = canonical_data_root(data["data_root"])
1429 data["data_root"] = root
1430 if data.get("documents_dir") in (None, _UNSET_PATH):
1431 data["documents_dir"] = root / "documents"
1432 if data.get("data_dir") in (None, _UNSET_PATH):
1433 data["data_dir"] = root / "data"
1434 if data.get("lancedb_dir") in (None, _UNSET_PATH):
1435 data["lancedb_dir"] = root / "data" / "lancedb"
1436 if data.get("models_dir") in (None, _UNSET_PATH):
1437 data["models_dir"] = canonical_models_dir()
1439 return data
1441 @classmethod
1442 def settings_customise_sources(
1443 cls,
1444 settings_cls: type[BaseSettings],
1445 init_settings: Any,
1446 env_settings: Any,
1447 dotenv_settings: Any,
1448 file_secret_settings: Any,
1449 ) -> tuple[Any, ...]:
1450 from lilbee.core.system import canonical_data_root, default_data_dir, find_local_root
1452 # .strip() to match _resolve_defaults; a padded value would otherwise
1453 # send the root and its config.toml to different directories.
1454 data_env = os.environ.get("LILBEE_DATA", "").strip()
1455 if data_env:
1456 toml_dir = Path(data_env)
1457 else:
1458 local = find_local_root()
1459 toml_dir = local if local else default_data_dir()
1460 # Same call as the root itself, so this looks where the root resolves to;
1461 # a "~/lilbee" value would otherwise search a literal ./~ and find nothing.
1462 toml_path = canonical_data_root(toml_dir) / CONFIG_FILE_NAME
1464 plain_env = _PlainEnvSource(settings_cls, env_prefix="LILBEE_")
1465 sources: list[Any] = [init_settings, plain_env]
1466 if toml_path.exists() and os.environ.get("LILBEE_SKIP_TOML_CONFIG") != "1":
1467 sources.append(_TomlSource(settings_cls, toml_path))
1468 return tuple(sources)
1470 @property
1471 def model_defaults(self) -> Any:
1472 """Per-model generation defaults (read-only). Set via apply_model_defaults()."""
1473 return self._model_defaults
1475 def apply_model_defaults(self, defaults: Any) -> None:
1476 """Store per-model generation defaults for 3-layer merge."""
1477 object.__setattr__(self, "_model_defaults", defaults)
1479 def clear_model_defaults(self) -> None:
1480 """Reset per-model defaults to None."""
1481 object.__setattr__(self, "_model_defaults", None)
1483 def generation_options(self, **overrides: Any) -> dict[str, Any]:
1484 """Merge model defaults, user config, and per-call overrides, dropping None."""
1485 result = _model_defaults_dict(self._model_defaults)
1486 # One name for the output cap inside lilbee; the provider translators rename it.
1487 if "max_tokens" in result:
1488 result["num_predict"] = result.pop("max_tokens")
1489 user_fields: dict[str, Any] = {
1490 "temperature": self.temperature,
1491 "top_p": self.top_p,
1492 "top_k": self.top_k_sampling,
1493 "repeat_penalty": self.repeat_penalty,
1494 "num_ctx": self.num_ctx,
1495 "seed": self.seed,
1496 "num_predict": self.max_tokens,
1497 }
1498 for k, v in user_fields.items():
1499 if v is not None:
1500 result[k] = v
1501 for k, v in overrides.items():
1502 if v is not None:
1503 result[k] = v
1504 return result
1507def _model_defaults_dict(defaults: Any) -> dict[str, Any]:
1508 """Non-None fields of a ModelDefaults instance as a dict."""
1509 if defaults is None:
1510 return {}
1511 from dataclasses import fields as dc_fields
1513 return {
1514 f.name: getattr(defaults, f.name)
1515 for f in dc_fields(defaults)
1516 if getattr(defaults, f.name) is not None
1517 }
1520class _PlainEnvSource:
1521 """Reads LILBEE_* env vars as plain strings so field validators handle parsing."""
1523 def __init__(self, settings_cls: type[BaseSettings], env_prefix: str) -> None:
1524 self._prefix = env_prefix
1525 self._fields = set(settings_cls.model_fields)
1527 def __call__(self) -> dict[str, Any]:
1528 result: dict[str, Any] = {}
1529 for field_name in self._fields:
1530 raw = os.environ.get(f"{self._prefix}{field_name.upper()}")
1531 if value_is_set(field_name, raw):
1532 result[field_name] = raw
1533 return result
1536class _TomlSource:
1537 """Custom pydantic-settings source that reads config.toml."""
1539 def __init__(self, settings_cls: type[BaseSettings], path: Path) -> None:
1540 self._path = path
1542 def __call__(self) -> dict[str, Any]:
1543 import tomllib
1545 try:
1546 with self._path.open("rb") as f:
1547 data = tomllib.load(f)
1548 except (ValueError, OSError):
1549 log.warning("Failed to read %s, ignoring", self._path)
1550 return {}
1551 # An empty string is unset (the field default applies, since pydantic
1552 # cannot coerce "" to int|None), except on a clearable model role,
1553 # where it clears the model. TOML's native types pass through as-is.
1554 return {k: v for k, v in data.items() if value_is_set(k, v)}
1557def _build_cfg() -> tuple[Config, Exception | None]:
1558 """Build cfg; on stale-config validation failure, fall back to defaults.
1560 A persisted ``config.toml`` from before a breaking schema change can
1561 contain values the new validators reject. Crashing at module import
1562 means every command (``lilbee --help`` included) emits a Python
1563 traceback. Falling back to env+defaults lets the package load; the
1564 CLI / TUI surfaces the original error before doing real work.
1565 """
1566 try:
1567 return Config(), None
1568 except Exception as exc:
1569 os.environ["LILBEE_SKIP_TOML_CONFIG"] = "1"
1570 try:
1571 return Config(), exc
1572 finally:
1573 os.environ.pop("LILBEE_SKIP_TOML_CONFIG", None)
1576cfg, config_load_error = _build_cfg()
1578# Canonicalize LILBEE_DATA at the cfg.data_root resolution boundary so
1579# spawn-context worker subprocesses inherit the same data root.
1580# ``setdefault`` preserves a user-set value.
1581os.environ.setdefault("LILBEE_DATA", str(cfg.data_root))