Coverage for src/lilbee/core/config/model.py: 100%

545 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""The :class:`Config` dataclass and the ``cfg`` singleton. 

2 

3The settings sources, TOML parser, and the resilient builder that falls 

4back to defaults on stale-config validation failures live here too. Every 

5``from lilbee.core.config import cfg`` resolves through ``lilbee.core.config.__init__`` 

6to the same instance defined at module bottom. 

7""" 

8 

9import logging 

10import os 

11import re 

12from pathlib import Path 

13from typing import Any, ClassVar 

14 

15from pydantic import Field, ValidationInfo, field_validator, model_validator 

16from pydantic_settings import BaseSettings, SettingsConfigDict 

17 

18from lilbee.core.system import scaled_chat_ctx_target_default 

19 

20from .defaults import ( 

21 CONFIG_FILE_NAME, 

22 DEFAULT_ALLOWED_NER_LABELS, 

23 DEFAULT_CORS_ORIGIN_REGEX, 

24 DEFAULT_CRAWL_EXCLUDE_PATTERNS, 

25 DEFAULT_GENERAL_SYSTEM_PROMPT, 

26 DEFAULT_IGNORE_DIRS, 

27 DEFAULT_RAG_SYSTEM_PROMPT, 

28) 

29from .enums import ( 

30 ChatMode, 

31 ClustererBackend, 

32 CrawlRenderMode, 

33 FtsLanguage, 

34 KvCacheType, 

35 LlmProvider, 

36 OcrPageStrategy, 

37 ReasoningMode, 

38 RerankerType, 

39 TableModel, 

40 WikiEntityMode, 

41) 

42from .parsing import parse_bool 

43from .validators import ConfigField 

44 

45log = logging.getLogger(__name__) 

46 

47# Sentinel for unset Path-typed fields. ``Field(default=Path())`` produces an 

48# instance equal to this, so the model_validator can distinguish "user passed 

49# the default" from "user explicitly set a value". 

50_UNSET_PATH = Path() 

51 

52# A Tesseract language code: ISO 639 letters plus script or orientation suffixes 

53# (eng, en, chi_sim, jpn_vert). xberg rejects anything else before extracting. 

54_TESSERACT_LANGUAGE_CODE = re.compile(r"[a-z]{2,3}(?:_[a-z]+)*") 

55 

56# Model roles that can be off. An empty LILBEE_<FIELD> or config.toml value 

57# clears one of these; on every other field an empty value counts as unset. 

58CLEARABLE_MODEL_FIELDS = frozenset({"vision_model", "reranker_model"}) 

59 

60 

61def value_is_set(field_name: str, raw: object) -> bool: 

62 """Whether an env or config.toml value is set: non-empty, or empty on a clearable model role.""" 

63 if raw is None: 

64 return False 

65 return raw != "" or field_name in CLEARABLE_MODEL_FIELDS 

66 

67 

68def _as_int(item: Any) -> int: 

69 """A force_ocr_pages entry as an int: an int, or a string of digits.""" 

70 if isinstance(item, str) and item.strip().lstrip("-").isdigit(): 

71 return int(item) 

72 # bool is an int subclass; True is not a page number. 

73 if isinstance(item, int) and not isinstance(item, bool): 

74 return item 

75 raise ValueError(f"force_ocr_pages: {item!r} is not a page number") 

76 

77 

78def _split_page_item(item: Any) -> list[Any]: 

79 """A string force_ocr_pages item split on commas and newlines; other items as-is.""" 

80 if isinstance(item, str): 

81 return item.replace("\n", ",").split(",") 

82 return [item] 

83 

84 

85def _page_number(item: Any) -> int: 

86 """One force_ocr_pages entry as a 1-indexed page.""" 

87 page = _as_int(item) 

88 if page < 1: 

89 raise ValueError(f"force_ocr_pages: page numbers start at 1 (got {page})") 

90 return page 

91 

92 

93class Config(BaseSettings): 

94 """Runtime configuration: one singleton instance, mutated by CLI overrides.""" 

95 

96 model_config = SettingsConfigDict( 

97 env_prefix="LILBEE_", 

98 validate_assignment=True, 

99 arbitrary_types_allowed=True, 

100 extra="ignore", 

101 ) 

102 

103 # Paths: resolved from env/defaults in model_validator(mode='before') 

104 data_root: Path = Field( 

105 default=Path(), 

106 description=( 

107 "Root directory for this library. Resolved at start from LILBEE_DATA, " 

108 "then a .lilbee/ directory walked up from the working directory, then " 

109 "the platform default. Every other path below hangs off it" 

110 ), 

111 ) 

112 # Writable so plugin-managed servers can pivot storage to a vault path on 

113 # first boot; rebuild the index after migrating. 

114 documents_dir: Path = ConfigField(default=Path(), writable=True) 

115 # External source roots ``add`` registered, mapping label -> absolute path. 

116 # lilbee indexes the files where they live (no copy, no symlink); the label 

117 # prefixes their source keys so a root at /data/corpus keys as ``corpus/…``. 

118 # Managed by ``add`` / ``remove``, so it is writable (persisted to 

119 # config.toml) but not surfaced in the settings UI. 

120 linked_roots: dict[str, str] = ConfigField( 

121 default_factory=dict, 

122 writable=True, 

123 public=False, 

124 description=( 

125 "External source roots that `add` registered, as label -> absolute path. " 

126 "`add` and `remove` maintain it; do not edit it by hand" 

127 ), 

128 ) 

129 data_dir: Path = Field( 

130 default=Path(), 

131 description="Directory holding the database. Defaults to data_root/data", 

132 ) 

133 lancedb_dir: Path = Field( 

134 default=Path(), 

135 description=( 

136 "Directory holding the LanceDB vector tables. Defaults to data_root/data/lancedb" 

137 ), 

138 ) 

139 models_dir: Path = Field( 

140 default=Path(), 

141 description=( 

142 "Directory holding downloaded model files. Shared across libraries, so a " 

143 "model pulled for one is available to all" 

144 ), 

145 ) 

146 # Markdown vault root; when set, search results carry a vault-relative 

147 # ``vault_path`` so a host UI can deep-link into the vault. 

148 vault_base: Path | None = ConfigField(default=None, writable=True) 

149 

150 # Human-readable label for the active lilbee. Empty falls back to 

151 # "global" for the platform default dir, otherwise the project path 

152 # (~-substituted and left-truncated to a hard cap). 

153 lilbee_name: str = ConfigField(default="", writable=True) 

154 # If True, the status bar pill shows the full absolute path: expands 

155 # "global" to the on-disk platform-default path and skips the 

156 # ~-substitution / left-truncation for project paths. Toggled by F4. 

157 show_lilbee_path: bool = ConfigField(default=False, writable=True) 

158 

159 # Whether an agent launcher (opencode, hermes) registers lilbee's MCP search 

160 # tool into the agent's config. Per-launch --mcp/--no-mcp overrides it. 

161 agent_mcp_enabled: bool = ConfigField(default=True, writable=True) 

162 

163 # Empty = not configured, same convention as vision_model. A fresh install 

164 # has no models; the catalog assigns these on the first download. 

165 chat_model: str = Field(default="") 

166 embedding_model: str = Field(default="") 

167 # Vision OCR model for scanned PDFs and image-only pages. Empty = disabled; 

168 # there is no cross-role fallback onto the chat model even if multimodal. 

169 vision_model: str = ConfigField(default="", public=True) 

170 embedding_dim: int = Field( 

171 default=768, 

172 ge=1, 

173 description=( 

174 "Vector width the index is built with. The embedding model sets it; " 

175 "change it only to match a model lilbee cannot introspect" 

176 ), 

177 ) 

178 chunk_size: int = ConfigField(default=512, ge=64, writable=True, reindex=True) 

179 chunk_overlap: int = ConfigField(default=100, ge=0, writable=True, reindex=True) 

180 # A file over this many chunks is skipped before embedding; 0 lifts the ceiling. 

181 max_chunks_per_file: int = ConfigField(default=3_000, ge=0, writable=True) 

182 # Workers for the parallel discovery/hash planning pass. 0 = auto, sized to 

183 # the container-aware CPU budget (see runtime.cpu.available_cpu_count). 

184 # `add --max-cpus N` sets this per invocation. Sizes only the planning pass, 

185 # not the GPU-fed extract/embed batch. 

186 ingest_workers: int = ConfigField(default=0, ge=0, writable=True) 

187 # Worker PROCESSES for a bulk ingest (distinct from ingest_workers, which sizes 

188 # the planning pass's threads). Each owns a GPU, a private store and its own 

189 # slice of the corpus, and the shards are folded into one index at the end. 

190 # 0 = auto: one worker per visible card, used once the corpus is big enough to 

191 # pay for them. N pins the count; worker i takes card i % card_count, so more 

192 # workers than cards share a card's engine rather than double-booking it. 

193 ingest_processes: int = ConfigField(default=0, ge=0, writable=True) 

194 # Passages packed into one embed request. Larger batches keep a GPU's 

195 # continuous-batching slots full: small per-passage requests leave the card 

196 # batch-starved (~96% util, low throughput). The engine still re-splits to 

197 # its physical batch, so raising this only helps up to the server's --batch. 

198 embed_batch_sequences: int = ConfigField( 

199 default=64, 

200 ge=1, 

201 writable=True, 

202 description=( 

203 "Passages packed into one embed request. Larger batches keep a GPU's " 

204 "continuous-batching slots full. The engine re-splits to its physical " 

205 "batch, so raising this helps only up to the server's --batch" 

206 ), 

207 ) 

208 # Files allowed in their compute phase at once during ingest. 0 = auto: the 

209 # ceiling scales with the detected embed fleet (replicas x per-replica 

210 # in-flight) so a multi-GPU box is kept fed without a manual cap, falling back 

211 # to the CPU quota on a single card. Set a positive value only to override the 

212 # auto sizing. Sizes the extract+embed fan-out, not the plan pass. 

213 ingest_max_inflight: int = ConfigField( 

214 default=0, 

215 ge=0, 

216 writable=True, 

217 description=( 

218 "Files allowed in their compute phase at once during ingest. 0 = auto, " 

219 "scaled to the detected embed fleet. Sizes the extract and embed fan-out, " 

220 "not the planning pass" 

221 ), 

222 ) 

223 # Gate for the pre-ask sync; --no-sync overrides per invocation. 

224 auto_sync: bool = ConfigField(default=True, writable=True) 

225 max_embed_chars: int = Field( 

226 default=2000, 

227 ge=1, 

228 description="Maximum characters sent to the embedding model per chunk. Longer text is cut", 

229 ) 

230 top_k: int = ConfigField(default=12, ge=1, writable=True) 

231 max_distance: float = ConfigField(default=0.75, ge=0.0, writable=True) 

232 # Abstention floor against the [0, 1] fused relevance score (0.0 = no 

233 # filtering). When every retrieved chunk falls below it, ask refuses instead 

234 # of feeding noise as context. The fused score normalizes against the 

235 # configured weight budget (a constant), so an arm's top hit scores a stable 

236 # share of it; useful floors start around 0.4. Tune against your own corpus. 

237 min_relevance_score: float = ConfigField(default=0.0, ge=0.0, writable=True) 

238 adaptive_threshold: bool = ConfigField(default=False, writable=True) 

239 rag_system_prompt: str = ConfigField( 

240 default=DEFAULT_RAG_SYSTEM_PROMPT, min_length=1, writable=True 

241 ) 

242 general_system_prompt: str = ConfigField( 

243 default=DEFAULT_GENERAL_SYSTEM_PROMPT, min_length=1, writable=True 

244 ) 

245 chat_mode: ChatMode = ConfigField(default=ChatMode.SEARCH, writable=True) 

246 ignore_dirs: frozenset[str] = Field( 

247 default=DEFAULT_IGNORE_DIRS, 

248 description=( 

249 "Directory names ingest never walks (.git, node_modules, and similar). " 

250 "Use a .lilbeeignore file for per-library patterns" 

251 ), 

252 ) 

253 # OCR for scanned PDFs via vision-capable chat model. 

254 # None = auto-detect (use OCR if chat model is vision-capable). 

255 # True = force OCR regardless of detection. 

256 # False = disable OCR entirely. 

257 enable_ocr: bool | None = ConfigField(default=None, writable=True) 

258 # Per-page timeout in seconds for vision OCR (0 = no limit). Sized so a dense 

259 # full-page scan finishes on modest hardware; a raised vision_ocr_max_tokens 

260 # needs matching headroom here. 

261 ocr_timeout: float = ConfigField(default=300.0, ge=0.0, writable=True) 

262 # Outer wall-clock budget for the streamed pool drain: load grace plus 

263 # per_page * pages. Tune up for slow hardware (M1 Pro vision is 

264 # ~5min/page) or down for fast hardware. ocr_timeout still governs the 

265 # per-page expectation that drives the total budget. 

266 vision_load_budget_s: float = ConfigField(default=300.0, ge=0.0, writable=True) 

267 # Hard cap on tokens generated per OCR page. A real page is well under this; 

268 # the vision request's repeat penalty stops a page looping one line, and the cap 

269 # bounds any loop that still escapes it. Raising it lengthens per-page 

270 # generation on dense scans, so give ocr_timeout matching headroom. 

271 vision_ocr_max_tokens: int = ConfigField(default=4096, ge=256, writable=True) 

272 # Pages OCR'd concurrently, and the vision server's continuous-batching slots. 

273 # A single-page decode underutilizes a modern GPU (~half SM); batching several 

274 # pages raises throughput. Each slot adds KV cache, so lower it on small GPUs. 

275 vision_ocr_concurrency: int = ConfigField(default=4, ge=1, writable=True) 

276 

277 # Tesseract OCR language codes for the scanned-document fallback (used when no 

278 # vision model is set), e.g. ["eng"] or ["eng", "deu"]. Set via env as 

279 # LILBEE_OCR_LANGUAGE="eng+deu". xberg requires a non-empty list. 

280 ocr_language: list[str] = ConfigField(default_factory=lambda: ["eng"], writable=True) 

281 # PDF pages xberg OCRs. auto = pages whose native text fails its quality 

282 # check; scanned_pages also OCRs every page graded as a scan. 

283 ocr_strategy: OcrPageStrategy = ConfigField(default=OcrPageStrategy.AUTO, writable=True) 

284 # Scan-grade threshold for scanned_pages. A slide with a full-bleed 

285 # background image grades 0.5, so lower this to OCR such slides too. 

286 ocr_scan_confidence: float = ConfigField(default=0.7, ge=0.0, le=1.0, writable=True) 

287 # 1-indexed pages that lilbee OCRs in every PDF; while set, it replaces the 

288 # ocr_strategy page selection. Env form: LILBEE_FORCE_OCR_PAGES="1,3". 

289 force_ocr_pages: list[int] = ConfigField(default_factory=list, writable=True) 

290 # Typed entity table for exact counting/cross-referencing; corpus-scale pass, off by default. 

291 entity_extraction: bool = ConfigField(default=False, writable=True) 

292 semantic_chunking: bool = ConfigField(default=False, writable=True) 

293 topic_threshold: float = ConfigField(default=0.75, ge=0.0, le=1.0, writable=True) 

294 # Size chunks in real tokens via the embedder's tokenizer backend, not the 

295 # chars-per-token heuristic. Plain/heading chunkers only; semantic sizes by chars. 

296 token_sizing: bool = ConfigField(default=False, writable=True, reindex=True) 

297 # Index each recognized table as its own markdown-serialized chunk. 

298 table_extraction: bool = ConfigField(default=False, writable=True, reindex=True) 

299 # Layout-aware PDF extraction (reading-order sort, header/footer stripping), 

300 # run in xberg's AUTO strategy so detection only fires when it helps. Off by 

301 # default: enabling it downloads the ONNX layout and table-structure models 

302 # and adds per-page inference, which a CPU-only ingest pays for. 

303 layout_detection: bool = ConfigField(default=False, writable=True, reindex=True) 

304 # Table structure model; only applied when layout_detection is on. 

305 table_model: TableModel = ConfigField( 

306 default=TableModel.SLANET_AUTO, writable=True, reindex=True 

307 ) 

308 # Wall-clock cap per file for one xberg extraction, seconds. 0 = no cap. 

309 # xberg's own default is 600s and applies on its batch path, so leaving this 

310 # unset drops a slow file at ten minutes with no lilbee knob to raise it. 

311 # Zero matches the uncapped single-file path, and ingest is background work. 

312 extraction_timeout: int = ConfigField(default=0, ge=0, writable=True) 

313 # Coalesce concurrent extractions into one xberg extract_batch call. 

314 batch_extraction: bool = ConfigField(default=False, writable=True) 

315 batch_extraction_size: int = ConfigField(default=8, ge=1, writable=True) 

316 # xberg's shared thread budget: PDF rendering, OCR and ONNX inference. It 

317 # also bounds concurrent Tesseract sessions, which xberg further limits to 

318 # what free memory holds. 0 = auto, runtime.cpu.cpu_quota() (half the usable 

319 # CPUs). The rayon pool is fixed at the first extraction, so a change takes 

320 # full effect after a restart. 

321 extraction_threads: int = ConfigField(default=0, ge=0, writable=True) 

322 # Size of anyio's thread pool: synchronous handlers (MCP tools, sync routes) 

323 # that may run off the event loop at once. The ceiling on agents one daemon 

324 # serves before their calls queue. 

325 mcp_tool_threads: int = ConfigField(default=40, ge=1, writable=True) 

326 # Crawled pages converted to markdown on anyio's thread pool at once. The 

327 # conversion is synchronous, so this keeps it off the event loop that serves 

328 # requests. 0 converts inline on the loop. 

329 crawl_convert_workers: int = ConfigField(default=2, ge=0, writable=True) 

330 server_host: str = Field( 

331 default="127.0.0.1", 

332 description=( 

333 "Address `lilbee serve` binds. Loopback by default; set 0.0.0.0 to accept " 

334 "connections from the network" 

335 ), 

336 ) 

337 server_port: int = Field( 

338 default=0, 

339 ge=0, 

340 le=65535, 

341 description="Port `lilbee serve` binds. 0 picks a free port and prints it", 

342 ) 

343 cors_origins: list[str] = Field( 

344 default_factory=list, 

345 description=( 

346 "Extra browser origins the HTTP server accepts, in addition to cors_origin_regex" 

347 ), 

348 ) 

349 cors_origin_regex: str = Field( 

350 default=DEFAULT_CORS_ORIGIN_REGEX, 

351 description="Regular expression matching browser origins the HTTP server accepts", 

352 ) 

353 # Seconds between SSE heartbeat events when the producer queue is idle. 

354 # Must stay well below the plugin's STREAM_IDLE_TIMEOUT_MS (120s) so a 

355 # single long-running vision OCR page can't starve the client into aborting. 

356 sse_heartbeat_interval: float = ConfigField(default=30.0, ge=0.0, writable=True) 

357 json_mode: bool = Field( 

358 default=False, 

359 description=( 

360 "Emit structured JSON from CLI commands. The --json flag sets it for one invocation" 

361 ), 

362 ) 

363 temperature: float | None = ConfigField(default=0.1, ge=0.0, writable=True) 

364 top_p: float | None = ConfigField(default=0.9, ge=0.0, le=1.0, writable=True) 

365 top_k_sampling: int | None = ConfigField(default=40, ge=1, writable=True) 

366 # 1.1 is llama.cpp's default. Leaving this at None caused n-gram loops 

367 # ("tire tire tire...") on some open-weights models. 

368 repeat_penalty: float | None = ConfigField(default=1.1, ge=0.0, writable=True) 

369 num_ctx: int | None = ConfigField(default=None, ge=1, writable=True) 

370 max_tokens: int | None = ConfigField(default=4096, ge=1, writable=True) 

371 seed: int | None = ConfigField(default=None, writable=True) 

372 llm_provider: LlmProvider = ConfigField(default=LlmProvider.AUTO, writable=True) 

373 # Path to a llama-server binary. Empty = use the bundled lilbee-engine 

374 # wheel binary, else a llama-server on PATH. 

375 llama_server_path: str = ConfigField(default="", writable=True) 

376 # Per-server local model-manager URLs. Blank means "use the server's spec 

377 # default" (resolved in providers.local_servers.config_urls); the default 

378 # URL literal lives only in the spec, which core must not import. 

379 ollama_base_url: str = ConfigField(default="", writable=True) 

380 lm_studio_base_url: str = ConfigField(default="", writable=True) 

381 llm_api_key: str = ConfigField(default="", writable=True, write_only=True) 

382 openrouter_api_key: str = ConfigField(default="", writable=True, write_only=True) 

383 gemini_api_key: str = ConfigField(default="", writable=True, write_only=True) 

384 anthropic_api_key: str = ConfigField(default="", writable=True, write_only=True) 

385 openai_api_key: str = ConfigField(default="", writable=True, write_only=True) 

386 mistral_api_key: str = ConfigField(default="", writable=True, write_only=True) 

387 deepseek_api_key: str = ConfigField(default="", writable=True, write_only=True) 

388 hf_token: str = ConfigField(default="", writable=True, write_only=True) 

389 

390 # Retrieval quality knobs. 

391 

392 # Max chunks per source in top-k; prevents one large file monopolizing results. 

393 diversity_max_per_source: int = ConfigField(default=5, ge=1, writable=True) 

394 

395 # MMR relevance/diversity tradeoff; 0 = max diversity, 1 = pure relevance 

396 # (Carbonell & Goldstein 1998). 

397 mmr_lambda: float = ConfigField(default=0.5, ge=0.0, le=1.0, writable=True) 

398 

399 # Vector-only search retrieves this many candidates per final result so 

400 # MMR reranking has a pool to diversify from. Hybrid search ignores it: 

401 # fusion arms stay exactly top_k deep. 

402 candidate_multiplier: int = ConfigField(default=3, ge=1, writable=True) 

403 

404 # Third lexical arm in hybrid search: BM25 over document titles, fused with 

405 # the vector and chunk arms so a query naming a document by title surfaces 

406 # its chunks. Off by default until the eval harness measures it. 

407 title_search: bool = ConfigField(default=False, writable=True) 

408 

409 # Title arm weight relative to a full arm in rank fusion (1.0 = equal voice 

410 # with the vector and chunk arms). 

411 title_search_weight: float = ConfigField(default=0.5, ge=0.0, le=1.0, writable=True) 

412 

413 # Lexical (BM25) arm weight relative to the vector arm in rank fusion. 

414 # 1.0 gives the two arms equal voice; lowering it lets 

415 # a strong dense embedder dominate on corpora where the lexical arm adds 

416 # noise rather than signal. The right value is corpus-dependent and set by 

417 # the retrieval benchmark, not guessed here. 

418 lexical_fusion_weight: float = ConfigField(default=1.0, ge=0.0, le=1.0, writable=True) 

419 

420 # Adaptive fusion: scale the BM25 arm per query by vector-arm confidence 

421 # instead of a fixed lexical_fusion_weight (a peaked dense ranking downweights 

422 # lexical, a flat one keeps it). OFF by default, pending a benchmark run to 

423 # confirm it beats the fixed weight. lexical_fusion_weight is the ceiling the 

424 # rule scales down from. Set adaptive_fusion=true to enable it. 

425 adaptive_fusion: bool = ConfigField(default=False, writable=True) 

426 

427 # Vector-similarity margin at which the lexical arm is fully silenced; smaller 

428 # = more aggressive downweighting. 0 disables adaptation entirely (the lexical 

429 # arm keeps its full fixed weight). 

430 adaptive_fusion_margin: float = ConfigField(default=0.15, ge=0.0, le=2.0, writable=True) 

431 

432 # Stemmer/stop-word language for the BM25 (FTS) indexes, a tantivy language 

433 # name ("English", "German", "French", ...). Applied when an index is 

434 # (re)built, so changing it needs `lilbee rebuild` on an existing store. 

435 # A bad name would otherwise fail index creation quietly and hybrid search 

436 # would degrade to vector-only. 

437 fts_language: FtsLanguage = ConfigField( 

438 default=FtsLanguage.ENGLISH, writable=True, reindex=True 

439 ) 

440 

441 @field_validator("fts_language", mode="before") 

442 @classmethod 

443 def _validate_fts_language(cls, value: Any) -> FtsLanguage: 

444 """Accept a language name in any casing, with surrounding whitespace.""" 

445 try: 

446 return FtsLanguage(str(value).strip().title()) 

447 except ValueError as exc: 

448 valid = ", ".join(member.value for member in FtsLanguage) 

449 raise ValueError(f"fts_language must be one of: {valid}") from exc 

450 

451 # Prefix each chunk's document title to its embedding input (the stored 

452 # chunk text is unchanged). Changes the embedding space: toggling it needs 

453 # `lilbee rebuild`, so it ships off. 

454 embed_titles: bool = ConfigField(default=False, writable=True, reindex=True) 

455 

456 # Contextual retrieval: prepend one LLM-written sentence situating each 

457 # chunk in its document to the embedding input. One generation per chunk, 

458 # so ingest slows substantially; stored text and citations stay verbatim. 

459 # Toggling needs `lilbee rebuild`. 

460 contextual_enrichment: bool = ConfigField(default=False, writable=True, reindex=True) 

461 

462 # Drop tables-of-contents and classification-banner cover/title pages from 

463 # search results. OFF by default; validate per corpus, since the cover-page 

464 # heuristic can also fire on short banner-carrying body pages. A query-matched 

465 # or top-ranked page is never dropped, so removal is limited to structural 

466 # chunks the query did not hit. 

467 filter_structural_chunks: bool = ConfigField(default=False, writable=True) 

468 

469 # Chunk count at/above which sync builds an approximate (ANN) vector index 

470 # so search stays fast at millions of vectors. Below this, search uses exact 

471 # flat scan (faster and exact for small vaults). 0 disables the ANN index. 

472 ann_index_threshold: int = ConfigField(default=50_000, ge=0, writable=True) 

473 

474 # Condense a follow-up question into a standalone retrieval query using 

475 # the chat history (one LLM call; skipped when there is no history). 

476 # Without it, "what about his brother?" is embedded and BM25-matched 

477 # with its pronouns. 

478 history_rewrite: bool = ConfigField(default=False, writable=True) 

479 

480 # Route questions by shape before top-k retrieval: a question naming a 

481 # document resolves to that document's chunks; a count-shaped question 

482 # runs a full-corpus scan (a count is a corpus property top-k cannot 

483 # answer). Unrecognized shapes take the topical path unchanged. 

484 intent_routing: bool = ConfigField(default=True, writable=True) 

485 

486 # Ask the chat model to classify count questions the deterministic 

487 # patterns miss (phrasing variants, other languages). Adds one short LLM 

488 # call to every turn the patterns don't already route, so it's opt-in. 

489 intent_llm: bool = ConfigField(default=False, writable=True) 

490 

491 # LLM-generated alternative queries for expansion. 0 disables. 

492 query_expansion_count: int = ConfigField(default=3, ge=0, writable=True) 

493 

494 # Skip LLM expansion when tokenized query length ≤ this. The LLM round-trip 

495 # dominates latency on small local models; short queries already have strong 

496 # BM25/vector signal. Concept-graph expansion still runs. 0 disables the skip. 

497 expansion_short_query_tokens: int = ConfigField(default=2, ge=0, writable=True) 

498 

499 # Cosine-distance step when adaptive-widening retry kicks in. 

500 adaptive_threshold_step: float = ConfigField(default=0.2, gt=0.0, writable=True) 

501 

502 # Reject expansion variants below expansion_similarity_threshold. 

503 expansion_guardrails: bool = ConfigField(default=True, writable=True) 

504 

505 # Min cosine similarity between question and variant embeddings. 

506 expansion_similarity_threshold: float = ConfigField(default=0.5, ge=0.0, le=1.0, writable=True) 

507 

508 # Saturating BM25 confidence (s / (s + 5)) above which query expansion is 

509 # skipped; 0.8 corresponds to a raw BM25 score of 20. 

510 expansion_skip_threshold: float = Field( 

511 default=0.8, 

512 ge=0.0, 

513 le=1.0, 

514 description=( 

515 "Saturating BM25 confidence above which query expansion is skipped. " 

516 "0.8 corresponds to a raw BM25 score of 20" 

517 ), 

518 ) 

519 

520 # Min relative BM25 top-1 vs top-2 gap ((top - second) / top) to skip expansion. 

521 expansion_skip_gap: float = Field( 

522 default=0.15, 

523 ge=0.0, 

524 le=1.0, 

525 description=( 

526 "Minimum relative BM25 gap between the top two hits that skips query expansion" 

527 ), 

528 ) 

529 

530 # Chunks included in LLM context after adaptive selection. 

531 max_context_sources: int = ConfigField(default=8, ge=1, writable=True) 

532 

533 # Adjacent chunks pulled from the same source on each side of every 

534 # selected chunk and merged into one contiguous passage, so a hit that 

535 # lands mid-argument regains the text before and after it. 0 disables. 

536 # Capped: it is a small chunk radius (useful values are single digits), and 

537 # the merged text is token-budget-bounded anyway, so a large value only 

538 # inflates per-query fetch cost -- and a misread as a token count (e.g. 

539 # 50000) would build a megabyte-long IN-predicate per source. 

540 neighbor_expansion: int = ConfigField(default=0, ge=0, le=100, writable=True) 

541 

542 # HyDE (Gao et al. 2022): hypothetical-answer embedding search. +~500ms. 

543 hyde: bool = ConfigField(default=False, writable=True) 

544 

545 # HyDE result weight relative to real-doc search (0.0-1.0). 

546 hyde_weight: float = ConfigField(default=0.7, ge=0.0, le=1.0, writable=True) 

547 

548 # HyDE prompt template. Must contain {question} placeholder. 

549 hyde_prompt: str = Field( 

550 default=( 

551 "Write a 50-100 word passage that directly answers this question as if " 

552 "it were an excerpt from a real document. Do not include any preamble, " 

553 "just write the passage.\n\nQuestion: {question}" 

554 ), 

555 description=( 

556 "Prompt template HyDE uses to write the hypothetical answer. Must contain {question}" 

557 ), 

558 ) 

559 

560 # Reranker model ref. Empty disables reranking. Native GGUFs run on 

561 # llama-server (rank pooling or LLM logprob scoring); hosted refs 

562 # (cohere/voyage/jina/together/hf-tei) need the backend extra. 

563 reranker_model: str = ConfigField(default="", public=True) 

564 

565 # auto detects cross-encoder vs LLM reranker by GGUF arch; override forces one. 

566 reranker_type: RerankerType = ConfigField(default=RerankerType.AUTO, writable=True, public=True) 

567 # Relevance prompt for LLM rerankers; empty uses the built-in generic template. 

568 # A format string with {query} and {document} placeholders. 

569 reranker_prompt: str = ConfigField(default="", writable=True, public=True) 

570 

571 # Recommend safety-stripped models in Picks and Discover. Off keeps them 

572 # in browse and search only; on restores them to the recommendations. 

573 include_uncensored: bool = ConfigField(default=False, writable=True) 

574 

575 # Long-term chat memory. Off by default (opt-in): when disabled the whole 

576 # subsystem is dormant and the write surfaces respond with an enable hint. 

577 memory_enabled: bool = ConfigField(default=False, writable=True) 

578 

579 # Facts recalled by similarity per turn (preferences are always injected). 

580 memory_top_k: int = ConfigField(default=5, ge=0, writable=True) 

581 

582 # Cosine-distance ceiling for fact recall; stricter than the document default 

583 # because a tiny memory corpus floods at the wider document threshold. 

584 memory_max_distance: float = ConfigField(default=0.6, ge=0.0, le=1.0, writable=True) 

585 

586 # Char/4 token budget for the injected memory block. 

587 memory_token_budget: int = ConfigField(default=512, ge=0, writable=True) 

588 

589 # Per-owner soft cap; oldest memories evicted past it (runaway-write guard). 

590 memory_max_per_owner: int = ConfigField(default=200, ge=1, writable=True) 

591 

592 # Cosine distance below which a new memory is treated as a duplicate of an 

593 # existing same-owner memory and updates it in place instead of inserting. 

594 memory_dedup_distance: float = ConfigField(default=0.05, ge=0.0, le=1.0, writable=True) 

595 

596 # LLM pass that extracts memories from the chat loop. Off by default; extracted 

597 # memories are saved directly and recalled like any other memory. 

598 memory_auto_extract: bool = ConfigField(default=False, writable=True) 

599 

600 # Candidate count sent to the reranker. 

601 rerank_candidates: int = ConfigField(default=60, ge=1, writable=True, public=True) 

602 

603 # Blend reranker scores with the retrieval fusion signal (position-aware). 

604 # Off = the cross-encoder's own ordering stands unblended, which isolates 

605 # the reranker's effect when measuring it. 

606 rerank_blend: bool = ConfigField(default=True, writable=True, public=True) 

607 

608 # Drop candidates whose RAW reranker score falls below this; unset = off. 

609 # The scale is provider/model specific (bge logits can be negative, hosted 

610 # rerankers use 0..1), so set it against observed scores. 

611 rerank_min_score: float | None = ConfigField(default=None, writable=True, public=True) 

612 

613 # Date-range filter; only fires when a temporal keyword is detected. 

614 temporal_filtering: bool = ConfigField(default=True, writable=True) 

615 

616 # If True, emit <think>…</think> content as separate SSE reasoning events; 

617 # if False, strip it silently. 

618 show_reasoning: bool = ConfigField(default=False, writable=True) 

619 

620 # How /v1/chat/completions presents a reasoning model's thinking. ``separate`` 

621 # reports it in ``reasoning_content`` (OpenAI-compatible); ``inline`` keeps it 

622 # in ``content`` as <think> text for clients that never render 

623 # ``reasoning_content``; ``off`` asks the model not to think. A request's 

624 # ``reasoning`` field overrides this per call. 

625 completions_reasoning: ReasoningMode = ConfigField( 

626 default=ReasoningMode.SEPARATE, writable=True 

627 ) 

628 

629 # How /v1/messages presents a reasoning model's thinking. ``separate`` reports 

630 # it as a ``thinking`` block (Anthropic-compatible); ``inline`` folds it into 

631 # the answer text for clients that never render thinking blocks; ``off`` asks 

632 # the model not to think and drops any thinking it produces anyway. A 

633 # request's ``thinking`` parameter overrides this per call. 

634 messages_reasoning: ReasoningMode = ConfigField(default=ReasoningMode.SEPARATE, writable=True) 

635 

636 # Maximum reasoning characters before lilbee forces the model to answer. 

637 # Per-model overrides apply on top of this default. Approx N/4 tokens. 

638 # 0 disables the cap (unlimited reasoning; accept the runaway-loop risk). 

639 max_reasoning_chars: int = ConfigField(default=64_000, ge=0, writable=True) 

640 

641 # Web crawling. 

642 

643 # How crawls fetch pages. ``http`` (default) uses a plain HTTP client with 

644 # no browser, the lightweight path for static / server-rendered sites. 

645 # ``browser`` launches a tuned Chromium with JavaScript enabled for sites 

646 # that render content client-side, at a much higher memory cost. 

647 crawl_render_mode: CrawlRenderMode = ConfigField(default=CrawlRenderMode.HTTP, writable=True) 

648 

649 # Browser-mode memory levers (only used when crawl_render_mode is browser). 

650 # Recycle the Chromium process every N fetched pages to cap RSS growth on a 

651 # long recursive crawl; 0 disables recycling. Raise on a roomy machine for 

652 # fewer restarts, lower it if memory is tight. 

653 crawl_browser_recycle_pages: int = ConfigField(default=50, ge=0, writable=True) 

654 

655 # Extra Chromium launch flags for browser-mode crawls. Defaults trim shared 

656 # memory and GPU use; override to pass site- or environment-specific flags. 

657 crawl_browser_extra_args: list[str] = ConfigField( 

658 default_factory=lambda: ["--disable-dev-shm-usage", "--disable-gpu"], 

659 writable=True, 

660 ) 

661 

662 # Optional global ceilings. None = no ceiling. 

663 crawl_max_depth: int | None = ConfigField(default=None, ge=0, writable=True) 

664 crawl_max_pages: int | None = ConfigField(default=None, ge=1, writable=True) 

665 

666 # Default page bound for an unbounded crawl (no explicit max_pages / 

667 # crawl_max_pages), so a hostile site can't exhaust the disk by default. 

668 # An explicit limit overrides it; raise this to crawl larger sites unbounded. 

669 crawl_safety_max_pages: int = ConfigField(default=5_000, ge=1, writable=True) 

670 

671 # Per-URL fetch timeout, seconds. 

672 crawl_timeout: int = ConfigField(default=30, ge=1, writable=True) 

673 

674 # 0 = unlimited, default = CPU count. 

675 crawl_max_concurrent: int = Field( 

676 default=0, 

677 ge=0, 

678 description="Pages fetched in parallel during a crawl. 0 = unlimited; default = CPU count", 

679 ) 

680 

681 # Seconds between periodic syncs during crawl. 0 = sync only at end. 

682 crawl_sync_interval: int = ConfigField(default=30, ge=0, writable=True) 

683 

684 # Per-request delay + jitter (defaults chosen to be gentler than crawl4ai's). 

685 crawl_mean_delay: float = ConfigField(default=0.5, ge=0.0, writable=True) 

686 crawl_max_delay_range: float = ConfigField(default=0.5, ge=0.0, writable=True) 

687 

688 # In-flight requests per crawl. 

689 crawl_concurrent_requests: int = ConfigField(default=3, ge=1, writable=True) 

690 

691 # Per-domain rate-limiter that backs off on HTTP 429/503 and retries. 

692 crawl_retry_on_rate_limit: bool = ConfigField(default=True, writable=True) 

693 crawl_retry_base_delay_min: float = ConfigField(default=1.0, ge=0.0, writable=True) 

694 crawl_retry_base_delay_max: float = ConfigField(default=3.0, ge=0.0, writable=True) 

695 crawl_retry_max_backoff: float = ConfigField(default=30.0, ge=0.0, writable=True) 

696 crawl_retry_max_attempts: int = ConfigField(default=3, ge=0, writable=True) 

697 

698 # Regex patterns dropped at link-discovery time. Defaults block CMS 

699 # scaffolding (WordPress admin, archives, tracking params, etc.). 

700 crawl_exclude_patterns: list[str] = ConfigField( 

701 default_factory=lambda: list(DEFAULT_CRAWL_EXCLUDE_PATTERNS), 

702 writable=True, 

703 ) 

704 

705 # Fraction of GPU/unified memory reserved for loaded models. 

706 gpu_memory_fraction: float = ConfigField(default=0.75, ge=0.1, le=1.0, writable=True) 

707 

708 # Share of a card placement may charge, leaving room for allocator 

709 # fragmentation and driver overhead. Tunable because it decides admission: at 

710 # the default, a 16 GB machine whose chat model needs 12-13 GB can be refused 

711 # chat entirely, and the owner is the one who knows whether that card has the 

712 # room. Raising it trades safety margin for the ability to serve at all. 

713 usable_vram_fraction: float = ConfigField(default=0.9, ge=0.5, le=1.0, writable=True) 

714 

715 # RAM held back for the OS when placing against system memory, in GiB. Capped 

716 # at a quarter of total RAM either way, so a small host keeps its proportional 

717 # reserve however this is set. 

718 system_memory_reserve_gb: float = ConfigField(default=4.0, ge=0.0, le=64.0, writable=True) 

719 

720 # Data-parallel replicas of the embed / vision role across GPUs: N independent 

721 # servers, round-robined, so large-scale ingest fans the embedding / OCR work 

722 # across the whole box. 0 means "auto": one replica per detected GPU, capped by 

723 # the VRAM left after the persistent query fleet (chat, one embed, rerank, one 

724 # vision) is reserved. A positive value pins the count. The extra replicas are 

725 # ingest-only and reclaimed when ingest ends; the persistent query embedder / 

726 # vision (replica 0) always exists if its model fits. 

727 embed_replicas: int = ConfigField(default=0, ge=0, writable=True) 

728 vision_replicas: int = ConfigField(default=0, ge=0, writable=True) 

729 

730 # Seconds a model stays loaded after last use. 0 = unload immediately. 

731 model_keep_alive: int = ConfigField(default=300, ge=0, writable=True) 

732 

733 # Spawn every configured role server at startup instead of on first use. 

734 # Trades a slower TUI mount (the role servers cold-start in parallel) for a 

735 # responsive first interaction. Roles whose model is unset are skipped, so a 

736 # setup with only chat + embed never spawns rerank or vision. Set to false 

737 # for headless / scripted use where the first call doesn't need to be fast. 

738 worker_pool_eager_start: bool = ConfigField(default=True, writable=True) 

739 

740 # Leave the engine fleet running on quit so the next launch adopts it warm. 

741 # On keeps the engine process alive across app close so the next launch 

742 # binds instantly; its weights still follow engine_idle_ttl_minutes. Off 

743 # (default): the engine stops when the last lilbee process exits, leaving 

744 # the machine clean. 

745 keep_engine_warm: bool = ConfigField(default=False, writable=True) 

746 

747 # Hugging Face's high-performance transfer mode: more connections and much 

748 # larger in-flight buffers. Off by default, those ceilings suit a server 

749 # rather than a laptop also holding a model in memory. 

750 fast_model_downloads: bool = ConfigField(default=False, writable=True) 

751 

752 # Idle minutes before the engine unloads its weights (llama-swap ttl), in 

753 # every mode: even a persistent engine naps when unused. 0 keeps weights 

754 # loaded until the engine stops. 

755 engine_idle_ttl_minutes: int = ConfigField(default=5, writable=True) 

756 

757 # Working n_ctx the dynamic picker aims for. Default scales with 

758 # total host RAM (see core.system.chat_ctx_target_for_total_bytes): 

759 # <16 GiB -> 8192, 16-32 -> 12288, 32-64 -> 16384, 64-128 -> 24576, 

760 # >=128 -> 65536 (an agent-capable window on server-class hosts). 

761 # 8192 is the floor; the picker still clamps to training_ctx and 

762 # host headroom. 

763 chat_n_ctx_target: int = ConfigField( 

764 default_factory=scaled_chat_ctx_target_default, 

765 ge=512, 

766 writable=True, 

767 ) 

768 

769 # Condense turns that outgrow chat_n_ctx_target into carried notes instead 

770 # of dropping them. Off: zero model calls; the oldest turns drop and the 

771 # context chip shows it. On: each firing blocks on a summarize call 

772 # (measured: 1.3-2.5s per 60-turn fold on a datacenter GPU, 0.7-2s on an 

773 # 8-core CPU with a 0.6B-4B model). 

774 chat_compaction: bool = ConfigField(default=False, writable=True) 

775 

776 # Persist conversations and expose the Sessions drawer, tab, and commands. 

777 # On by default; turning it off stops chats being written to disk, hides the 

778 # ctrl+o binding from the footer, and gates the Sessions view behind a notice. 

779 # Governs the human surfaces (TUI, HTTP, CLI); agent sessions have their own 

780 # flag below, so the two domains the store already separates stay separate. 

781 sessions_enabled: bool = ConfigField(default=True, writable=True) 

782 

783 # The agent (MCP) half of the same feature, off by default: agent hosts 

784 # generally track their own conversation history, and the seven session 

785 # tools cost schema on every request whether or not anything uses them. 

786 mcp_sessions_enabled: bool = ConfigField(default=False, writable=True) 

787 

788 # Explicit ceiling for the dynamic n_ctx picker. ``None`` (default) 

789 # lets the model's training_ctx from GGUF metadata be the ceiling, 

790 # so a 128K-context model can reach for it on a host with the RAM 

791 # to back it. Set explicitly to cap below the model's training_ctx. 

792 num_ctx_max: int | None = ConfigField(default=None, ge=512, writable=True) 

793 

794 # Flash attention. None (default) = on, True = force on, False = off 

795 # for backends or models where it misbehaves. 

796 # Resolves the 'padding V cache to 1024' warning on models with 

797 # uneven per-layer V dims (e.g. Gemma3) and saves ~25% KV memory. 

798 flash_attention: bool | None = ConfigField(default=None, writable=True) 

799 

800 # KV cache element type. q8_0 (default) halves cache memory vs f16 

801 # with no measurable quality loss for chat; q4_0 quarters it with a 

802 # small quality cost. Both require flash attention to be enabled. 

803 kv_cache_type: KvCacheType = ConfigField(default=KvCacheType.Q8_0, writable=True) 

804 

805 # Number of model layers to offload to GPU. None (default) = all 

806 # layers, 0 = CPU only, positive int = partial offload. Useful when a 

807 # discrete GPU has less VRAM than the model needs. 

808 n_gpu_layers: int | None = ConfigField(default=None, writable=True) 

809 

810 # Keep a MoE model's expert weights in system memory, attention and shared 

811 # layers on the GPU. Lets a sparse model run on a card too small to hold it. 

812 # No effect on dense models, which have no expert tensors. 

813 cpu_moe: bool = ConfigField(default=False, writable=True) 

814 

815 # Offload only the first N layers' experts. Takes precedence over cpu_moe; 

816 # a smaller N keeps more of the model resident. 

817 n_cpu_moe: int | None = ConfigField(default=None, writable=True) 

818 

819 # GPU device picker for dual-GPU machines (typical laptop case: 

820 # discrete NVIDIA + integrated Intel/AMD). The Vulkan backend 

821 # enumerates every adapter the system exposes and may pick the 

822 # integrated one first, producing stalls or OOMs that look like 

823 # llama.cpp bugs. Setting ``gpu_devices`` constrains visibility 

824 # before the servers spawn, pinning inference to the chosen device(s). 

825 # 

826 # Accepts a comma-separated list of device indexes ("0", "1", 

827 # "0,1") and applies it to every backend simultaneously: 

828 # ``GGML_VK_VISIBLE_DEVICES`` for Vulkan, ``CUDA_VISIBLE_DEVICES`` 

829 # for CUDA, ``HIP_VISIBLE_DEVICES`` / ``ROCR_VISIBLE_DEVICES`` for 

830 # ROCm. Setting one variable that the active backend ignores is 

831 # harmless, so we set all four rather than detecting the build. 

832 # 

833 # Must be set before the first llama.cpp call; in practice that 

834 # means via ``LILBEE_GPU_DEVICES`` or ``config.toml`` (TUI edits 

835 # only take effect after a restart). ``None`` (default) hands off 

836 # to the autodetect in ``providers/fleet/gpu_select.py``, 

837 # which parses ``vulkaninfo --summary`` and pins the discrete 

838 # adapter when one is present. The autodetect is silent on failure 

839 # (no vulkaninfo, single device, parse error), leaving the 

840 # Vulkan-loader's default ordering in place. 

841 gpu_devices: str | None = ConfigField(default=None, writable=True) 

842 

843 # Primary GPU index passed to ``Llama(main_gpu=...)``. Only matters 

844 # when multiple devices remain visible after ``gpu_devices``; with 

845 # a single visible device, llama.cpp ignores this. ``None`` 

846 # (default) lets llama.cpp pick (index 0). 

847 main_gpu: int | None = ConfigField(default=None, writable=True) 

848 

849 # Manual GPU placement override stored as a JSON scalar (the config.toml store 

850 # is flat, and core must not depend on the provider PlacementSpec type). When 

851 # set, it fully replaces the automatic placement planner: each active role pins 

852 # to the listed device indices, with an optional tensor_split and replica count. 

853 # Edited via the placement CLI/MCP/HTTP/TUI surfaces rather than the generic 

854 # settings list, so public=False. None hands off to the VRAM-aware auto planner. 

855 placement: str | None = ConfigField( 

856 default=None, 

857 writable=True, 

858 public=False, 

859 description=( 

860 "Manual multi-GPU placement spec. It fully replaces the automatic planner: " 

861 "each active role pins to the listed device indices. Edit it with the " 

862 "placement commands, not the settings list. Empty uses the VRAM-aware planner" 

863 ), 

864 ) 

865 

866 # Allow PUT/DELETE /api/placement to apply or clear placement over HTTP. 

867 # Off by default because applying placement restarts the shared fleet's moved roles, which 

868 # is unsafe across concurrent HTTP clients. Turn it on (LILBEE_ALLOW_HTTP_PLACEMENT=1) 

869 # only for a single-client / owned deployment: the plugin's managed local 

870 # server, or a personally-owned pod where one operator runs `lilbee serve`. 

871 allow_http_placement: bool = Field( 

872 default=False, 

873 description=( 

874 "Let PUT and DELETE /api/placement change model placement over HTTP. Off by " 

875 "default because applying placement restarts the moved roles, which is unsafe " 

876 "with concurrent clients. Turn it on only for a deployment you alone use" 

877 ), 

878 ) 

879 

880 # True = Markdown widget for chat; False = plain Static (faster). 

881 markdown_rendering: bool = Field( 

882 default=True, 

883 description=( 

884 "Render chat replies as Markdown in the TUI. Off draws plain text, which is faster" 

885 ), 

886 ) 

887 

888 # TUI theme name; persists the last Ctrl+T pick across sessions. 

889 theme: str = ConfigField(default="rose-pine", writable=True) 

890 

891 # Per-model generation defaults set via apply_model_defaults(). 

892 _model_defaults: Any = None 

893 

894 # Wiki layer. LLM-maintained synthesis pages with citation provenance. 

895 # Off by default; flip to True (or set LILBEE_WIKI=1) to enable. When off, 

896 # the Wiki view tab and the chat ModelBar's scope picker are both hidden. 

897 wiki: bool = ConfigField(default=False, writable=True) 

898 # Whether a sync regenerates touched wiki pages on its own. Off by 

899 # default: enabling the wiki never starts generating by itself, the 

900 # user wikifies explicitly via `lilbee wiki build` / `wiki update`. 

901 wiki_auto_update: bool = ConfigField(default=False, writable=True) 

902 # Read-only: changing the directory at runtime strands prior wiki pages 

903 # under the old path. Users who want a different location set it via 

904 # LILBEE_WIKI_DIR / config.toml before the first wiki_build. 

905 wiki_dir: str = "wiki" 

906 wiki_prune_raw: bool = ConfigField(default=False, writable=True) 

907 

908 # Minimum cosine similarity between a page body and the mean of its 

909 # source chunk vectors before a page is published (below → drafts). 

910 # Replaces the old LLM-based faithfulness score: mean-of-chunks is a 

911 # deterministic, zero-LLM-call signal that routes topic-drifted 

912 # pages to drafts without the 0.0 to 1.0 ambiguity of a model-emitted 

913 # number. Tuning knob: swap to per-chunk max or top-K-mean if the 

914 # default 0.5 produces false drafts. 

915 wiki_embedding_faithfulness_threshold: float = ConfigField( 

916 default=0.5, ge=0.0, le=1.0, writable=True 

917 ) 

918 

919 # Per-call output token cap for wiki generation. Without this a 

920 # reasoning model (Qwen3, DeepSeek-R1) can burn the full context 

921 # window emitting <think> tokens before the actual answer, taking 

922 # minutes per page. Default leaves headroom for a typical reasoning 

923 # budget plus a real response (~1000 output + ~1000 slack). 

924 wiki_summary_max_tokens: int = ConfigField(default=2048, ge=256, writable=True) 

925 

926 # Wiki generation is a structured-output task: the model must emit the 

927 # block separators, the citation footnotes, and verbatim quotes. The 

928 # usual chat default (~0.8) is too creative for that. Lowering the 

929 # sampling temperature makes the model stick to the template and quote 

930 # more faithfully. 0.1 leaves just enough slack to avoid hard loops. 

931 wiki_temperature: float = ConfigField(default=0.1, ge=0.0, le=2.0, writable=True) 

932 

933 # Fraction of citations that must be stale before a wiki page is flagged. 

934 wiki_stale_citation_threshold: float = ConfigField(default=0.5, ge=0.0, le=1.0, writable=True) 

935 

936 # Fraction of content changed that triggers human-review drift guard. 

937 wiki_drift_threshold: float = ConfigField(default=0.3, ge=0.0, le=1.0, writable=True) 

938 

939 # LLM prompt templates for wiki page generation: wiki_synthesis_prompt 

940 # for cross-source synthesis pages, wiki_entity_batch_prompt (below) 

941 # for the per-source batched call. Writable so advanced users can 

942 # override them from /settings, config.toml, or ``LILBEE_WIKI_*_PROMPT`` 

943 # env vars. Templates must keep the expected ``{placeholders}``. If you 

944 # remove one the generator will crash on first use. 

945 wiki_synthesis_prompt: str = ConfigField( 

946 writable=True, 

947 default=( 

948 "You are a knowledge compiler. Given source chunks from MULTIPLE documents " 

949 "about related concepts, write a synthesis wiki page in markdown that connects " 

950 "ideas across sources.\n\n" 

951 "Rules:\n" 

952 "1. Every factual claim MUST have an inline citation [^src1], [^src2], etc. " 

953 "Never cite by chunk label: [Chunk N] labels only organize the " 

954 "chunks below and must not appear in the page.\n" 

955 "2. Cite the EXACT text from the source that supports each claim by quoting it.\n" 

956 "3. For connections, interpretations, or patterns you identify across sources, " 

957 "mark with [*inference*].\n" 

958 "4. Use blockquotes (>) for directly cited facts.\n" 

959 "5. Reference each source by its filename when drawing connections.\n" 

960 "6. End with a citation block in this format:\n\n" 

961 "---\n" 

962 "<!-- citations (auto-generated from _citations table -- do not edit) -->\n" 

963 '[^src1]: {{source_name}}, excerpt: "exact quoted text"\n' 

964 '[^src2]: {{source_name}}, excerpt: "exact quoted text"\n\n' 

965 "Topic: {topic}\n\n" 

966 "Sources:\n{source_list}\n\n" 

967 "Chunks:\n{chunks_text}\n\n" 

968 "Write the synthesis page now. Start with a heading." 

969 ), 

970 ) 

971 

972 # Wiki synthesis clusterer backend. CONCEPTS requires the [graph] extra 

973 # and falls back to EMBEDDING when unavailable. 

974 wiki_clusterer: ClustererBackend = ConfigField( 

975 default=ClustererBackend.EMBEDDING, writable=True 

976 ) 

977 

978 # Neighborhood size for the mutual-kNN graph. 0 = auto-scale from corpus size. 

979 wiki_clusterer_k: int = ConfigField(default=0, ge=0, writable=True) 

980 

981 # LazyGraphRAG-style concept graph. Requires the [graph] extra. 

982 concept_graph: bool = ConfigField(default=True, writable=True) 

983 

984 # Weight of concept overlap boost relative to vector similarity. 

985 concept_boost_weight: float = ConfigField(default=0.3, ge=0.0, le=1.0, writable=True) 

986 

987 # Max noun-phrase concepts extracted per chunk. 

988 concept_max_per_chunk: int = ConfigField(default=5, ge=1, writable=True) 

989 

990 # spaCy NER labels kept by the wiki entity extractor. Anything not 

991 # in this set (QUANTITY, CARDINAL, DATE, TIME, MONEY, PERCENT, 

992 # ORDINAL, ...) is dropped before aggregation. Override via 

993 # LILBEE_CONCEPT_ALLOWED_ENT_TYPES as a comma-separated list. 

994 concept_allowed_ent_types: frozenset[str] = Field( 

995 default=DEFAULT_ALLOWED_NER_LABELS, 

996 description=( 

997 "spaCy NER labels the wiki entity extractor keeps. It drops everything else " 

998 "(QUANTITY, CARDINAL, DATE, and so on) before aggregation" 

999 ), 

1000 ) 

1001 

1002 # Strategy used to extract entities for the concept/entity wiki. 

1003 # NER_ENTITIES (default) pulls typed NER entities with spaCy; concept 

1004 # pages are proposed by the LLM inside the per-source batched call, 

1005 # not by the extractor. NER_CONCEPTS_PLUS_LLM_TYPES layers an 

1006 # LLM-proposed domain schema on top. LLM_TAGGED asks the LLM to tag 

1007 # every chunk (most expensive). Unimplemented modes fall back to 

1008 # NER_ENTITIES. 

1009 wiki_entity_mode: WikiEntityMode = ConfigField( 

1010 default=WikiEntityMode.NER_ENTITIES, writable=True 

1011 ) 

1012 

1013 # Minimum distinct chunk mentions before an entity or concept earns 

1014 # its own wiki page. Filters one-off noise. 

1015 wiki_entity_min_mentions: int = ConfigField(default=3, ge=1, writable=True) 

1016 wiki_stub_max_chunk_refs: int = ConfigField(default=50, ge=1, writable=True) 

1017 

1018 # Auto-update cap: if a single sync touches more than this many 

1019 # concept or entity pages, skip the per-slug regeneration and tell 

1020 # the user to run `lilbee wiki update` explicitly. Keeps a surprise 

1021 # bulk import from firing hundreds of LLM calls. 

1022 wiki_ingest_update_cap: int = ConfigField(default=20, ge=1, writable=True) 

1023 

1024 # Whether the per-source batched call asks the LLM to curate 

1025 # concept pages alongside the pre-extracted entity list. False → 

1026 # entity sections only, no concept curation (incremental ingest 

1027 # path uses this to avoid churning concept slugs per source-touch). 

1028 wiki_extract_concepts: bool = ConfigField(default=True, writable=True) 

1029 

1030 # Minimum chunk count a source must contribute before it is eligible 

1031 # for concept curation. Sources below the floor still get a batched 

1032 # call when they have entities (the prompt writes entity-only 

1033 # sections); sources below the floor with zero entities are skipped 

1034 # entirely. Prevents boilerplate / TOC / appendix documents from 

1035 # burning an LLM call to invent "concepts". 

1036 wiki_batch_min_chunks: int = ConfigField(default=3, ge=1, writable=True) 

1037 

1038 # Prompt template for the per-source batched call. Placeholders: 

1039 # {source}, {entity_list}, {chunks_text}, {concept_instruction}. 

1040 # {concept_instruction} is filled with a concept-curation paragraph 

1041 # when concepts are requested, or the empty string otherwise. 

1042 # Single-entity page written on demand from that entity's chunks across 

1043 # every source naming it. The batched prompt above cannot serve this: it 

1044 # writes every section for one source in one call. 

1045 wiki_entity_page_prompt: str = ConfigField( 

1046 writable=True, 

1047 default=( 

1048 "You are a knowledge compiler. Given source chunks that mention " 

1049 "ONE subject, write a wiki page about that subject in markdown.\n\n" 

1050 "Rules:\n" 

1051 "1. Every factual claim MUST have an inline citation [^src1], [^src2], etc. " 

1052 "Never cite by chunk label: [Chunk N] labels only organize the " 

1053 "chunks below and must not appear in the page.\n" 

1054 "2. Cite the EXACT text from the source that supports each claim by quoting it.\n" 

1055 "3. Write only what the chunks support. Mark anything you infer with " 

1056 "[*inference*].\n" 

1057 "4. Use blockquotes (>) for directly cited facts.\n" 

1058 "5. When sources disagree, say so and cite both.\n" 

1059 "6. End with a citation block in this format:\n\n" 

1060 "---\n" 

1061 "<!-- citations (auto-generated from _citations table -- do not edit) -->\n" 

1062 '[^src1]: {{source_name}}, excerpt: "exact quoted text"\n' 

1063 '[^src2]: {{source_name}}, excerpt: "exact quoted text"\n\n' 

1064 "Subject: {topic}\n\n" 

1065 "Sources:\n{source_list}\n\n" 

1066 "Chunks:\n{chunks_text}\n\n" 

1067 "Write the page now. Start with a heading naming the subject." 

1068 ), 

1069 ) 

1070 wiki_entity_batch_prompt: str = ConfigField( 

1071 writable=True, 

1072 default=( 

1073 "You are writing wiki sections based on these chunks from {source}.\n\n" 

1074 "{concept_instruction}" 

1075 "Write a wiki section for each of these NER ENTITIES: {entity_list}\n\n" 

1076 "Format each section exactly as:\n" 

1077 "## Name\n" 

1078 "{{content with [^src1]-style citations}}\n\n" 

1079 "Rules:\n" 

1080 "1. Every factual claim MUST have an inline citation [^src1], [^src2], etc. " 

1081 "Never cite by chunk label: [Chunk N] labels only organize the " 

1082 "chunks below and must not appear in the page.\n" 

1083 "2. Cite the EXACT text from the source that supports each claim by quoting it.\n" 

1084 "3. For interpretations or connections not directly stated, mark with [*inference*].\n" 

1085 "4. Use blockquotes (>) for directly cited facts.\n" 

1086 "5. End the response with a citation block in this format:\n\n" 

1087 "---\n" 

1088 "<!-- citations (auto-generated from _citations table -- do not edit) -->\n" 

1089 '[^src1]: {{source_name}}, excerpt: "exact quoted text"\n' 

1090 '[^src2]: {{source_name}}, excerpt: "exact quoted text"\n\n' 

1091 "Source chunks:\n{chunks_text}\n" 

1092 ), 

1093 ) 

1094 

1095 # Class variable: not a settings field 

1096 _toml_cache: ClassVar[dict[str, Any]] = {} 

1097 

1098 @field_validator("lilbee_name", mode="after") 

1099 @classmethod 

1100 def _strip_lilbee_name(cls, value: str) -> str: 

1101 """Strip whitespace; an empty string signals 'use the path-derived label'.""" 

1102 return value.strip() 

1103 

1104 @field_validator( 

1105 "temperature", 

1106 "top_p", 

1107 "repeat_penalty", 

1108 "top_k_sampling", 

1109 "num_ctx", 

1110 "seed", 

1111 mode="before", 

1112 ) 

1113 @classmethod 

1114 def _empty_string_to_none(cls, v: Any) -> Any: 

1115 if isinstance(v, str) and v.strip() == "": 

1116 return None 

1117 return v 

1118 

1119 @field_validator("chat_mode", mode="before") 

1120 @classmethod 

1121 def _normalize_chat_mode(cls, v: Any) -> ChatMode: 

1122 """Coerce chat_mode to a ChatMode value; default ChatMode.SEARCH.""" 

1123 if v is None or v == "": 

1124 return ChatMode.SEARCH 

1125 candidate = str(v).strip().lower() 

1126 try: 

1127 return ChatMode(candidate) 

1128 except ValueError as exc: 

1129 valid = ", ".join(repr(m.value) for m in ChatMode) 

1130 raise ValueError(f"chat_mode must be one of {{{valid}}}, got {v!r}") from exc 

1131 

1132 @field_validator("enable_ocr", mode="before") 

1133 @classmethod 

1134 def _parse_enable_ocr(cls, v: Any) -> bool | None: 

1135 """Parse enable_ocr from env var string or direct value. 

1136 

1137 Accepts: true/false/1/0/yes/no (case-insensitive), empty string 

1138 or None for auto-detect. 

1139 """ 

1140 if v is None: 

1141 return None 

1142 if isinstance(v, bool): 

1143 return v 

1144 if isinstance(v, str): 

1145 if v.strip().lower() in ("", "auto", "none"): 

1146 return None 

1147 try: 

1148 return parse_bool(v) 

1149 except ValueError: 

1150 # bool() on a non-empty string is True, so falling through here 

1151 # turned an unparseable value into "on". Warn and auto-detect, 

1152 # matching the sibling validators. 

1153 log.warning("Invalid LILBEE_ENABLE_OCR=%r, using auto", v) 

1154 return None 

1155 return bool(v) 

1156 

1157 @field_validator("ocr_language", mode="before") 

1158 @classmethod 

1159 def _parse_ocr_language(cls, v: Any) -> list[str]: 

1160 """Accept a list or a ``+``/comma/newline-separated string; never empty. 

1161 

1162 Tesseract joins languages with ``+`` (e.g. ``eng+deu``), so that is the 

1163 canonical user-facing form. Commas are also accepted. Newlines are 

1164 accepted because ``app.settings`` joins list values with ``\\n`` when it 

1165 persists them to config.toml; without splitting on it a multi-language 

1166 value would reload as one malformed token. Blank input falls back to 

1167 English, since xberg errors on an empty list. 

1168 """ 

1169 if isinstance(v, str): 

1170 v = v.replace("+", ",").replace("\n", ",").split(",") 

1171 items = v or [] 

1172 langs = [s.strip() for s in items if isinstance(s, str) and s.strip()] 

1173 for lang in langs: 

1174 if not _TESSERACT_LANGUAGE_CODE.fullmatch(lang): 

1175 raise ValueError( 

1176 f"ocr_language: {lang!r} is not a Tesseract language code " 

1177 "(examples: eng, deu, chi_sim, jpn_vert)" 

1178 ) 

1179 return langs or ["eng"] 

1180 

1181 @field_validator("force_ocr_pages", mode="before") 

1182 @classmethod 

1183 def _parse_force_ocr_pages(cls, v: Any) -> list[int]: 

1184 """Accept a list, a string or one int; split string items on commas and newlines. 

1185 

1186 Newlines are accepted because ``app.settings`` joins list values with 

1187 ``\\n`` when it persists them to config.toml. Returns sorted unique pages. 

1188 """ 

1189 items = v if isinstance(v, list) else [v] 

1190 parts = [p for item in items for p in _split_page_item(item)] 

1191 return sorted({_page_number(part) for part in parts if str(part).strip()}) 

1192 

1193 @field_validator("flash_attention", mode="before") 

1194 @classmethod 

1195 def _parse_flash_attention(cls, v: Any) -> bool | None: 

1196 """Auto/on/off tri-state: empty/auto/none -> None, else parse bool.""" 

1197 if v is None: 

1198 return None 

1199 if isinstance(v, bool): 

1200 return v 

1201 if isinstance(v, str): 

1202 if v.strip().lower() in ("", "auto", "none"): 

1203 return None 

1204 try: 

1205 return parse_bool(v) 

1206 except ValueError: 

1207 log.warning("Invalid flash_attention=%r, using auto", v) 

1208 return None 

1209 return bool(v) 

1210 

1211 @field_validator("n_gpu_layers", mode="before") 

1212 @classmethod 

1213 def _parse_n_gpu_layers(cls, v: Any) -> int | None: 

1214 """Auto -> None, ``cpu`` alias -> 0, integers parsed verbatim.""" 

1215 if v is None: 

1216 return None 

1217 if isinstance(v, str): 

1218 label = v.strip().lower() 

1219 if label in ("", "auto", "none"): 

1220 return None 

1221 if label == "cpu": 

1222 return 0 

1223 try: 

1224 return int(label) 

1225 except ValueError: 

1226 log.warning("Invalid LILBEE_N_GPU_LAYERS=%r, using auto", v) 

1227 return None 

1228 return int(v) 

1229 

1230 @field_validator("main_gpu", mode="before") 

1231 @classmethod 

1232 def _parse_main_gpu(cls, v: Any) -> int | None: 

1233 """Empty/auto strings -> None, integers parsed verbatim.""" 

1234 if v is None: 

1235 return None 

1236 if isinstance(v, str): 

1237 label = v.strip().lower() 

1238 if label in ("", "auto", "none"): 

1239 return None 

1240 try: 

1241 return int(label) 

1242 except ValueError: 

1243 log.warning("Invalid LILBEE_MAIN_GPU=%r, using auto", v) 

1244 return None 

1245 return int(v) 

1246 

1247 @field_validator("gpu_devices", mode="before") 

1248 @classmethod 

1249 def _parse_gpu_devices(cls, v: Any) -> str | None: 

1250 """Normalize device list: strip whitespace, drop empties, keep order.""" 

1251 if v is None: 

1252 return None 

1253 if isinstance(v, str): 

1254 label = v.strip().lower() 

1255 if label in ("", "auto", "all", "none"): 

1256 return None 

1257 parts = [p.strip() for p in v.split(",") if p.strip()] 

1258 if not parts: 

1259 return None 

1260 for part in parts: 

1261 if not part.lstrip("-").isdigit(): 

1262 log.warning("Invalid LILBEE_GPU_DEVICES=%r, ignoring", v) 

1263 return None 

1264 return ",".join(parts) 

1265 return str(v) 

1266 

1267 @field_validator("placement", mode="before") 

1268 @classmethod 

1269 def _parse_placement(cls, v: Any) -> str | None: 

1270 """Blank/None -> None; validate a JSON string or PlacementSpec; store JSON.""" 

1271 from lilbee.providers.fleet.placement_spec import PlacementError, PlacementSpec 

1272 

1273 if v is None: 

1274 return None 

1275 if isinstance(v, PlacementSpec): 

1276 json_str = v.to_json() 

1277 PlacementSpec.from_json(json_str) # re-validate a directly-built spec 

1278 return json_str 

1279 if isinstance(v, str): 

1280 if v.strip() == "": 

1281 return None 

1282 PlacementSpec.from_json(v) 

1283 return v 

1284 raise PlacementError("placement must be a JSON string or PlacementSpec") 

1285 

1286 @field_validator("semantic_chunking", mode="before") 

1287 @classmethod 

1288 def _parse_semantic_chunking(cls, v: Any) -> bool: 

1289 """Parse from env string; invalid values warn and fall back to False.""" 

1290 if isinstance(v, bool): 

1291 return v 

1292 if isinstance(v, str): 

1293 try: 

1294 return parse_bool(v) 

1295 except ValueError: 

1296 log.warning("Invalid LILBEE_SEMANTIC_CHUNKING=%r, using default False", v) 

1297 return False 

1298 return bool(v) 

1299 

1300 @field_validator( 

1301 "chat_model", "embedding_model", "vision_model", "reranker_model", mode="after" 

1302 ) 

1303 @classmethod 

1304 def _normalize_model_tag(cls, v: str, info: ValidationInfo) -> str: 

1305 """Validate and canonicalize a model ref; blank means the role is unconfigured.""" 

1306 if not v or not v.strip(): 

1307 return "" 

1308 from lilbee.providers.model_ref import parse_model_ref 

1309 

1310 return parse_model_ref(v).for_openai_prefix() 

1311 

1312 @field_validator("ollama_base_url", "lm_studio_base_url", mode="after") 

1313 @classmethod 

1314 def _strip_trailing_slash(cls, v: str) -> str: 

1315 """Canonicalize a local-server URL once at the write boundary.""" 

1316 return v.rstrip("/") 

1317 

1318 @field_validator("cors_origins", mode="before") 

1319 @classmethod 

1320 def _split_cors_origins(cls, v: Any) -> Any: 

1321 if isinstance(v, str): 

1322 return [o.strip() for o in v.split(",") if o.strip()] 

1323 return v 

1324 

1325 @field_validator("crawl_browser_extra_args", mode="before") 

1326 @classmethod 

1327 def _split_crawl_browser_extra_args(cls, v: Any) -> Any: 

1328 """Accept a newline-separated string, matching how the field is persisted. 

1329 

1330 ``app.settings`` joins list values with newlines before writing them to 

1331 ``config.toml`` as a scalar string. Without this inverse, reload cannot 

1332 coerce that string to ``list[str]`` and the whole config.toml is dropped. 

1333 TOML lists and JSON arrays pass through unchanged. 

1334 """ 

1335 if isinstance(v, str): 

1336 return [a.strip() for a in v.splitlines() if a.strip()] 

1337 return v 

1338 

1339 @field_validator("crawl_exclude_patterns", mode="before") 

1340 @classmethod 

1341 def _split_crawl_exclude_patterns(cls, v: Any) -> Any: 

1342 """Accept newline-separated strings from env vars / plain-text config. 

1343 

1344 Regex commonly uses commas (e.g. `{2,4}`) and pipes (alternation), so 

1345 newline is the only separator safe to use for this field. TOML lists 

1346 and JSON arrays pass through unchanged. 

1347 """ 

1348 if isinstance(v, str): 

1349 return [p.strip() for p in v.splitlines() if p.strip()] 

1350 return v 

1351 

1352 @field_validator("crawl_exclude_patterns", mode="after") 

1353 @classmethod 

1354 def _validate_crawl_exclude_patterns(cls, v: list[str]) -> list[str]: 

1355 """Reject any entry that isn't a valid Python regex. 

1356 

1357 These patterns are compiled at crawl time. An invalid pattern there 

1358 surfaces as an opaque mid-crawl error; catching it at PATCH time gives 

1359 the user a 400 with a pointer to the bad entry. 

1360 """ 

1361 import re 

1362 

1363 bad: list[str] = [] 

1364 for i, pattern in enumerate(v): 

1365 try: 

1366 re.compile(pattern) 

1367 except re.error as exc: 

1368 bad.append(f"[{i}] {pattern!r}: {exc}") 

1369 if bad: 

1370 raise ValueError("invalid regex in crawl_exclude_patterns:\n " + "\n ".join(bad)) 

1371 return v 

1372 

1373 @field_validator("ignore_dirs", mode="before") 

1374 @classmethod 

1375 def _merge_ignore_dirs(cls, v: Any) -> frozenset[str]: 

1376 if isinstance(v, str): 

1377 extra = frozenset(name.strip() for name in v.split(",") if name.strip()) 

1378 return DEFAULT_IGNORE_DIRS | extra 

1379 if isinstance(v, (set, frozenset, list)): 

1380 return DEFAULT_IGNORE_DIRS | frozenset(v) 

1381 return DEFAULT_IGNORE_DIRS 

1382 

1383 @field_validator("concept_allowed_ent_types", mode="before") 

1384 @classmethod 

1385 def _parse_ent_types(cls, v: Any) -> frozenset[str]: 

1386 """Replace-semantics override: a narrowed set is used as-is, 

1387 not unioned with defaults. A user asking for ``PERSON,ORG`` 

1388 wants exactly those kinds. Accepts comma-separated strings 

1389 from env and list / set / frozenset from code. Empty input 

1390 falls back to :data:`DEFAULT_ALLOWED_NER_LABELS` so an empty 

1391 env var does not silently disable the gate. 

1392 """ 

1393 if isinstance(v, str): 

1394 parts = frozenset(name.strip().upper() for name in v.split(",") if name.strip()) 

1395 return parts or DEFAULT_ALLOWED_NER_LABELS 

1396 if isinstance(v, (set, frozenset, list)): 

1397 parts = frozenset(str(x).upper() for x in v) 

1398 return parts or DEFAULT_ALLOWED_NER_LABELS 

1399 return DEFAULT_ALLOWED_NER_LABELS 

1400 

1401 @model_validator(mode="before") 

1402 @classmethod 

1403 def _resolve_defaults(cls, data: Any) -> Any: 

1404 from lilbee.core.system import ( 

1405 canonical_data_root, 

1406 canonical_models_dir, 

1407 default_data_dir, 

1408 find_local_root, 

1409 ) 

1410 

1411 if not isinstance(data, dict): 

1412 return data 

1413 

1414 # An empty LILBEE_DATA_ROOT (delivered as "") must fall through to default 

1415 # resolution like an unset one, not become Path(".") = the process cwd. 

1416 if isinstance(data.get("data_root"), str) and not data["data_root"].strip(): 

1417 data["data_root"] = None 

1418 if data.get("data_root") in (None, _UNSET_PATH): 

1419 data_env = os.environ.get("LILBEE_DATA", "").strip() 

1420 if data_env: 

1421 data["data_root"] = Path(data_env) 

1422 else: 

1423 local = find_local_root() 

1424 data["data_root"] = local if local is not None else default_data_dir() 

1425 # Every child path below derives from this, and the server lock keys on 

1426 # those, so canonicalizing here is what makes one directory key one lock. 

1427 # Also coerces a raw string (LILBEE_DATA_ROOT) to Path. 

1428 root = canonical_data_root(data["data_root"]) 

1429 data["data_root"] = root 

1430 if data.get("documents_dir") in (None, _UNSET_PATH): 

1431 data["documents_dir"] = root / "documents" 

1432 if data.get("data_dir") in (None, _UNSET_PATH): 

1433 data["data_dir"] = root / "data" 

1434 if data.get("lancedb_dir") in (None, _UNSET_PATH): 

1435 data["lancedb_dir"] = root / "data" / "lancedb" 

1436 if data.get("models_dir") in (None, _UNSET_PATH): 

1437 data["models_dir"] = canonical_models_dir() 

1438 

1439 return data 

1440 

1441 @classmethod 

1442 def settings_customise_sources( 

1443 cls, 

1444 settings_cls: type[BaseSettings], 

1445 init_settings: Any, 

1446 env_settings: Any, 

1447 dotenv_settings: Any, 

1448 file_secret_settings: Any, 

1449 ) -> tuple[Any, ...]: 

1450 from lilbee.core.system import canonical_data_root, default_data_dir, find_local_root 

1451 

1452 # .strip() to match _resolve_defaults; a padded value would otherwise 

1453 # send the root and its config.toml to different directories. 

1454 data_env = os.environ.get("LILBEE_DATA", "").strip() 

1455 if data_env: 

1456 toml_dir = Path(data_env) 

1457 else: 

1458 local = find_local_root() 

1459 toml_dir = local if local else default_data_dir() 

1460 # Same call as the root itself, so this looks where the root resolves to; 

1461 # a "~/lilbee" value would otherwise search a literal ./~ and find nothing. 

1462 toml_path = canonical_data_root(toml_dir) / CONFIG_FILE_NAME 

1463 

1464 plain_env = _PlainEnvSource(settings_cls, env_prefix="LILBEE_") 

1465 sources: list[Any] = [init_settings, plain_env] 

1466 if toml_path.exists() and os.environ.get("LILBEE_SKIP_TOML_CONFIG") != "1": 

1467 sources.append(_TomlSource(settings_cls, toml_path)) 

1468 return tuple(sources) 

1469 

1470 @property 

1471 def model_defaults(self) -> Any: 

1472 """Per-model generation defaults (read-only). Set via apply_model_defaults().""" 

1473 return self._model_defaults 

1474 

1475 def apply_model_defaults(self, defaults: Any) -> None: 

1476 """Store per-model generation defaults for 3-layer merge.""" 

1477 object.__setattr__(self, "_model_defaults", defaults) 

1478 

1479 def clear_model_defaults(self) -> None: 

1480 """Reset per-model defaults to None.""" 

1481 object.__setattr__(self, "_model_defaults", None) 

1482 

1483 def generation_options(self, **overrides: Any) -> dict[str, Any]: 

1484 """Merge model defaults, user config, and per-call overrides, dropping None.""" 

1485 result = _model_defaults_dict(self._model_defaults) 

1486 # One name for the output cap inside lilbee; the provider translators rename it. 

1487 if "max_tokens" in result: 

1488 result["num_predict"] = result.pop("max_tokens") 

1489 user_fields: dict[str, Any] = { 

1490 "temperature": self.temperature, 

1491 "top_p": self.top_p, 

1492 "top_k": self.top_k_sampling, 

1493 "repeat_penalty": self.repeat_penalty, 

1494 "num_ctx": self.num_ctx, 

1495 "seed": self.seed, 

1496 "num_predict": self.max_tokens, 

1497 } 

1498 for k, v in user_fields.items(): 

1499 if v is not None: 

1500 result[k] = v 

1501 for k, v in overrides.items(): 

1502 if v is not None: 

1503 result[k] = v 

1504 return result 

1505 

1506 

1507def _model_defaults_dict(defaults: Any) -> dict[str, Any]: 

1508 """Non-None fields of a ModelDefaults instance as a dict.""" 

1509 if defaults is None: 

1510 return {} 

1511 from dataclasses import fields as dc_fields 

1512 

1513 return { 

1514 f.name: getattr(defaults, f.name) 

1515 for f in dc_fields(defaults) 

1516 if getattr(defaults, f.name) is not None 

1517 } 

1518 

1519 

1520class _PlainEnvSource: 

1521 """Reads LILBEE_* env vars as plain strings so field validators handle parsing.""" 

1522 

1523 def __init__(self, settings_cls: type[BaseSettings], env_prefix: str) -> None: 

1524 self._prefix = env_prefix 

1525 self._fields = set(settings_cls.model_fields) 

1526 

1527 def __call__(self) -> dict[str, Any]: 

1528 result: dict[str, Any] = {} 

1529 for field_name in self._fields: 

1530 raw = os.environ.get(f"{self._prefix}{field_name.upper()}") 

1531 if value_is_set(field_name, raw): 

1532 result[field_name] = raw 

1533 return result 

1534 

1535 

1536class _TomlSource: 

1537 """Custom pydantic-settings source that reads config.toml.""" 

1538 

1539 def __init__(self, settings_cls: type[BaseSettings], path: Path) -> None: 

1540 self._path = path 

1541 

1542 def __call__(self) -> dict[str, Any]: 

1543 import tomllib 

1544 

1545 try: 

1546 with self._path.open("rb") as f: 

1547 data = tomllib.load(f) 

1548 except (ValueError, OSError): 

1549 log.warning("Failed to read %s, ignoring", self._path) 

1550 return {} 

1551 # An empty string is unset (the field default applies, since pydantic 

1552 # cannot coerce "" to int|None), except on a clearable model role, 

1553 # where it clears the model. TOML's native types pass through as-is. 

1554 return {k: v for k, v in data.items() if value_is_set(k, v)} 

1555 

1556 

1557def _build_cfg() -> tuple[Config, Exception | None]: 

1558 """Build cfg; on stale-config validation failure, fall back to defaults. 

1559 

1560 A persisted ``config.toml`` from before a breaking schema change can 

1561 contain values the new validators reject. Crashing at module import 

1562 means every command (``lilbee --help`` included) emits a Python 

1563 traceback. Falling back to env+defaults lets the package load; the 

1564 CLI / TUI surfaces the original error before doing real work. 

1565 """ 

1566 try: 

1567 return Config(), None 

1568 except Exception as exc: 

1569 os.environ["LILBEE_SKIP_TOML_CONFIG"] = "1" 

1570 try: 

1571 return Config(), exc 

1572 finally: 

1573 os.environ.pop("LILBEE_SKIP_TOML_CONFIG", None) 

1574 

1575 

1576cfg, config_load_error = _build_cfg() 

1577 

1578# Canonicalize LILBEE_DATA at the cfg.data_root resolution boundary so 

1579# spawn-context worker subprocesses inherit the same data root. 

1580# ``setdefault`` preserves a user-set value. 

1581os.environ.setdefault("LILBEE_DATA", str(cfg.data_root))