Coverage for src/lilbee/app/settings_map.py: 100%
53 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
1"""Shared settings map for interactive configuration."""
3from __future__ import annotations
5from dataclasses import dataclass, field, replace
6from enum import StrEnum
8from pydantic_core import PydanticUndefined
10from lilbee.app.themes import DARK_THEMES
11from lilbee.core.config import cfg
12from lilbee.core.config.model import CLEARABLE_MODEL_FIELDS
13from lilbee.core.config.schema import field_value_set
16class RenderStyle(StrEnum):
17 """How a setting is displayed in /settings."""
19 COMPACT = "compact"
20 FULL = "full"
21 MULTILINE = "multiline"
24class SettingGroup(StrEnum):
25 """Logical bucket names rendered by ``/settings`` and ``settings_list``."""
27 MODELS = "Models"
28 GENERATION = "Generation"
29 RETRIEVAL = "Retrieval"
30 INGEST = "Ingest"
31 WIKI = "Wiki"
32 MEMORY = "Memory"
33 CRAWLING = "Crawling"
34 LOCAL_SERVERS = "Local-Servers"
35 API_KEYS = "API-Keys"
36 SYSTEM = "System"
37 DISPLAY = "Display"
38 GENERAL = "General"
41@dataclass(frozen=True)
42class SettingDef:
43 """Metadata for an interactive setting.
45 ``writable`` is a TUI rendering hint: fields marked ``writable=False``
46 (the model role slots) get a dedicated picker rather than an inline
47 editor, and the ``/set`` slash command refuses them, except that it
48 sets or clears a model role that can be off. The actual
49 write contract for HTTP / MCP / programmatic surfaces lives in
50 ``config_meta.WRITABLE_CONFIG_FIELDS`` + ``MODEL_ROLE_FIELDS`` and
51 is enforced by ``app.settings.apply_settings_update``.
53 ``choices`` is the closed value set a picker renders. It is read off
54 the Config field's own type, so only a field whose value set lives
55 outside that type declares one.
57 ``hidden`` keeps the setting out of the TUI settings screen while
58 leaving it reachable via ``lilbee set`` and the ``LILBEE_*`` env
59 var: use it for transport/server knobs that aren't relevant to a
60 typical TUI session.
61 """
63 type: type
64 nullable: bool
65 writable: bool = True
66 render: RenderStyle = field(default=RenderStyle.COMPACT)
67 group: SettingGroup = SettingGroup.GENERAL
68 help_text: str = ""
69 choices: tuple[str, ...] | None = None
70 hidden: bool = False
71 # List editors validate each line as a regex only when this is set; flag-style
72 # lists (e.g. crawl_browser_extra_args) would be wrongly rejected otherwise.
73 validate_regex: bool = False
74 # Credentials: the TUI masks the editor so the value is never on screen in
75 # plain text, including while it is being pasted.
76 secret: bool = False
79def get_default(key: str) -> object:
80 """Return the cfg default for a setting key."""
81 field_info = type(cfg).model_fields[key]
82 if field_info.default_factory is not None:
83 return field_info.default_factory() # type: ignore[call-arg]
84 if field_info.default is PydanticUndefined:
85 return None
86 return field_info.default
89SETTINGS_MAP: dict[str, SettingDef] = {
90 "chat_model": SettingDef(
91 str,
92 nullable="chat_model" in CLEARABLE_MODEL_FIELDS,
93 writable=False,
94 group=SettingGroup.MODELS,
95 help_text="LLM used for chat generation (vision and reranking are separate slots)",
96 ),
97 "vision_model": SettingDef(
98 str,
99 nullable="vision_model" in CLEARABLE_MODEL_FIELDS,
100 writable=False,
101 group=SettingGroup.MODELS,
102 help_text=(
103 "Vision model for scanned PDF OCR; when set it is used instead of Tesseract. "
104 "Clear it (empty value, including an empty LILBEE_VISION_MODEL) to use Tesseract"
105 ),
106 ),
107 "enable_ocr": SettingDef(
108 bool,
109 nullable=True,
110 group=SettingGroup.INGEST,
111 help_text=(
112 "OCR for scanned PDFs: the vision model when one is set, else Tesseract "
113 "(empty or true = on, since only false is checked; false = off for every "
114 "backend, the vision model included)"
115 ),
116 ),
117 "ocr_timeout": SettingDef(
118 float,
119 nullable=False,
120 group=SettingGroup.INGEST,
121 help_text="Per-page timeout in seconds for vision OCR (0 = no limit)",
122 ),
123 "vision_load_budget_s": SettingDef(
124 float,
125 nullable=False,
126 group=SettingGroup.INGEST,
127 help_text=(
128 "Wall-clock seconds reserved for the vision worker to load the"
129 " model. Total PDF-OCR budget = load_budget + ocr_timeout * pages."
130 ),
131 ),
132 "vision_ocr_max_tokens": SettingDef(
133 int,
134 nullable=False,
135 group=SettingGroup.INGEST,
136 help_text=(
137 "Hard cap on tokens generated per OCR page (bounds runaway repetition"
138 " loops); raising it lengthens page generation, so give ocr_timeout headroom"
139 ),
140 ),
141 "vision_ocr_concurrency": SettingDef(
142 int,
143 nullable=False,
144 group=SettingGroup.INGEST,
145 help_text="Pages OCR'd concurrently per vision server; each slot adds KV cache memory",
146 ),
147 "extraction_timeout": SettingDef(
148 int,
149 nullable=False,
150 group=SettingGroup.INGEST,
151 help_text=(
152 "Wall-clock seconds one file gets to extract before ingest gives up"
153 " on it (0 = no limit)"
154 ),
155 ),
156 "ingest_workers": SettingDef(
157 int,
158 nullable=False,
159 group=SettingGroup.INGEST,
160 help_text="Workers for discovering and hashing files (0 = auto, all available cores)",
161 ),
162 "ingest_processes": SettingDef(
163 int,
164 nullable=False,
165 group=SettingGroup.INGEST,
166 help_text=(
167 "Ingest worker processes, one GPU each (0 = auto, one per card). Used"
168 " once the corpus is big enough to pay for them; 1 keeps ingest in this"
169 " process"
170 ),
171 ),
172 "mcp_tool_threads": SettingDef(
173 int,
174 nullable=False,
175 group=SettingGroup.LOCAL_SERVERS,
176 help_text=(
177 "Threads for synchronous MCP tool handlers; the ceiling on how many agents"
178 " one daemon serves before retrieval calls queue"
179 ),
180 ),
181 "crawl_convert_workers": SettingDef(
182 int,
183 nullable=False,
184 group=SettingGroup.CRAWLING,
185 help_text=(
186 "Crawled pages converted to markdown on worker threads at once, so a crawl"
187 " does not block request handling; 0 converts on the event loop"
188 ),
189 ),
190 "auto_sync": SettingDef(
191 bool,
192 nullable=False,
193 group=SettingGroup.INGEST,
194 help_text="Run a sync before `lilbee ask` (disable on large static corpora)",
195 ),
196 "entity_extraction": SettingDef(
197 bool,
198 nullable=False,
199 group=SettingGroup.INGEST,
200 help_text="Extract typed entities automatically at sync (schema induced on first run)",
201 ),
202 "semantic_chunking": SettingDef(
203 bool,
204 nullable=False,
205 group=SettingGroup.INGEST,
206 help_text="Opt-in topic-aware chunker (default off; may fragment numbered procedures)",
207 ),
208 "topic_threshold": SettingDef(
209 float,
210 nullable=False,
211 group=SettingGroup.INGEST,
212 help_text="Topic-boundary similarity threshold, 0.0-1.0, used when semantic chunking is on",
213 ),
214 "token_sizing": SettingDef(
215 bool,
216 nullable=False,
217 group=SettingGroup.INGEST,
218 help_text=(
219 "Size chunks by real embedder tokens, not chars; on by itself when the "
220 "embedder's window is below the character budget (changes invalidate the index)"
221 ),
222 ),
223 "table_extraction": SettingDef(
224 bool,
225 nullable=False,
226 group=SettingGroup.INGEST,
227 help_text="Index each extracted table as its own chunk (changes invalidate the index)",
228 ),
229 "layout_detection": SettingDef(
230 bool,
231 nullable=False,
232 group=SettingGroup.INGEST,
233 help_text=(
234 "Layout-aware PDF extraction: reading order plus header/footer "
235 "stripping (changes invalidate the index)"
236 ),
237 ),
238 "table_model": SettingDef(
239 str,
240 nullable=False,
241 group=SettingGroup.INGEST,
242 help_text=(
243 "Table structure model used when layout detection is on: slanet_auto "
244 "(docling-parity default), other slanet variants, tatr, or disabled "
245 "(changes invalidate the index)"
246 ),
247 ),
248 "batch_extraction": SettingDef(
249 bool,
250 nullable=False,
251 group=SettingGroup.INGEST,
252 help_text="Coalesce concurrent extractions into one xberg batch call",
253 ),
254 "batch_extraction_size": SettingDef(
255 int,
256 nullable=False,
257 group=SettingGroup.INGEST,
258 help_text="Max files per extract_batch call when batch extraction is on",
259 ),
260 "extraction_threads": SettingDef(
261 int,
262 nullable=False,
263 group=SettingGroup.INGEST,
264 help_text=(
265 "Threads xberg uses for PDF rendering, OCR and layout models, and the"
266 " most Tesseract OCR sessions that run at once (0 = auto, half the"
267 " available cores). Takes full effect after a restart."
268 ),
269 ),
270 "embedding_model": SettingDef(
271 str,
272 nullable="embedding_model" in CLEARABLE_MODEL_FIELDS,
273 writable=False,
274 group=SettingGroup.MODELS,
275 help_text="Model used to embed document chunks",
276 ),
277 "reranker_model": SettingDef(
278 str,
279 nullable="reranker_model" in CLEARABLE_MODEL_FIELDS,
280 writable=False,
281 group=SettingGroup.MODELS,
282 help_text="Cross-encoder model for result reranking (empty = off)",
283 ),
284 "reranker_type": SettingDef(
285 str,
286 nullable=False,
287 group=SettingGroup.MODELS,
288 help_text=(
289 "Reranker serving mode: auto (detect cross-encoder vs LLM by model), "
290 "cross_encoder, or llm"
291 ),
292 ),
293 "reranker_prompt": SettingDef(
294 str,
295 nullable=False,
296 group=SettingGroup.MODELS,
297 help_text="Relevance prompt for LLM rerankers (blank uses the built-in template)",
298 ),
299 "include_uncensored": SettingDef(
300 bool,
301 nullable=False,
302 group=SettingGroup.MODELS,
303 help_text=(
304 "Recommend safety-stripped models in Picks and Discover"
305 " (browse and search always list them)"
306 ),
307 ),
308 "temperature": SettingDef(
309 float,
310 nullable=True,
311 group=SettingGroup.GENERATION,
312 help_text="Sampling temperature (higher = more creative)",
313 ),
314 "top_p": SettingDef(
315 float,
316 nullable=True,
317 group=SettingGroup.GENERATION,
318 help_text="Nucleus sampling cutoff probability",
319 ),
320 "top_k_sampling": SettingDef(
321 int,
322 nullable=True,
323 group=SettingGroup.GENERATION,
324 help_text="Top-K sampling: number of tokens to consider",
325 ),
326 "repeat_penalty": SettingDef(
327 float,
328 nullable=True,
329 group=SettingGroup.GENERATION,
330 help_text="Penalty for repeating tokens",
331 ),
332 "num_ctx": SettingDef(
333 int,
334 nullable=True,
335 group=SettingGroup.GENERATION,
336 help_text=(
337 "Context window size in tokens. Leave empty to size automatically "
338 "(aims for chat_n_ctx_target, ceiling at num_ctx_max or training_ctx)."
339 ),
340 ),
341 "num_ctx_max": SettingDef(
342 int,
343 nullable=True,
344 group=SettingGroup.GENERATION,
345 help_text=(
346 "Explicit ceiling for the dynamic context picker. Leave empty to "
347 "use the model's training_ctx from GGUF metadata as the only "
348 "ceiling. Set to cap below training_ctx (saves KV memory)."
349 ),
350 ),
351 "chat_n_ctx_target": SettingDef(
352 int,
353 nullable=False,
354 group=SettingGroup.GENERATION,
355 help_text=(
356 "Working context the dynamic picker aims for. Fits a RAG turn "
357 "with reasoning headroom; raise for long-document chat."
358 ),
359 ),
360 "flash_attention": SettingDef(
361 bool,
362 nullable=True,
363 group=SettingGroup.GENERATION,
364 help_text=(
365 "Flash attention. Empty (auto) enables it; disable for backends or "
366 "models where it misbehaves. Resolves the V-cache padding warning "
367 "on models with uneven per-layer V dims."
368 ),
369 ),
370 "kv_cache_type": SettingDef(
371 str,
372 nullable=False,
373 group=SettingGroup.GENERATION,
374 help_text=(
375 "KV cache element type. q8_0 / q4_0 halve or quarter cache memory "
376 "but require flash attention to be enabled."
377 ),
378 ),
379 "n_gpu_layers": SettingDef(
380 int,
381 nullable=True,
382 group=SettingGroup.GENERATION,
383 help_text=(
384 "Layers to offload to GPU. Empty = all (recommended), 0 = CPU only, "
385 "positive int = partial offload for tight VRAM."
386 ),
387 ),
388 "cpu_moe": SettingDef(
389 bool,
390 nullable=False,
391 group=SettingGroup.GENERATION,
392 help_text=(
393 "Keep a mixture-of-experts model's expert weights in system memory so "
394 "it fits a smaller GPU. No effect on dense models."
395 ),
396 ),
397 "n_cpu_moe": SettingDef(
398 int,
399 nullable=True,
400 group=SettingGroup.GENERATION,
401 help_text=(
402 "Offload only the first N layers' experts to system memory. Takes "
403 "precedence over the offload-everything setting; smaller N stays faster."
404 ),
405 ),
406 "fast_model_downloads": SettingDef(
407 bool,
408 nullable=False,
409 group=SettingGroup.GENERATION,
410 help_text=(
411 "Warning: Hugging Face states this mode uses all available bandwidth "
412 "and CPU cores, and buffers far more of the download in memory. "
413 "Faster on a fast connection, at the cost of everything else running "
414 "on the machine. Leave it off unless the machine can spare that. "
415 "Requires a restart to take effect."
416 ),
417 ),
418 "gpu_devices": SettingDef(
419 str,
420 nullable=True,
421 group=SettingGroup.GENERATION,
422 help_text=(
423 "Restrict llama.cpp to specific GPU indexes on dual-GPU machines "
424 "(e.g. NVIDIA dGPU + integrated). Comma-separated, like '0' or '0,1'. "
425 "Applies to Vulkan, CUDA, and ROCm. Requires a restart to take effect."
426 ),
427 ),
428 "main_gpu": SettingDef(
429 int,
430 nullable=True,
431 group=SettingGroup.GENERATION,
432 help_text=(
433 "Primary GPU index for llama.cpp when multiple devices are visible. "
434 "Empty = let llama.cpp pick (index 0). Set this together with "
435 "gpu_devices to pin inference to a specific card. Requires a restart "
436 "to take effect."
437 ),
438 ),
439 "seed": SettingDef(
440 int,
441 nullable=True,
442 group=SettingGroup.GENERATION,
443 help_text="Random seed for reproducible output",
444 ),
445 "rag_system_prompt": SettingDef(
446 str,
447 nullable=False,
448 render=RenderStyle.MULTILINE,
449 group=SettingGroup.GENERATION,
450 help_text="System prompt sent when answering with retrieved context",
451 ),
452 "general_system_prompt": SettingDef(
453 str,
454 nullable=False,
455 render=RenderStyle.MULTILINE,
456 group=SettingGroup.GENERATION,
457 help_text="System prompt sent when there are no documents to ground the answer",
458 ),
459 "chat_compaction": SettingDef(
460 bool,
461 nullable=False,
462 group=SettingGroup.GENERATION,
463 help_text=(
464 "Off (default): when a chat outgrows the model's context window the oldest "
465 "turns are dropped. They stay on screen but the model stops seeing them, and "
466 "the context chip by the prompt shows the window filling. Costs nothing. "
467 "On: those turns are condensed into a short summary the model keeps reading, "
468 "so it still knows roughly what was said. That costs one extra model call each "
469 "time it fires, pausing the reply for a few seconds on a GPU and considerably "
470 "longer on a CPU-only machine. Worth turning on if your hardware is quick."
471 ),
472 ),
473 "sessions_enabled": SettingDef(
474 bool,
475 nullable=False,
476 group=SettingGroup.GENERATION,
477 help_text=(
478 "On (default): conversations are saved automatically, and you can list, "
479 "resume, rename, and delete them from the Sessions drawer (ctrl+o), the "
480 "Sessions tab, and the /sessions command. Off: nothing is written to disk, "
481 "the ctrl+o binding leaves the footer, and opening the Sessions view shows a "
482 "notice that sessions are turned off. Turn it off if you would rather your "
483 "chats not persist. Covers the TUI, the HTTP server, and the CLI; agent "
484 "sessions have their own setting."
485 ),
486 ),
487 "mcp_sessions_enabled": SettingDef(
488 bool,
489 nullable=False,
490 group=SettingGroup.GENERATION,
491 help_text=(
492 "Off (default): the session tools are not offered over MCP, and a connected "
493 "agent cannot create or read agent sessions. On: an agent can keep its own "
494 "saved conversations, separate from yours. Most agent hosts already track "
495 "their own history, and the tools cost context on every request, so this "
496 "stays off unless you want an agent owning conversations."
497 ),
498 ),
499 "chat_mode": SettingDef(
500 str,
501 nullable=False,
502 group=SettingGroup.GENERATION,
503 help_text="search runs every chat turn through document retrieval; chat skips it",
504 ),
505 "top_k": SettingDef(
506 int,
507 nullable=False,
508 group=SettingGroup.RETRIEVAL,
509 help_text="Number of chunks returned by search",
510 ),
511 "rerank_candidates": SettingDef(
512 int,
513 nullable=False,
514 group=SettingGroup.RETRIEVAL,
515 help_text="Candidate pool size for reranking",
516 ),
517 "rerank_blend": SettingDef(
518 bool,
519 nullable=False,
520 group=SettingGroup.RETRIEVAL,
521 help_text="Blend reranker scores with retrieval fusion (off = pure reranker order)",
522 ),
523 "rerank_min_score": SettingDef(
524 float,
525 nullable=True,
526 group=SettingGroup.RETRIEVAL,
527 help_text="Drop candidates whose raw reranker score is below this (unset = off)",
528 ),
529 "show_reasoning": SettingDef(
530 bool,
531 nullable=False,
532 group=SettingGroup.DISPLAY,
533 help_text="Show model reasoning/thinking tokens in output",
534 ),
535 "completions_reasoning": SettingDef(
536 str,
537 nullable=False,
538 group=SettingGroup.GENERATION,
539 help_text=(
540 "How /v1/chat/completions presents thinking: separate "
541 "reasoning_content field, inline thinking as plain content text, "
542 "or off (ask the model not to think)"
543 ),
544 ),
545 "messages_reasoning": SettingDef(
546 str,
547 nullable=False,
548 group=SettingGroup.GENERATION,
549 help_text=(
550 "How /v1/messages presents thinking: separate thinking block, "
551 "inline thinking as plain answer text, or off (ask the model not "
552 "to think)"
553 ),
554 ),
555 "lilbee_name": SettingDef(
556 str,
557 nullable=False,
558 group=SettingGroup.DISPLAY,
559 help_text=(
560 "Human-readable label for this lilbee, shown in the status bar. "
561 "Empty falls back to 'global' for the platform default dir or "
562 "to the project path (~-substituted and left-truncated)."
563 ),
564 ),
565 "show_lilbee_path": SettingDef(
566 bool,
567 nullable=False,
568 group=SettingGroup.DISPLAY,
569 help_text=(
570 "Show the full absolute path in the status bar: expands 'global' "
571 "to its on-disk path and skips ~ substitution / truncation."
572 ),
573 ),
574 "theme": SettingDef(
575 str,
576 nullable=False,
577 group=SettingGroup.DISPLAY,
578 help_text="TUI color theme. Cycle with Ctrl+T; the active theme persists across sessions.",
579 choices=tuple(DARK_THEMES),
580 ),
581 "wiki": SettingDef(
582 bool,
583 nullable=False,
584 group=SettingGroup.WIKI,
585 help_text=(
586 "Enable the wiki layer (cited concept and entity pages). "
587 "GPU-heavy: a build spends one LLM call per source document, "
588 "so a large library takes hours. Enabling this generates nothing "
589 "on its own: you wikify explicitly, or turn on wiki_auto_update"
590 ),
591 ),
592 "wiki_auto_update": SettingDef(
593 bool,
594 nullable=False,
595 group=SettingGroup.WIKI,
596 help_text="Regenerate touched wiki pages after each sync (off: wikify explicitly)",
597 ),
598 "wiki_dir": SettingDef(
599 str,
600 nullable=False,
601 writable=False,
602 group=SettingGroup.WIKI,
603 help_text=(
604 "Directory under data_root where wiki pages live (set via env / config.toml only)"
605 ),
606 ),
607 "wiki_prune_raw": SettingDef(
608 bool,
609 nullable=False,
610 group=SettingGroup.WIKI,
611 help_text="Delete raw chunks after summarizing into the wiki",
612 ),
613 "wiki_embedding_faithfulness_threshold": SettingDef(
614 float,
615 nullable=False,
616 group=SettingGroup.WIKI,
617 help_text=(
618 "Minimum cosine similarity (0-1) between a generated page and "
619 "the mean of its source chunk vectors before publishing. "
620 "Pages below the threshold route to drafts/."
621 ),
622 ),
623 "wiki_stale_citation_threshold": SettingDef(
624 float,
625 nullable=False,
626 group=SettingGroup.WIKI,
627 help_text="Fraction of stale citations before a page is flagged by wiki prune",
628 ),
629 "wiki_drift_threshold": SettingDef(
630 float,
631 nullable=False,
632 group=SettingGroup.WIKI,
633 help_text="Max fraction of changed lines before regeneration requires review",
634 ),
635 "wiki_clusterer": SettingDef(
636 str,
637 nullable=False,
638 group=SettingGroup.WIKI,
639 help_text="Synthesis clusterer backend (embedding or concepts)",
640 ),
641 "wiki_entity_mode": SettingDef(
642 str,
643 nullable=False,
644 group=SettingGroup.WIKI,
645 help_text=(
646 "Entity extraction strategy. ner_entities (typed spaCy NER) is the "
647 "only implemented mode; the other values fall back to it with a warning"
648 ),
649 ),
650 "wiki_entity_min_mentions": SettingDef(
651 int,
652 nullable=False,
653 group=SettingGroup.WIKI,
654 help_text="Minimum chunk mentions before an entity or concept gets its own page",
655 ),
656 "wiki_stub_max_chunk_refs": SettingDef(
657 int,
658 nullable=False,
659 group=SettingGroup.WIKI,
660 help_text=(
661 "How many source chunks a page kept for lazy generation draws on. "
662 "Caps the browse index's size; already more than one page's context "
663 "budget admits, so raising it rarely changes what a page says"
664 ),
665 ),
666 "wiki_ingest_update_cap": SettingDef(
667 int,
668 nullable=False,
669 group=SettingGroup.WIKI,
670 help_text=(
671 "Touched-page cap for auto-update after sync. "
672 "Beyond this count, run `lilbee wiki update` manually."
673 ),
674 ),
675 "wiki_synthesis_prompt": SettingDef(
676 str,
677 nullable=False,
678 render=RenderStyle.FULL,
679 group=SettingGroup.WIKI,
680 help_text=(
681 "Prompt for cross-source synthesis pages. "
682 "Must keep {topic}, {source_list}, and {chunks_text}."
683 ),
684 ),
685 "wiki_entity_page_prompt": SettingDef(
686 str,
687 nullable=False,
688 render=RenderStyle.FULL,
689 group=SettingGroup.WIKI,
690 help_text=(
691 "Prompt for a single page generated on demand from one subject's "
692 "chunks across every source naming it. "
693 "Must keep {topic}, {source_list}, and {chunks_text}."
694 ),
695 ),
696 "wiki_entity_batch_prompt": SettingDef(
697 str,
698 nullable=False,
699 render=RenderStyle.FULL,
700 group=SettingGroup.WIKI,
701 help_text=(
702 "Prompt for the per-source batched call. "
703 "Must keep {source}, {entity_list}, {chunks_text}, and {concept_instruction}."
704 ),
705 ),
706 "wiki_extract_concepts": SettingDef(
707 bool,
708 nullable=False,
709 group=SettingGroup.WIKI,
710 help_text=(
711 "Whether the per-source batched call asks the LLM to curate concept pages "
712 "alongside the pre-extracted entity list."
713 ),
714 ),
715 "wiki_batch_min_chunks": SettingDef(
716 int,
717 nullable=False,
718 group=SettingGroup.WIKI,
719 help_text=(
720 "Minimum chunks a source must contribute before its batched call includes "
721 "concept curation. Sources below the floor skip the concept-curation "
722 "instruction; sources with zero entities AND below the floor are skipped entirely."
723 ),
724 ),
725 "wiki_clusterer_k": SettingDef(
726 int,
727 nullable=False,
728 group=SettingGroup.WIKI,
729 help_text="Mutual-kNN neighborhood size for the clusterer (0 = auto)",
730 ),
731 "memory_enabled": SettingDef(
732 bool,
733 nullable=False,
734 group=SettingGroup.MEMORY,
735 help_text="Master switch for long-term chat memory (off by default)",
736 ),
737 "memory_auto_extract": SettingDef(
738 bool,
739 nullable=False,
740 group=SettingGroup.MEMORY,
741 help_text="Auto-save durable facts and preferences from each TUI turn (needs memory on)",
742 ),
743 "memory_top_k": SettingDef(
744 int,
745 nullable=False,
746 group=SettingGroup.MEMORY,
747 help_text="Maximum facts recalled into context per turn",
748 ),
749 "memory_max_distance": SettingDef(
750 float,
751 nullable=False,
752 group=SettingGroup.MEMORY,
753 help_text="Recall cutoff distance, 0.0-1.0 (lower is stricter)",
754 ),
755 "memory_token_budget": SettingDef(
756 int,
757 nullable=False,
758 group=SettingGroup.MEMORY,
759 help_text="Token cap on the recalled-memory block added to the prompt",
760 ),
761 "memory_max_per_owner": SettingDef(
762 int,
763 nullable=False,
764 group=SettingGroup.MEMORY,
765 help_text="Soft cap before the oldest memories are evicted",
766 hidden=True,
767 ),
768 "memory_dedup_distance": SettingDef(
769 float,
770 nullable=False,
771 group=SettingGroup.MEMORY,
772 help_text="Near-duplicate distance below which a new memory updates the old",
773 hidden=True,
774 ),
775 "crawl_max_depth": SettingDef(
776 int,
777 nullable=True,
778 group=SettingGroup.CRAWLING,
779 help_text="Optional recursion-depth cap (blank = no cap; per-crawl values win)",
780 ),
781 "crawl_render_mode": SettingDef(
782 str,
783 nullable=False,
784 group=SettingGroup.CRAWLING,
785 help_text=(
786 "How crawls fetch pages. http = lightweight, no browser (default, best "
787 "for static and server-rendered sites). browser = Chromium with "
788 "JavaScript enabled for client-rendered sites, at much higher memory cost."
789 ),
790 ),
791 "crawl_browser_recycle_pages": SettingDef(
792 int,
793 nullable=False,
794 group=SettingGroup.CRAWLING,
795 help_text=(
796 "Browser mode: recycle the Chromium process every N pages to cap memory "
797 "growth on long crawls (0 = never recycle)."
798 ),
799 ),
800 "crawl_browser_extra_args": SettingDef(
801 list,
802 nullable=False,
803 group=SettingGroup.CRAWLING,
804 help_text=(
805 "Browser mode: extra Chromium launch flags, one per line. "
806 "Defaults trim shared-memory and GPU use."
807 ),
808 ),
809 "crawl_max_pages": SettingDef(
810 int,
811 nullable=True,
812 group=SettingGroup.CRAWLING,
813 help_text="Optional global cap on total pages per crawl (blank = no cap).",
814 ),
815 "crawl_safety_max_pages": SettingDef(
816 int,
817 nullable=False,
818 group=SettingGroup.CRAWLING,
819 help_text="Default page bound for an unbounded crawl, so a hostile site cannot "
820 "exhaust the disk. An explicit max-pages overrides it; raise this to crawl "
821 "larger sites unbounded.",
822 ),
823 "crawl_timeout": SettingDef(
824 int,
825 nullable=False,
826 group=SettingGroup.CRAWLING,
827 help_text="Per-page fetch timeout in seconds",
828 ),
829 "crawl_sync_interval": SettingDef(
830 int,
831 nullable=False,
832 group=SettingGroup.CRAWLING,
833 help_text="Seconds between periodic re-syncs during a crawl (0 = sync only at end)",
834 ),
835 "crawl_mean_delay": SettingDef(
836 float,
837 nullable=False,
838 group=SettingGroup.CRAWLING,
839 help_text="Seconds between in-flight requests within a single crawl",
840 ),
841 "crawl_max_delay_range": SettingDef(
842 float,
843 nullable=False,
844 group=SettingGroup.CRAWLING,
845 help_text="Random jitter (seconds) added on top of mean delay",
846 ),
847 "crawl_concurrent_requests": SettingDef(
848 int,
849 nullable=False,
850 group=SettingGroup.CRAWLING,
851 help_text="Concurrent in-flight URLs within one crawl",
852 ),
853 "crawl_retry_on_rate_limit": SettingDef(
854 bool,
855 nullable=False,
856 group=SettingGroup.CRAWLING,
857 help_text="Enable per-domain backoff and retries on HTTP 429/503",
858 ),
859 "crawl_retry_base_delay_min": SettingDef(
860 float,
861 nullable=False,
862 group=SettingGroup.CRAWLING,
863 help_text="Minimum base-delay (seconds) on rate-limit responses",
864 ),
865 "crawl_retry_base_delay_max": SettingDef(
866 float,
867 nullable=False,
868 group=SettingGroup.CRAWLING,
869 help_text="Maximum base-delay (seconds) on rate-limit responses",
870 ),
871 "crawl_retry_max_backoff": SettingDef(
872 float,
873 nullable=False,
874 group=SettingGroup.CRAWLING,
875 help_text="Upper bound on any single backoff wait (seconds)",
876 ),
877 "crawl_retry_max_attempts": SettingDef(
878 int,
879 nullable=False,
880 group=SettingGroup.CRAWLING,
881 help_text="Retry count per URL when a rate-limit code comes back",
882 ),
883 "crawl_exclude_patterns": SettingDef(
884 list,
885 nullable=False,
886 group=SettingGroup.CRAWLING,
887 validate_regex=True,
888 help_text=(
889 "Regex patterns that skip URLs at link-discovery time during "
890 "recursive crawls. One per line."
891 ),
892 ),
893 "openrouter_api_key": SettingDef(
894 str,
895 nullable=False,
896 group=SettingGroup.API_KEYS,
897 secret=True,
898 help_text="OpenRouter API key (enables frontier models in chat picker)",
899 ),
900 "gemini_api_key": SettingDef(
901 str,
902 nullable=False,
903 group=SettingGroup.API_KEYS,
904 secret=True,
905 help_text="Google Gemini API key (enables frontier models in chat picker)",
906 ),
907 "anthropic_api_key": SettingDef(
908 str,
909 nullable=False,
910 group=SettingGroup.API_KEYS,
911 secret=True,
912 help_text="Anthropic API key (enables frontier models in chat picker)",
913 ),
914 "openai_api_key": SettingDef(
915 str,
916 nullable=False,
917 group=SettingGroup.API_KEYS,
918 secret=True,
919 help_text="OpenAI API key (enables frontier models in chat picker)",
920 ),
921 "mistral_api_key": SettingDef(
922 str,
923 nullable=False,
924 group=SettingGroup.API_KEYS,
925 secret=True,
926 help_text="Mistral API key (enables frontier models in chat picker)",
927 ),
928 "deepseek_api_key": SettingDef(
929 str,
930 nullable=False,
931 group=SettingGroup.API_KEYS,
932 secret=True,
933 help_text="DeepSeek API key (enables frontier models in chat picker)",
934 ),
935 "llm_api_key": SettingDef(
936 str,
937 nullable=False,
938 group=SettingGroup.API_KEYS,
939 secret=True,
940 help_text="API key for the remote OpenAI-compatible endpoint (llm_provider = remote)",
941 ),
942 "hf_token": SettingDef(
943 str,
944 nullable=False,
945 group=SettingGroup.SYSTEM,
946 secret=True,
947 help_text=(
948 "HuggingFace access token. Avoids the unauthenticated download "
949 "rate limit and unlocks gated repos. Stored in plain text in "
950 "config.toml. Env vars (LILBEE_HF_TOKEN, HF_TOKEN) override."
951 ),
952 ),
953 "chunk_size": SettingDef(
954 int,
955 nullable=False,
956 group=SettingGroup.INGEST,
957 help_text="Document chunk size in tokens (changes invalidate the index)",
958 ),
959 "chunk_overlap": SettingDef(
960 int,
961 nullable=False,
962 group=SettingGroup.INGEST,
963 help_text="Tokens of overlap between adjacent chunks (preserves context across boundaries)",
964 ),
965 "max_chunks_per_file": SettingDef(
966 int,
967 nullable=False,
968 group=SettingGroup.INGEST,
969 help_text=(
970 "Most chunks one file can add to the index; a file over the limit is skipped, "
971 "not embedded (0 = no limit). Raise it for a long document such as a "
972 "thousand-page manual at a small chunk_size, then retry skipped files"
973 ),
974 ),
975 "ocr_language": SettingDef(
976 list,
977 nullable=False,
978 group=SettingGroup.INGEST,
979 help_text="Tesseract OCR languages when no vision model is set; '+'-join, e.g. eng+deu",
980 ),
981 "ocr_strategy": SettingDef(
982 str,
983 nullable=False,
984 group=SettingGroup.INGEST,
985 help_text=(
986 "PDF pages to OCR: auto (pages whose text layer is missing or garbled) or"
987 " scanned_pages (also every page that looks like a scan, e.g. a scanned"
988 " page with a hidden text layer)"
989 ),
990 ),
991 "ocr_scan_confidence": SettingDef(
992 float,
993 nullable=False,
994 group=SettingGroup.INGEST,
995 help_text=(
996 "How sure xberg must be that a page is a scan before it OCRs it (0-1)."
997 " Applies only when ocr_strategy is scanned_pages. Lower it to 0.5 to"
998 " also OCR slides with a full-page background image"
999 ),
1000 ),
1001 "force_ocr_pages": SettingDef(
1002 list,
1003 nullable=False,
1004 group=SettingGroup.INGEST,
1005 help_text=(
1006 "PDF page numbers that lilbee OCRs in every PDF, comma-separated (e.g. 1,3)"
1007 " or one per line"
1008 ),
1009 ),
1010 "worker_pool_eager_start": SettingDef(
1011 bool,
1012 nullable=False,
1013 group=SettingGroup.INGEST,
1014 help_text=(
1015 "Spawn every configured role server at TUI startup instead of on first use. "
1016 "Trades cold-start time per role for first-call latency"
1017 ),
1018 ),
1019 "keep_engine_warm": SettingDef(
1020 bool,
1021 nullable=False,
1022 group=SettingGroup.SYSTEM,
1023 help_text=(
1024 "Let the engine outlive lilbee for warm launches; off stops it on last "
1025 "exit unless another lilbee sharing the engine asked to keep it"
1026 ),
1027 ),
1028 "engine_idle_ttl_minutes": SettingDef(
1029 int,
1030 nullable=False,
1031 group=SettingGroup.SYSTEM,
1032 help_text="Idle minutes before the engine unloads its weights; 0 keeps them loaded",
1033 ),
1034 "agent_mcp_enabled": SettingDef(
1035 bool,
1036 nullable=False,
1037 group=SettingGroup.SYSTEM,
1038 help_text=(
1039 "Register lilbee's MCP search tool into agent launchers (opencode, hermes). "
1040 "Disable to bring your own MCP servers; lilbee stays the model provider"
1041 ),
1042 ),
1043 "max_tokens": SettingDef(
1044 int,
1045 nullable=True,
1046 group=SettingGroup.GENERATION,
1047 help_text="Hard cap on generated tokens per response (blank = no cap)",
1048 ),
1049 "max_reasoning_chars": SettingDef(
1050 int,
1051 nullable=False,
1052 group=SettingGroup.GENERATION,
1053 help_text=(
1054 "Maximum reasoning characters before lilbee forces the model to answer "
1055 "(0 = unlimited; per-model overrides apply on top)"
1056 ),
1057 ),
1058 "model_keep_alive": SettingDef(
1059 int,
1060 nullable=False,
1061 group=SettingGroup.GENERATION,
1062 help_text="Seconds the loaded model stays warm between calls (0 = unload immediately)",
1063 ),
1064 "gpu_memory_fraction": SettingDef(
1065 float,
1066 nullable=False,
1067 group=SettingGroup.GENERATION,
1068 help_text="Fraction of GPU memory the model is allowed to claim (0.1-1.0)",
1069 ),
1070 "usable_vram_fraction": SettingDef(
1071 float,
1072 nullable=False,
1073 group=SettingGroup.GENERATION,
1074 help_text=(
1075 "Share of a GPU placement may fill, leaving room for fragmentation and driver "
1076 "overhead (0.5-1.0). Raise it if a model that should fit is being refused; "
1077 "lower it if loads fail near the top of the card."
1078 ),
1079 ),
1080 "system_memory_reserve_gb": SettingDef(
1081 float,
1082 nullable=False,
1083 group=SettingGroup.GENERATION,
1084 help_text=(
1085 "RAM held back for the OS in GiB when serving from system memory (no discrete "
1086 "GPU). Capped at a quarter of total RAM either way."
1087 ),
1088 ),
1089 "embed_replicas": SettingDef(
1090 int,
1091 nullable=False,
1092 group=SettingGroup.GENERATION,
1093 help_text="Embedding servers in parallel (0 = auto, one per GPU; positive pins the count)",
1094 ),
1095 "vision_replicas": SettingDef(
1096 int,
1097 nullable=False,
1098 group=SettingGroup.GENERATION,
1099 help_text="Vision OCR servers in parallel (0 = auto, one per GPU; positive pins the count)",
1100 ),
1101 "candidate_multiplier": SettingDef(
1102 int,
1103 nullable=False,
1104 group=SettingGroup.RETRIEVAL,
1105 help_text="Candidate-pool multiplier over top_k before reranking",
1106 ),
1107 "title_search": SettingDef(
1108 bool,
1109 nullable=False,
1110 group=SettingGroup.RETRIEVAL,
1111 help_text="Match queries against document titles as a third hybrid-search arm",
1112 ),
1113 "title_search_weight": SettingDef(
1114 float,
1115 nullable=False,
1116 group=SettingGroup.RETRIEVAL,
1117 help_text="Title arm weight in rank fusion (1.0 = equal voice with the other arms)",
1118 ),
1119 "lexical_fusion_weight": SettingDef(
1120 float,
1121 nullable=False,
1122 group=SettingGroup.RETRIEVAL,
1123 help_text="BM25 arm weight in fusion (1.0 = equal to vector; lower to favor dense)",
1124 ),
1125 "adaptive_fusion": SettingDef(
1126 bool,
1127 nullable=False,
1128 group=SettingGroup.RETRIEVAL,
1129 help_text="Scale the BM25 weight per query by vector-arm confidence, not a fixed value",
1130 ),
1131 "adaptive_fusion_margin": SettingDef(
1132 float,
1133 nullable=False,
1134 group=SettingGroup.RETRIEVAL,
1135 help_text="Vector-similarity margin at which adaptive fusion fully silences the BM25 arm",
1136 ),
1137 "filter_structural_chunks": SettingDef(
1138 bool,
1139 nullable=False,
1140 group=SettingGroup.RETRIEVAL,
1141 help_text="Drop tables-of-contents and classification-banner cover pages from results",
1142 ),
1143 "fts_language": SettingDef(
1144 str,
1145 nullable=False,
1146 group=SettingGroup.RETRIEVAL,
1147 help_text="Stemmer/stop-word language for BM25 indexes (rebuild to apply)",
1148 ),
1149 "embed_titles": SettingDef(
1150 bool,
1151 nullable=False,
1152 group=SettingGroup.RETRIEVAL,
1153 help_text="Prefix document titles to chunk embeddings (rebuild to apply)",
1154 ),
1155 "contextual_enrichment": SettingDef(
1156 bool,
1157 nullable=False,
1158 group=SettingGroup.RETRIEVAL,
1159 help_text="LLM context sentence per chunk embedding (slow ingest; rebuild to apply)",
1160 ),
1161 "history_rewrite": SettingDef(
1162 bool,
1163 nullable=False,
1164 group=SettingGroup.RETRIEVAL,
1165 help_text=(
1166 "Rewrite a follow-up that refers to earlier turns into a standalone "
1167 "retrieval query (one extra chat-model call on those turns)"
1168 ),
1169 ),
1170 "intent_routing": SettingDef(
1171 bool,
1172 nullable=False,
1173 group=SettingGroup.RETRIEVAL,
1174 help_text="Route document-name lookups to exact retrieval, count questions to a scan",
1175 ),
1176 "intent_llm": SettingDef(
1177 bool,
1178 nullable=False,
1179 group=SettingGroup.RETRIEVAL,
1180 help_text=(
1181 "Classify count questions with the chat model when the fast patterns "
1182 "miss (covers phrasing variants and other languages; adds one short "
1183 "LLM call to those turns)"
1184 ),
1185 ),
1186 "ann_index_threshold": SettingDef(
1187 int,
1188 nullable=False,
1189 group=SettingGroup.RETRIEVAL,
1190 help_text="Chunk count to start building an ANN vector index (0 = always flat search)",
1191 ),
1192 "max_distance": SettingDef(
1193 float,
1194 nullable=False,
1195 group=SettingGroup.RETRIEVAL,
1196 help_text="Maximum vector distance for retrieval matches (lower = stricter)",
1197 ),
1198 "min_relevance_score": SettingDef(
1199 float,
1200 nullable=False,
1201 group=SettingGroup.RETRIEVAL,
1202 help_text="Minimum RRF relevance score for hybrid search results (0.0 = no filter)",
1203 ),
1204 "max_context_sources": SettingDef(
1205 int,
1206 nullable=False,
1207 group=SettingGroup.RETRIEVAL,
1208 help_text="Maximum unique sources contributing chunks to a single answer",
1209 ),
1210 "neighbor_expansion": SettingDef(
1211 int,
1212 nullable=False,
1213 group=SettingGroup.RETRIEVAL,
1214 help_text="Adjacent chunks merged into each retrieved passage per side (0 = off)",
1215 ),
1216 "diversity_max_per_source": SettingDef(
1217 int,
1218 nullable=False,
1219 group=SettingGroup.RETRIEVAL,
1220 help_text="Maximum chunks accepted from any one source (caps source dominance)",
1221 ),
1222 "mmr_lambda": SettingDef(
1223 float,
1224 nullable=False,
1225 group=SettingGroup.RETRIEVAL,
1226 help_text=(
1227 "MMR lambda balancing relevance vs diversity (0 = max diversity, 1 = max relevance)"
1228 ),
1229 ),
1230 "temporal_filtering": SettingDef(
1231 bool,
1232 nullable=False,
1233 group=SettingGroup.RETRIEVAL,
1234 help_text="Detect temporal queries and bias retrieval toward recent chunks",
1235 ),
1236 "hyde": SettingDef(
1237 bool,
1238 nullable=False,
1239 group=SettingGroup.RETRIEVAL,
1240 help_text="Use HyDE (hypothetical answer expansion) to broaden retrieval",
1241 ),
1242 "hyde_weight": SettingDef(
1243 float,
1244 nullable=False,
1245 group=SettingGroup.RETRIEVAL,
1246 help_text="Weight on the HyDE-generated query vector when blending with the original",
1247 ),
1248 "query_expansion_count": SettingDef(
1249 int,
1250 nullable=False,
1251 group=SettingGroup.RETRIEVAL,
1252 help_text="Number of paraphrase expansions per query (0 disables expansion)",
1253 ),
1254 "expansion_similarity_threshold": SettingDef(
1255 float,
1256 nullable=False,
1257 group=SettingGroup.RETRIEVAL,
1258 help_text="Minimum cosine similarity an expansion must keep with the original query",
1259 ),
1260 "expansion_short_query_tokens": SettingDef(
1261 int,
1262 nullable=False,
1263 group=SettingGroup.RETRIEVAL,
1264 help_text="Queries at or below this token count skip expansion (saves a model call)",
1265 ),
1266 "expansion_guardrails": SettingDef(
1267 bool,
1268 nullable=False,
1269 group=SettingGroup.RETRIEVAL,
1270 help_text="Drop expansions that diverge from the original intent",
1271 ),
1272 "adaptive_threshold": SettingDef(
1273 bool,
1274 nullable=False,
1275 group=SettingGroup.RETRIEVAL,
1276 help_text="Widen the distance cutoff when too few results pass (vector-only fallback path)",
1277 ),
1278 "adaptive_threshold_step": SettingDef(
1279 float,
1280 nullable=False,
1281 group=SettingGroup.RETRIEVAL,
1282 help_text="Step size for adaptive relevance-score relaxation when initial recall is empty",
1283 ),
1284 "concept_graph": SettingDef(
1285 bool,
1286 nullable=False,
1287 group=SettingGroup.RETRIEVAL,
1288 help_text="Boost retrieval scores for chunks that share concepts with the query",
1289 ),
1290 "concept_boost_weight": SettingDef(
1291 float,
1292 nullable=False,
1293 group=SettingGroup.RETRIEVAL,
1294 help_text="Maximum boost (0-1) the concept graph can add to a chunk's relevance",
1295 ),
1296 "concept_max_per_chunk": SettingDef(
1297 int,
1298 nullable=False,
1299 group=SettingGroup.RETRIEVAL,
1300 help_text="Maximum concept tags stored per chunk (caps graph density)",
1301 ),
1302 "documents_dir": SettingDef(
1303 str,
1304 nullable=False,
1305 group=SettingGroup.SYSTEM,
1306 help_text="Local documents root that lilbee sync ingests (blank = data_root/documents)",
1307 ),
1308 "vault_base": SettingDef(
1309 str,
1310 nullable=True,
1311 group=SettingGroup.SYSTEM,
1312 help_text="Markdown vault root; results carry a vault-relative path (blank = none)",
1313 ),
1314 "sse_heartbeat_interval": SettingDef(
1315 float,
1316 nullable=False,
1317 group=SettingGroup.SYSTEM,
1318 help_text="Seconds between SSE keep-alive frames sent to idle HTTP stream clients",
1319 hidden=True,
1320 ),
1321 "llm_provider": SettingDef(
1322 str,
1323 nullable=False,
1324 group=SettingGroup.API_KEYS,
1325 help_text=(
1326 "Inference provider: auto (default, runs models locally on llama-server) "
1327 "or remote (external OpenAI-compatible endpoint)"
1328 ),
1329 ),
1330 "ollama_base_url": SettingDef(
1331 str,
1332 nullable=False,
1333 group=SettingGroup.LOCAL_SERVERS,
1334 help_text="Ollama server URL (blank uses http://localhost:11434)",
1335 ),
1336 "lm_studio_base_url": SettingDef(
1337 str,
1338 nullable=False,
1339 group=SettingGroup.LOCAL_SERVERS,
1340 help_text="LM Studio server URL (blank uses http://localhost:1234/v1)",
1341 ),
1342 "llama_server_path": SettingDef(
1343 str,
1344 nullable=False,
1345 group=SettingGroup.API_KEYS,
1346 help_text="Path to a llama-server binary (empty: bundled wheel or PATH)",
1347 ),
1348 "wiki_summary_max_tokens": SettingDef(
1349 int,
1350 nullable=False,
1351 group=SettingGroup.WIKI,
1352 help_text="Maximum tokens generated per wiki page",
1353 ),
1354 "wiki_temperature": SettingDef(
1355 float,
1356 nullable=False,
1357 group=SettingGroup.WIKI,
1358 help_text="Temperature used for wiki page synthesis (low = stay close to sources)",
1359 ),
1360}
1363def _fill_value_sets(settings: dict[str, SettingDef]) -> None:
1364 """Give every setting the closed value set Config declares for it.
1366 An explicit ``choices`` wins, for a field whose value set lives outside
1367 its type (``theme``) and for a deliberate narrowing.
1368 """
1369 for key, definition in settings.items():
1370 if definition.choices is not None:
1371 continue
1372 declared = field_value_set(key)
1373 if declared is not None:
1374 settings[key] = replace(definition, choices=declared)
1377_fill_value_sets(SETTINGS_MAP)