Coverage for src/lilbee/app/settings.py: 100%
282 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
1"""Canonical write boundary for lilbee configuration."""
3from __future__ import annotations
5import errno
6from dataclasses import dataclass
7from typing import TYPE_CHECKING, Any
9from pydantic_core import PydanticUndefined
11from lilbee.app.settings_map import SETTINGS_MAP, SettingDef, SettingGroup
12from lilbee.config_meta import (
13 MODEL_ROLE_FIELDS,
14 REINDEX_FIELDS,
15 WRITABLE_CONFIG_FIELDS,
16)
17from lilbee.core import settings as persistent_settings
18from lilbee.core.config import CONFIG_FILE_NAME, Config, cfg
19from lilbee.core.config.keys import (
20 LOAD_AFFECTING_KEYS,
21 PROVIDER_SWITCHING_KEYS,
22)
23from lilbee.core.config.schema import field_type_name
24from lilbee.providers.roles import MODEL_FIELD_TO_ROLE, ROLE_GATE_FIELD_TO_ROLE
25from lilbee.runtime.progress import OcrBackendUsed
27if TYPE_CHECKING:
28 from lilbee.modelhub.registry import ModelRegistry
30_MIN_CHUNK_SIZE = 64
32# Keys that decide which OCR engine runs, and whether a set vision model goes unused.
33OCR_SETTING_KEYS = frozenset({"enable_ocr", "vision_model"})
34OCR_OFF_WARNING = (
35 "OCR is off (enable_ocr = false), so the vision model {model} is not used. "
36 "Scanned PDFs without a text layer are skipped. Set enable_ocr to true to OCR them."
37)
38_OCR_ENGINE_NOTES = {
39 OcrBackendUsed.VISION: (
40 "A vision model is set ({model}), so it is used instead of Tesseract. "
41 "Clear vision_model to use Tesseract."
42 ),
43 OcrBackendUsed.TESSERACT: (
44 "No vision model is set, so Tesseract runs OCR. "
45 "Set vision_model to use a vision model instead."
46 ),
47}
49# Path-typed writable fields whose pydantic "default" is the unresolved
50# sentinel ``Path()`` (a literal "."). The actual default is computed by
51# the model_validator at process start (data_root/documents, vault_base
52# stays as None). Resetting these via the boundary would corrupt the
53# install, so they are refused at the reset gate.
54_NO_RESET_FIELDS: frozenset[str] = frozenset({"documents_dir"})
57@dataclass(frozen=True)
58class SettingInfo:
59 """Externally-facing description of a single writable setting."""
61 key: str
62 value: Any
63 default: Any
64 type: str
65 nullable: bool
66 group: SettingGroup
67 help_text: str
68 choices: tuple[str, ...] | None
69 reindex_required: bool
72@dataclass(frozen=True)
73class SettingsUpdateResult:
74 """Outcome of an ``apply_settings_update`` call."""
76 updated: list[str]
77 reindex_required: bool
78 warnings: tuple[str, ...] = ()
81def ocr_off_warning() -> str | None:
82 """The warning for a set vision model that OCR being off keeps unused, else None."""
83 if cfg.vision_model and cfg.enable_ocr is False:
84 return OCR_OFF_WARNING.format(model=cfg.vision_model)
85 return None
88def ocr_engine_note() -> str | None:
89 """Which OCR engine runs for scanned pages, or None when OCR is off."""
90 backend = OcrBackendUsed.chosen(cfg.enable_ocr, cfg.vision_model)
91 note = _OCR_ENGINE_NOTES.get(backend)
92 return note.format(model=cfg.vision_model) if note is not None else None
95def _update_warnings(changed_keys: set[str]) -> tuple[str, ...]:
96 """Warnings about the configuration an update leaves behind."""
97 warning = ocr_off_warning() if changed_keys & OCR_SETTING_KEYS else None
98 return (warning,) if warning is not None else ()
101def _setting_default(key: str) -> Any:
102 """Return the pydantic default for ``key``, or ``None`` if unset."""
103 info = Config.model_fields[key]
104 if info.default_factory is not None:
105 return info.default_factory() # type: ignore[call-arg]
106 if info.default is PydanticUndefined:
107 return None
108 return info.default
111def _is_write_only(key: str) -> bool:
112 """Return True for fields persisted but never read back (API keys, hf_token)."""
113 extra = Config.model_fields[key].json_schema_extra
114 if isinstance(extra, dict):
115 return bool(extra.get("write_only", False))
116 return False
119def _public_writable_keys() -> list[str]:
120 """Names of every writable config field minus write-only secrets."""
121 keys = set(WRITABLE_CONFIG_FIELDS) | set(MODEL_ROLE_FIELDS)
122 return sorted(k for k in keys if not _is_write_only(k))
125def _setting_help(key: str, definition: SettingDef | None) -> str:
126 """The one documented description for *key*.
128 ``SettingDef.help_text`` wins because it is what the TUI already shows;
129 a field with no settings-map entry falls back to its own description.
130 """
131 if definition is not None and definition.help_text:
132 return definition.help_text
133 return Config.model_fields[key].description or ""
136def _setting_info(key: str, definition: SettingDef | None) -> SettingInfo:
137 nullable = _is_nullable(key)
138 group = definition.group if definition else SettingGroup.MODELS
139 help_text = _setting_help(key, definition)
140 choices = definition.choices if definition else None
141 return SettingInfo(
142 key=key,
143 value=getattr(cfg, key),
144 default=_setting_default(key),
145 type=field_type_name(key),
146 nullable=nullable,
147 group=group,
148 help_text=help_text,
149 choices=choices,
150 reindex_required=key in REINDEX_FIELDS,
151 )
154def _parse_group(group: SettingGroup | str) -> SettingGroup:
155 """Resolve a group value or label to a ``SettingGroup``. Case-insensitive on the value."""
156 if isinstance(group, SettingGroup):
157 return group
158 normalized = group.strip().lower()
159 for candidate in SettingGroup:
160 if candidate.value.lower() == normalized:
161 return candidate
162 raise ValueError(
163 f"Unknown setting group: {group!r}. Valid groups: "
164 f"{', '.join(g.value for g in SettingGroup)}"
165 )
168def list_settings(group: SettingGroup | str | None = None) -> list[SettingInfo]:
169 """List every writable non-secret setting, optionally filtered by group (case-insensitive)."""
170 infos = [_setting_info(key, SETTINGS_MAP.get(key)) for key in _public_writable_keys()]
171 if group is not None:
172 wanted = _parse_group(group)
173 infos = [info for info in infos if info.group == wanted]
174 return sorted(infos, key=lambda info: (info.group.value, info.key))
177def get_setting(key: str) -> SettingInfo:
178 """Return the ``SettingInfo`` for one writable non-secret key."""
179 if not _is_settable(key):
180 raise KeyError(f"Unknown or read-only setting: {key}")
181 if _is_write_only(key):
182 raise KeyError(f"Setting '{key}' is write-only and cannot be read back")
183 return _setting_info(key, SETTINGS_MAP.get(key))
186def _is_settable(key: str) -> bool:
187 return key in WRITABLE_CONFIG_FIELDS or key in MODEL_ROLE_FIELDS
190def _is_nullable(key: str) -> bool:
191 """Return True if ``key`` accepts ``None`` to clear the persisted entry."""
192 if key in WRITABLE_CONFIG_FIELDS:
193 return WRITABLE_CONFIG_FIELDS[key]
194 return False
197def _as_int_setting(value: Any) -> int | None:
198 """Coerce a settings value to int the way pydantic will, or None if not numeric.
200 MCP settings_set forwards raw JSON, so a numeric setting can arrive as a
201 string (``{"chunk_overlap": "1000"}``). The cross-field guards must compare
202 the coerced int, not skip on the string and let pydantic accept an
203 unvalidated value downstream. ``bool`` is excluded (it is not a meaningful
204 chunk size) and non-numeric strings fall through to pydantic's type error.
205 """
206 if isinstance(value, bool):
207 return None
208 if isinstance(value, int):
209 return value
210 if isinstance(value, str):
211 try:
212 return int(value.strip())
213 except ValueError:
214 return None
215 return None
218def _validate(updates: dict[str, Any]) -> None:
219 """Reject unknown keys, null on non-nullable, and out-of-range chunk sizes."""
220 for key, value in updates.items():
221 if not _is_settable(key):
222 raise ValueError(f"Unknown or read-only setting: {key}")
223 if value is None and not _is_nullable(key):
224 raise ValueError(f"Setting '{key}' does not accept null")
225 new_ttl = _as_int_setting(updates.get("engine_idle_ttl_minutes"))
226 if new_ttl is not None and new_ttl < 0:
227 raise ValueError("engine_idle_ttl_minutes must be >= 0 (0 keeps weights loaded)")
228 new_chunk_size = _as_int_setting(updates.get("chunk_size"))
229 if new_chunk_size is not None and new_chunk_size < _MIN_CHUNK_SIZE:
230 raise ValueError(f"chunk_size must be >= {_MIN_CHUNK_SIZE}")
231 effective_chunk_size = new_chunk_size if new_chunk_size is not None else cfg.chunk_size
232 new_overlap = _as_int_setting(updates.get("chunk_overlap"))
233 # Compare the effective overlap against the effective chunk_size so that
234 # lowering chunk_size alone (below the already-persisted overlap) is caught,
235 # not just an explicit new overlap.
236 effective_overlap = new_overlap if new_overlap is not None else cfg.chunk_overlap
237 if effective_overlap >= effective_chunk_size:
238 raise ValueError(
239 f"chunk_overlap ({effective_overlap}) must be < chunk_size ({effective_chunk_size})"
240 )
243def _coerce_value(key: str, value: Any) -> Any:
244 """Canonicalize value before cfg assignment; model-role slots run task validation."""
245 if key in MODEL_ROLE_FIELDS and isinstance(value, str):
246 # heavy: role_validator pulls catalog + modelhub transitively (~300 ms)
247 from lilbee.modelhub.role_validator import validate_model_task_assignment
249 return validate_model_task_assignment(key, value)
250 return value
253def _apply_with_rollback(
254 updates: dict[str, Any],
255) -> tuple[dict[str, Any], list[str], dict[str, Any]]:
256 """Set each key on cfg with snapshot/rollback. Returns (persist, delete, snapshot)."""
257 snapshot = {k: getattr(cfg, k) for k in updates}
258 to_persist: dict[str, Any] = {}
259 to_delete: list[str] = []
260 try:
261 for key, raw in updates.items():
262 if raw is None:
263 setattr(cfg, key, None)
264 to_delete.append(key)
265 continue
266 setattr(cfg, key, _coerce_value(key, raw))
267 normalized = getattr(cfg, key)
268 if isinstance(normalized, list):
269 to_persist[key] = "\n".join(str(x) for x in normalized)
270 else:
271 # Hand the scalar over with its type intact so config.toml holds
272 # `true` and `2560`, not `"True"` and `"2560"`.
273 to_persist[key] = normalized
274 except Exception:
275 _restore_snapshot(snapshot)
276 raise
277 return to_persist, to_delete, snapshot
280def _restore_snapshot(snapshot: dict[str, Any]) -> None:
281 for key, value in snapshot.items():
282 setattr(cfg, key, value)
285def _reload_changed_roles(changed_keys: set[str]) -> None:
286 """Off-thread reload for each changed model-role server; full off-thread drop otherwise.
288 A model-role change (chat_model/embedding_model/reranker_model/vision_model)
289 respawns only that role's server via the per-role reload, so unrelated roles
290 keep serving uninterrupted; so does a setting that gates a role (enable_ocr).
291 A genuinely role-agnostic load key (num_ctx, kv_cache_type) has no single
292 owning role, so it falls back to dropping the whole fleet. Both paths run off
293 the caller's thread, so the settings write never blocks on a slow
294 stop-and-respawn.
295 """
296 from lilbee.app.services import peek_services
298 services = peek_services()
299 if services is None:
300 return
301 changed_role_fields = changed_keys & MODEL_ROLE_FIELDS
302 field_to_role = MODEL_FIELD_TO_ROLE | ROLE_GATE_FIELD_TO_ROLE
303 reloaded = {field_to_role[field] for field in changed_keys & field_to_role.keys()}
304 for role in sorted(reloaded):
305 services.reload_role(role)
306 if "vision_model" in changed_role_fields:
307 # Register/unregister lilbee's xberg OCR backend on any vision-model
308 # change (REST/MCP/TUI/CLI all funnel here), not just the REST route.
309 from lilbee.data.extract.backends import BackendKind, sync_xberg_backend
311 sync_xberg_backend(BackendKind.OCR, services.provider)
312 role_agnostic = (changed_keys & LOAD_AFFECTING_KEYS) - MODEL_ROLE_FIELDS
313 if role_agnostic:
314 services.provider.drop_loaded_models_async()
317def requires_services_reset(updates: dict[str, Any]) -> bool:
318 """True if applying *updates* would tear down and rebuild the Services singleton.
320 A provider switch reconstructs the provider via ``create_provider``, which
321 only runs at services init, so it forces a full ``reset_services()``. Callers
322 on the shared HTTP daemon use this to refuse the swap rather than tear the
323 singleton down under concurrent in-flight handlers.
324 """
325 return bool(set(updates) & PROVIDER_SWITCHING_KEYS)
328def provider_reset_refused_message(action: str) -> str:
329 """Shared user-facing refusal for a provider *action* on the HTTP server.
331 *action* is the verb shown to the user, e.g. ``"Switching"`` or
332 ``"Resetting"``. Kept in one place so the daemon entry points (MCP
333 settings_set / settings_reset, REST config) cannot drift apart.
334 """
335 return (
336 f"{action} the model provider is unavailable on the HTTP server: it rebuilds "
337 "the shared engine for every connected client. Change it from the CLI."
338 )
341def config_write_failure_message(exc: OSError) -> str:
342 """User-facing text for a failed config write; names the fix when the file is locked."""
343 detail = f"Could not write {CONFIG_FILE_NAME}: {exc}."
344 if exc.errno in (errno.EACCES, errno.EPERM):
345 detail += f" Close the program that holds {CONFIG_FILE_NAME} open and try again."
346 return detail
349def _invalidate_caches(changed_keys: set[str]) -> None:
350 """Drop every read-side cache whose freshness depends on a changed setting."""
351 if not changed_keys:
352 return
353 if changed_keys & MODEL_ROLE_FIELDS:
354 # heavy: model_info reads GGUF headers with the gguf parser (~130 ms)
355 from lilbee.modelhub.model_info import invalidate_cache as invalidate_arch_cache
357 invalidate_arch_cache()
358 if changed_keys & (LOAD_AFFECTING_KEYS | ROLE_GATE_FIELD_TO_ROLE.keys()):
359 # heavy: app.services pulls the provider stack + lancedb (~70 ms)
360 _reload_changed_roles(changed_keys)
361 if "token_sizing" in changed_keys:
362 # Unregister lilbee's xberg tokenizer backend when token_sizing is turned
363 # off (via any settings path); the chunker binds it on demand when on.
364 from lilbee.app.services import peek_services
365 from lilbee.data.extract.backends import BackendKind, sync_xberg_backend
367 services = peek_services()
368 if services is not None:
369 sync_xberg_backend(BackendKind.TOKENIZER, services.provider)
370 if changed_keys & PROVIDER_SWITCHING_KEYS:
371 # Swap requires reconstructing the provider singleton via
372 # providers.factory.create_provider, only called at services init.
373 from lilbee.app.services import reset_services
375 reset_services()
376 if "mcp_tool_threads" in changed_keys:
377 # Resize the running server's thread pool now instead of only at startup.
378 from lilbee.server.app import reapply_thread_pool_ceiling
380 reapply_thread_pool_ceiling()
381 if "include_uncensored" in changed_keys:
382 # The picks memoize per process; drop them so the toggle takes
383 # effect on the next read instead of the next restart.
384 from lilbee.catalog.picks import reset_picks
386 reset_picks()
389def apply_settings_update(
390 updates: dict[str, Any],
391 *,
392 allow_model_roles: bool = True,
393) -> SettingsUpdateResult:
394 """Validate, apply, persist, and invalidate caches for a batch of updates.
396 Atomic on validation: a rejection rolls every field back and writes
397 nothing. Atomic on disk failure: an ``OSError`` from the TOML write, or a
398 parse error reloading a corrupt config.toml, restores the in-memory
399 snapshot before re-raising. Cache invalidation runs only after a
400 successful persist.
402 Pass ``allow_model_roles=False`` to reject ``chat_model`` /
403 ``embedding_model`` / ``vision_model`` / ``reranker_model`` at the
404 boundary; the HTTP PATCH /api/config surface uses this to route role
405 writes through PUT /api/models/<role>.
406 """
407 if not allow_model_roles:
408 rejected = MODEL_ROLE_FIELDS & set(updates)
409 if rejected:
410 offender = sorted(rejected)[0]
411 raise ValueError(
412 f"'{offender}' must be set through the dedicated model route, "
413 "not the general settings update."
414 )
415 _validate(updates)
416 embed_in_batch = "embedding_model" in updates
417 # Derived (not user-writable) fields applied alongside the validated batch.
418 effective_updates = dict(updates)
419 if embed_in_batch:
420 # Pin the OLD ref into store meta before mutation, otherwise the
421 # next read lazy-initializes meta from the NEW cfg and silently
422 # hides the dimension drift. Runs even when the value is unchanged
423 # so a legacy meta row is always canonicalized on the first swap
424 # attempt.
425 _pin_legacy_store_meta()
426 # Track the new embedder's output width so a fresh index is built at
427 # the right dimension (embedding_dim is derived, not in SETTINGS_MAP).
428 dim = _embedder_dim_from_gguf(updates["embedding_model"])
429 if dim is not None:
430 effective_updates["embedding_dim"] = dim
431 to_persist, to_delete, snapshot = _apply_with_rollback(effective_updates)
432 # embedding_dim is derived and applied to cfg in-memory, but the overlay
433 # loader ignores it on reload (it is re-derived), so don't write it to disk.
434 to_persist.pop("embedding_dim", None)
435 try:
436 if to_persist:
437 persistent_settings.update_values(cfg.data_root, to_persist)
438 if to_delete:
439 persistent_settings.delete_values(cfg.data_root, to_delete)
440 except (OSError, ValueError):
441 # OSError from the write, or a TOMLDecodeError (ValueError) when
442 # update/delete reloads a corrupt on-disk config.toml: either way the
443 # in-memory snapshot must be restored so cfg matches what was persisted.
444 _restore_snapshot(snapshot)
445 raise
446 _invalidate_caches(set(effective_updates))
447 reindex_required = bool((REINDEX_FIELDS - _inert_reindex_keys()) & set(updates))
448 if embed_in_batch:
449 reindex_required = reindex_required or _embed_reindex_required()
450 return SettingsUpdateResult(
451 updated=sorted(updates),
452 reindex_required=reindex_required,
453 warnings=_update_warnings(set(updates)),
454 )
457def apply_ephemeral_model_swap(field: str, ref: str) -> None:
458 """Apply a chat/embedding model swap to cfg for this process only.
460 Performs the same embedding side effects as the persisted path (legacy
461 store meta pinned under the OLD ref first, then embedding_dim re-derived
462 for the new one) so the mismatch gate and table width stay correct, but
463 never writes config.toml.
464 """
465 if field == "embedding_model":
466 _pin_legacy_store_meta()
467 dim = _embedder_dim_from_gguf(ref)
468 setattr(cfg, field, ref)
469 if dim is not None:
470 cfg.embedding_dim = dim
471 return
472 setattr(cfg, field, ref)
475def _pin_legacy_store_meta() -> None:
476 """Pin the current embedding ref into store meta before swapping it."""
477 # heavy: ~100ms (lance + store init); only paid when embedding_model is in the batch.
478 from lilbee.app.services import get_services
480 get_services().store.initialize_meta_if_legacy()
483def _embedder_dim_from_gguf(ref: str, registry: ModelRegistry | None = None) -> int | None:
484 """The embedder's output width from its GGUF header (``<arch>.embedding_length``).
486 None when the model can't be resolved or the header lacks the field. Cheap: a
487 cached header read, no load. *registry* is forwarded to resolve the GGUF without
488 ``get_services()`` (callers running inside its construction).
489 """
490 from lilbee.providers.base import ProviderError
491 from lilbee.providers.engine_params import resolve_model_path
492 from lilbee.providers.gguf_meta import read_gguf_metadata
494 try:
495 # resolve_model_path raises ProviderError for a non-native (ollama/SDK) ref,
496 # which has no local GGUF -- those embedders carry no width to derive here.
497 meta = read_gguf_metadata(resolve_model_path(ref, registry))
498 except (ProviderError, ValueError, OSError, RuntimeError, TypeError):
499 return None
500 raw = meta.get("embedding_length") if meta else None
501 if not raw:
502 return None
503 try:
504 dim = int(raw)
505 except (TypeError, ValueError):
506 return None
507 return dim if dim > 0 else None
510def reconcile_embedding_dim(registry: ModelRegistry | None = None) -> None:
511 """Pin ``cfg.embedding_dim`` to the native embedder's GGUF width before the store
512 is built; no-op for non-native embedders or an already-matching dim."""
513 dim = _embedder_dim_from_gguf(cfg.embedding_model, registry)
514 if dim is not None and dim != cfg.embedding_dim:
515 cfg.embedding_dim = dim
518def _inert_reindex_keys() -> set[str]:
519 """Reindex keys that change no extraction output under the effective config.
521 xberg reads ``table_model`` only inside layout detection, so a change to it
522 while ``layout_detection`` is off is not worth a rebuild.
523 """
524 return set() if cfg.layout_detection else {"table_model"}
527def _embed_reindex_required() -> bool:
528 """True when the persisted index was built with another embedder than cfg now names.
530 Runs after the swap is applied, so the store compares its meta row against
531 the new ref and the new model's width: the same verdict search refuses on.
532 """
533 from lilbee.app.services import get_services
535 store = get_services().store
536 store.canonicalize_meta_if_legacy()
537 return store.index_mismatch() is not None
540def reset_settings(keys: list[str], *, skip_unresettable: bool = False) -> SettingsUpdateResult:
541 """Reset each key to its pydantic default and apply through the write boundary.
543 Fields whose default is a known sentinel (currently ``documents_dir``,
544 which resolves to ``data_root/documents`` at process start) are
545 refused so a reset doesn't write the literal sentinel back. Pass
546 ``skip_unresettable=True`` for bulk-reset gestures that should drop
547 those fields rather than failing the whole batch.
548 """
549 for key in keys:
550 if not _is_settable(key):
551 raise ValueError(f"Unknown or read-only setting: {key}")
552 if key in _NO_RESET_FIELDS and not skip_unresettable:
553 raise ValueError(
554 f"'{key}' has no resettable default; pass an explicit value via settings_set."
555 )
556 updates: dict[str, Any] = {}
557 for key in keys:
558 if key in _NO_RESET_FIELDS:
559 continue
560 default = _setting_default(key)
561 if default is None and _is_nullable(key):
562 updates[key] = None
563 else:
564 updates[key] = default
565 return apply_settings_update(updates)