Coverage for src/lilbee/providers/engine_params.py: 100%
119 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
1"""Engine-neutral parameter helpers: model-path resolution, context and GPU-layer
2sizing, and chat-option translation.
4These derive launch/generation parameters from cfg + a model's GGUF metadata
5without loading the model, so both the local llama-server engine and any other
6provider compute them identically. No native binding.
7"""
9from __future__ import annotations
11import logging
12from dataclasses import dataclass
13from pathlib import Path
14from typing import TYPE_CHECKING, Any
16if TYPE_CHECKING:
17 from lilbee.modelhub.registry import ModelRegistry
18from lilbee.core.config import DEFAULT_NUM_CTX, cfg
19from lilbee.core.config.enums import KV_CACHE_TYPE_BYTES, KvCacheType
20from lilbee.core.health_warnings import HealthWarning, WarningCode
21from lilbee.providers.base import (
22 CONTEXT_WINDOW_MARGIN_TOKENS,
23 GENERATION_RESERVE_TOKENS,
24 ProviderError,
25 ProviderErrorKind,
26 estimate_budget_tokens,
27 normalize_generation_options,
28)
29from lilbee.providers.gguf_meta import read_gguf_metadata, train_ctx_from_meta
30from lilbee.providers.model_cache import (
31 compute_dynamic_ctx,
32 get_available_memory,
33 kv_bytes_per_token,
34)
35from lilbee.providers.model_ref import is_loose_model_file
37log = logging.getLogger(__name__)
39EMBED_FALLBACK_CTX = 2048
40"""Context used for embed/rerank when a GGUF reports junk (e.g. context_length=0)."""
42# Sized above chunk_size so BOS re-added on re-tokenization doesn't overflow a full-chunk input.
43_EMBED_CTX_MARGIN = 8
46def resolve_embed_ctx(meta: dict[str, str] | None, model_path: Path) -> int:
47 """Embed/rerank context: worst-case chunk tokenization, capped by trained context.
49 ``chunk_size`` is token-denominated but the chunker enforces a CHARACTER
50 budget (``chunk_size * CHARS_PER_TOKEN``). A BPE token is at least one
51 character, so that char budget is also the provable token ceiling for any
52 chunk the chunker can emit: size the context to it and embed-time
53 truncation becomes impossible. Token-dense text (numeric tables, dense
54 identifiers) otherwise reaches ~2x chunk_size tokens against a 1x cap and
55 silently loses its tail at embed time. A trained context below that budget
56 caps the window instead, and the chunker then sizes chunks in the embedder's
57 own tokens at the served cap (see :func:`embed_token_cap`)."""
58 from lilbee.data.extract.chunk import CHARS_PER_TOKEN
60 train_ctx = train_ctx_from_meta(meta, fallback=EMBED_FALLBACK_CTX, model_path=model_path)
61 return min(train_ctx, cfg.chunk_size * CHARS_PER_TOKEN + _EMBED_CTX_MARGIN)
64def embed_token_cap(ctx: int) -> int:
65 """Tokens an embed server truncates one input to, given its per-slot context."""
66 return max(1, ctx - _EMBED_CTX_MARGIN)
69def embed_window_warning(cap: int) -> HealthWarning | None:
70 """The embed-window warning for a real, actionable gap, else None.
72 Plain chunking self-corrects to token sizing when *cap* binds, so it loses
73 no text and needs no warning. Semantic chunking ignores token sizing and
74 stays character-sized, so a bound *cap* loses text there regardless of the
75 ``token_sizing`` setting.
76 """
77 from lilbee.data.extract.chunk import CHARS_PER_TOKEN
79 if cfg.semantic_chunking:
80 chars = cfg.chunk_size * CHARS_PER_TOKEN
81 if cap >= chars:
82 return None
83 return HealthWarning(
84 code=WarningCode.EMBED_WINDOW_BELOW_CHUNK,
85 message=(
86 f"The embedding model accepts {cap} tokens per input, below the {chars} "
87 f"characters that chunk_size {cfg.chunk_size} allows. Semantic chunking "
88 f"sizes chunks by characters and ignores token sizing, so a chunk over "
89 f"{cap} tokens loses text at embedding time."
90 ),
91 remedy="Turn off semantic_chunking or lower chunk_size.",
92 )
93 if cfg.token_sizing and cap < cfg.chunk_size:
94 return HealthWarning(
95 code=WarningCode.EMBED_WINDOW_BELOW_CHUNK,
96 message=(
97 f"The embedding model accepts {cap} tokens per input, below the configured "
98 f"chunk_size {cfg.chunk_size} in tokens. Chunks are capped to {cap} tokens "
99 f"instead."
100 ),
101 remedy=f"Set chunk_size to {cap} to match the window.",
102 )
103 return None
106_LLM_RERANK_HEADROOM = 512
107"""Tokens reserved above chunk_size for an LLM reranker's query, prompt, and 1-token answer."""
110def resolve_llm_rerank_ctx(meta: dict[str, str] | None, model_path: Path) -> int:
111 """LLM-reranker context: a query+candidate pair, capped by the model's trained context."""
112 train_ctx = train_ctx_from_meta(meta, fallback=EMBED_FALLBACK_CTX, model_path=model_path)
113 return min(train_ctx, cfg.chunk_size + _LLM_RERANK_HEADROOM)
116_VISION_FALLBACK_N_CTX = 4096
117"""Context for a vision load when the GGUF reports no usable context_length."""
119_VISION_PAGE_CTX_CAP = 32768
120"""Per-page ceiling on a vision OCR server's context: covers a single high-res page's
121image tokens plus prompt, while keeping a long-context VLM placeable beside a chat giant."""
123N_GPU_LAYERS_AUTO = -1
124"""llama.cpp's "fit as many layers as the device holds" value for n_gpu_layers.
126The engine measures free VRAM at load and picks the count itself, spilling the
127rest to system memory. That is a better answer than any number lilbee can
128compute ahead of time, because it is taken on the real device after every other
129tenant, so the planner passes this rather than a layer count of its own.
130"""
131# llama.cpp's "offload nothing"; the user's CPU-only opt-out rather than a budget.
132_N_GPU_LAYERS_NONE = 0
135def chat_options_to_kwargs(options: dict[str, Any] | None) -> dict[str, Any]:
136 """Translate user-facing chat options into generation kwargs.
138 The output keys (``temperature``/``top_p``/``top_k``/``seed``/``max_tokens``/
139 ``repeat_penalty``) are accepted by llama-server's OpenAI body. ``top_k`` is
140 kept (local llama.cpp honors it), unlike the SDK/API translator which drops it.
141 ``think`` becomes ``chat_template_kwargs.enable_thinking``, which thinking
142 templates honor and others ignore.
143 """
144 kwargs = normalize_generation_options(options)
145 think = kwargs.pop("think", None)
146 if think is not None:
147 kwargs["chat_template_kwargs"] = {"enable_thinking": think}
148 return kwargs
151def resolve_model_path(model: str, registry: ModelRegistry | None = None) -> Path:
152 """Resolve a model name to a .gguf file path.
154 Resolution order: (1) registry (canonical source for installed models),
155 (2) an absolute path to an existing file. Pass *registry* to resolve without
156 reaching for ``get_services()`` (callers running inside its construction).
157 """
158 if not model:
159 raise ProviderError(
160 "No model is configured for this role. Pick one from the catalog "
161 "or run 'lilbee model pull <model>'.",
162 provider="llama-server",
163 kind=ProviderErrorKind.NOT_FOUND,
164 )
165 if registry is None:
166 # call-time import: keeps the app-layer container off this module's import graph
167 from lilbee.app.services import get_services
169 registry = get_services().registry
170 try:
171 return registry.resolve(model)
172 except (KeyError, ValueError):
173 pass
175 if is_loose_model_file(model):
176 return Path(model)
177 if Path(model).is_absolute():
178 raise ProviderError(
179 f"Model file not found: {model}",
180 provider="llama-server",
181 kind=ProviderErrorKind.NOT_FOUND,
182 )
184 raise ProviderError(
185 f"Model {model!r} is not installed. Run 'lilbee model pull {model}' to download it.",
186 provider="llama-server",
187 kind=ProviderErrorKind.NOT_FOUND,
188 )
191def chat_kv_elem_bytes() -> tuple[float, float]:
192 """Per-element (K, V) byte costs of the KV cache a chat launch allocates.
194 Reads the same flags the launch passes
195 (:func:`lilbee.providers.fleet.planning.chat_cache_type_flags`): K carries
196 ``cfg.kv_cache_type``, while V carries it only when flash attention is
197 certain to be on, because llama.cpp refuses a quantized V cache without it
198 and the launch then leaves V at f16. Budgeting from the launch flags keeps
199 the granted window in step with the cache the engine actually allocates.
200 """
201 # call-time import: planning imports this module at load
202 from lilbee.providers.fleet.planning import chat_cache_type_flags
204 def elem_bytes(flag: str | None) -> float:
205 return KV_CACHE_TYPE_BYTES[KvCacheType(flag) if flag else KvCacheType.F16]
207 k_flag, v_flag = chat_cache_type_flags()
208 return elem_bytes(k_flag), elem_bytes(v_flag)
211def chat_ctx_ceiling(meta: dict[str, str] | None, model_path: Path) -> int:
212 """Hard upper bound on a chat per-slot n_ctx: trained context, capped by ``cfg.num_ctx_max``."""
213 training_ctx = train_ctx_from_meta(meta, fallback=DEFAULT_NUM_CTX, model_path=model_path)
214 if cfg.num_ctx_max is not None:
215 return min(training_ctx, cfg.num_ctx_max)
216 return training_ctx
219@dataclass(frozen=True)
220class ChatFit:
221 """The GPU offload and per-slot window one chat launch runs with.
223 The pair is inseparable: the window was sized against the memory this
224 offload leaves free, so serving it at any other offload overruns the card.
225 """
227 gpu_layers: int
228 ctx: int
231def resolve_chat_fit(
232 model_path: Path, meta: dict[str, str] | None, *, available_bytes: int | None = None
233) -> ChatFit:
234 """Pick a single-GPU offload and n_ctx aiming for ``cfg.chat_n_ctx_target``,
235 clamped to model + host.
237 A gguf-parser fit answers first
238 (:func:`lilbee.providers.fleet.planning.fit_chat_ctx`), because it prices the
239 cache each layer of this architecture holds, and it may leave layers in
240 system memory to free the KV room a usable window needs. Header math takes
241 over when the estimator cannot answer; it charges every layer as dense
242 attention over the whole window at the configured offload, which
243 under-grants linear-attention, sliding-window and MLA models. Either way the
244 window stops at the smallest of the trained context, ``cfg.num_ctx_max`` and
245 the target.
247 A multi-GPU tensor-split chat is sized separately by the fleet against its
248 per-device headroom (see :func:`lilbee.providers.fleet.ctx.fit_split_ctx`).
249 ``available_bytes`` overrides the live host-memory read, and every caller
250 that is sizing a real launch passes it: the fleet and the surfaces that
251 mirror it hand over
252 :func:`lilbee.providers.fleet.planning.plan_sizing_budget`, which reports the
253 memory of the GPU that will run the model and holds a clean-box snapshot so
254 a reload sizes ctx like the boot did.
255 """
256 # call-time import: planning imports this module at load
257 from lilbee.providers.fleet.planning import fit_chat_ctx
259 training_ctx = train_ctx_from_meta(meta, fallback=DEFAULT_NUM_CTX, model_path=model_path)
260 ceiling = cfg.num_ctx_max if cfg.num_ctx_max is not None else training_ctx
261 if available_bytes is None:
262 available_bytes = get_available_memory(cfg.gpu_memory_fraction)
263 upper = min(training_ctx, ceiling, cfg.chat_n_ctx_target)
265 try:
266 return fit_chat_ctx(model_path, meta, available_bytes=available_bytes, ctx_ceiling=upper)
267 except (ProviderError, OSError, ValueError):
268 log.debug("gguf-parser ctx fit failed for %s, using header math", model_path, exc_info=True)
269 return ChatFit(
270 resolve_n_gpu_layers(embedding=False),
271 _header_math_chat_ctx(
272 model_path,
273 meta,
274 available_bytes=available_bytes,
275 training_ctx=training_ctx,
276 ceiling=ceiling,
277 ),
278 )
281def _header_math_chat_ctx(
282 model_path: Path,
283 meta: dict[str, str] | None,
284 *,
285 available_bytes: int,
286 training_ctx: int,
287 ceiling: int,
288) -> int:
289 """Window from GGUF header arithmetic: weights plus a dense-attention cache."""
290 try:
291 model_bytes = model_path.stat().st_size
292 kv_per_tok = kv_bytes_per_token(meta, *chat_kv_elem_bytes())
293 return compute_dynamic_ctx(
294 model_bytes=model_bytes,
295 available_bytes=available_bytes,
296 training_ctx=training_ctx,
297 kv_bytes_per_tok=kv_per_tok,
298 ceiling=ceiling,
299 target=cfg.chat_n_ctx_target,
300 )
301 except (OSError, ValueError):
302 log.debug("dynamic ctx sizing failed for %s, using static cap", model_path, exc_info=True)
303 return min(training_ctx, cfg.chat_n_ctx_target)
306def resolve_chat_ctx(
307 model_path: Path, meta: dict[str, str] | None, *, available_bytes: int | None = None
308) -> int:
309 """The window half of :func:`resolve_chat_fit`, for callers that only serve
310 or advertise the context."""
311 return resolve_chat_fit(model_path, meta, available_bytes=available_bytes).ctx
314# Tokens the minimum grounded prompt allows for the question plus the context
315# template's framing, beyond the system prompt and one retrieved source.
316_GROUNDED_QUESTION_TOKENS = 128
319def min_usable_chat_ctx() -> int:
320 """Smallest chat window that serves one grounded answer: the system prompt,
321 one retrieved source, the question, and the generation reserve plus margin."""
322 return (
323 estimate_budget_tokens(cfg.rag_system_prompt)
324 + cfg.chunk_size
325 + _GROUNDED_QUESTION_TOKENS
326 + GENERATION_RESERVE_TOKENS
327 + CONTEXT_WINDOW_MARGIN_TOKENS
328 )
331def resolve_n_gpu_layers(*, embedding: bool) -> int:
332 """Resolve ``cfg.n_gpu_layers`` (None=all) to llama.cpp's offload integer.
334 Zero is honoured for every role. It is not a layer budget but the way a user
335 says "run this on the CPU", and the search roles used to take the
336 full-offload sentinel before the setting was read, so embed, rerank and
337 vision kept loading onto the GPU that had just been excluded.
339 Any other value is a chat-shaped budget and says nothing useful about a small
340 embedding model, which still offloads fully.
341 """
342 if cfg.n_gpu_layers == _N_GPU_LAYERS_NONE:
343 return _N_GPU_LAYERS_NONE
344 if embedding or cfg.n_gpu_layers is None:
345 return N_GPU_LAYERS_AUTO
346 return cfg.n_gpu_layers
349def resolve_vision_ctx(model_path: Path) -> int:
350 """Pick n_ctx for a vision OCR load: the model's training context, capped per page.
352 Uses the model's ``<arch>.context_length`` (not the chat-tuned ``cfg.num_ctx``: a
353 vision pass packs image-token embeddings plus the prompt, and a small chat ctx
354 truncates OCR output) but caps it at ``_VISION_PAGE_CTX_CAP``. OCR processes one page
355 per request, so a single page never exceeds the cap, yet a long-context VLM's full
356 context would otherwise estimate too large to place alongside a chat giant.
357 """
358 try:
359 meta = read_gguf_metadata(model_path)
360 except Exception:
361 log.debug("read_gguf_metadata failed for vision %s", model_path, exc_info=True)
362 meta = None
363 train_ctx = train_ctx_from_meta(meta, fallback=_VISION_FALLBACK_N_CTX, model_path=model_path)
364 return min(train_ctx, _VISION_PAGE_CTX_CAP)