Coverage for src/lilbee/providers/engine_params.py: 100%

119 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""Engine-neutral parameter helpers: model-path resolution, context and GPU-layer 

2sizing, and chat-option translation. 

3 

4These derive launch/generation parameters from cfg + a model's GGUF metadata 

5without loading the model, so both the local llama-server engine and any other 

6provider compute them identically. No native binding. 

7""" 

8 

9from __future__ import annotations 

10 

11import logging 

12from dataclasses import dataclass 

13from pathlib import Path 

14from typing import TYPE_CHECKING, Any 

15 

16if TYPE_CHECKING: 

17 from lilbee.modelhub.registry import ModelRegistry 

18from lilbee.core.config import DEFAULT_NUM_CTX, cfg 

19from lilbee.core.config.enums import KV_CACHE_TYPE_BYTES, KvCacheType 

20from lilbee.core.health_warnings import HealthWarning, WarningCode 

21from lilbee.providers.base import ( 

22 CONTEXT_WINDOW_MARGIN_TOKENS, 

23 GENERATION_RESERVE_TOKENS, 

24 ProviderError, 

25 ProviderErrorKind, 

26 estimate_budget_tokens, 

27 normalize_generation_options, 

28) 

29from lilbee.providers.gguf_meta import read_gguf_metadata, train_ctx_from_meta 

30from lilbee.providers.model_cache import ( 

31 compute_dynamic_ctx, 

32 get_available_memory, 

33 kv_bytes_per_token, 

34) 

35from lilbee.providers.model_ref import is_loose_model_file 

36 

37log = logging.getLogger(__name__) 

38 

39EMBED_FALLBACK_CTX = 2048 

40"""Context used for embed/rerank when a GGUF reports junk (e.g. context_length=0).""" 

41 

42# Sized above chunk_size so BOS re-added on re-tokenization doesn't overflow a full-chunk input. 

43_EMBED_CTX_MARGIN = 8 

44 

45 

46def resolve_embed_ctx(meta: dict[str, str] | None, model_path: Path) -> int: 

47 """Embed/rerank context: worst-case chunk tokenization, capped by trained context. 

48 

49 ``chunk_size`` is token-denominated but the chunker enforces a CHARACTER 

50 budget (``chunk_size * CHARS_PER_TOKEN``). A BPE token is at least one 

51 character, so that char budget is also the provable token ceiling for any 

52 chunk the chunker can emit: size the context to it and embed-time 

53 truncation becomes impossible. Token-dense text (numeric tables, dense 

54 identifiers) otherwise reaches ~2x chunk_size tokens against a 1x cap and 

55 silently loses its tail at embed time. A trained context below that budget 

56 caps the window instead, and the chunker then sizes chunks in the embedder's 

57 own tokens at the served cap (see :func:`embed_token_cap`).""" 

58 from lilbee.data.extract.chunk import CHARS_PER_TOKEN 

59 

60 train_ctx = train_ctx_from_meta(meta, fallback=EMBED_FALLBACK_CTX, model_path=model_path) 

61 return min(train_ctx, cfg.chunk_size * CHARS_PER_TOKEN + _EMBED_CTX_MARGIN) 

62 

63 

64def embed_token_cap(ctx: int) -> int: 

65 """Tokens an embed server truncates one input to, given its per-slot context.""" 

66 return max(1, ctx - _EMBED_CTX_MARGIN) 

67 

68 

69def embed_window_warning(cap: int) -> HealthWarning | None: 

70 """The embed-window warning for a real, actionable gap, else None. 

71 

72 Plain chunking self-corrects to token sizing when *cap* binds, so it loses 

73 no text and needs no warning. Semantic chunking ignores token sizing and 

74 stays character-sized, so a bound *cap* loses text there regardless of the 

75 ``token_sizing`` setting. 

76 """ 

77 from lilbee.data.extract.chunk import CHARS_PER_TOKEN 

78 

79 if cfg.semantic_chunking: 

80 chars = cfg.chunk_size * CHARS_PER_TOKEN 

81 if cap >= chars: 

82 return None 

83 return HealthWarning( 

84 code=WarningCode.EMBED_WINDOW_BELOW_CHUNK, 

85 message=( 

86 f"The embedding model accepts {cap} tokens per input, below the {chars} " 

87 f"characters that chunk_size {cfg.chunk_size} allows. Semantic chunking " 

88 f"sizes chunks by characters and ignores token sizing, so a chunk over " 

89 f"{cap} tokens loses text at embedding time." 

90 ), 

91 remedy="Turn off semantic_chunking or lower chunk_size.", 

92 ) 

93 if cfg.token_sizing and cap < cfg.chunk_size: 

94 return HealthWarning( 

95 code=WarningCode.EMBED_WINDOW_BELOW_CHUNK, 

96 message=( 

97 f"The embedding model accepts {cap} tokens per input, below the configured " 

98 f"chunk_size {cfg.chunk_size} in tokens. Chunks are capped to {cap} tokens " 

99 f"instead." 

100 ), 

101 remedy=f"Set chunk_size to {cap} to match the window.", 

102 ) 

103 return None 

104 

105 

106_LLM_RERANK_HEADROOM = 512 

107"""Tokens reserved above chunk_size for an LLM reranker's query, prompt, and 1-token answer.""" 

108 

109 

110def resolve_llm_rerank_ctx(meta: dict[str, str] | None, model_path: Path) -> int: 

111 """LLM-reranker context: a query+candidate pair, capped by the model's trained context.""" 

112 train_ctx = train_ctx_from_meta(meta, fallback=EMBED_FALLBACK_CTX, model_path=model_path) 

113 return min(train_ctx, cfg.chunk_size + _LLM_RERANK_HEADROOM) 

114 

115 

116_VISION_FALLBACK_N_CTX = 4096 

117"""Context for a vision load when the GGUF reports no usable context_length.""" 

118 

119_VISION_PAGE_CTX_CAP = 32768 

120"""Per-page ceiling on a vision OCR server's context: covers a single high-res page's 

121image tokens plus prompt, while keeping a long-context VLM placeable beside a chat giant.""" 

122 

123N_GPU_LAYERS_AUTO = -1 

124"""llama.cpp's "fit as many layers as the device holds" value for n_gpu_layers. 

125 

126The engine measures free VRAM at load and picks the count itself, spilling the 

127rest to system memory. That is a better answer than any number lilbee can 

128compute ahead of time, because it is taken on the real device after every other 

129tenant, so the planner passes this rather than a layer count of its own. 

130""" 

131# llama.cpp's "offload nothing"; the user's CPU-only opt-out rather than a budget. 

132_N_GPU_LAYERS_NONE = 0 

133 

134 

135def chat_options_to_kwargs(options: dict[str, Any] | None) -> dict[str, Any]: 

136 """Translate user-facing chat options into generation kwargs. 

137 

138 The output keys (``temperature``/``top_p``/``top_k``/``seed``/``max_tokens``/ 

139 ``repeat_penalty``) are accepted by llama-server's OpenAI body. ``top_k`` is 

140 kept (local llama.cpp honors it), unlike the SDK/API translator which drops it. 

141 ``think`` becomes ``chat_template_kwargs.enable_thinking``, which thinking 

142 templates honor and others ignore. 

143 """ 

144 kwargs = normalize_generation_options(options) 

145 think = kwargs.pop("think", None) 

146 if think is not None: 

147 kwargs["chat_template_kwargs"] = {"enable_thinking": think} 

148 return kwargs 

149 

150 

151def resolve_model_path(model: str, registry: ModelRegistry | None = None) -> Path: 

152 """Resolve a model name to a .gguf file path. 

153 

154 Resolution order: (1) registry (canonical source for installed models), 

155 (2) an absolute path to an existing file. Pass *registry* to resolve without 

156 reaching for ``get_services()`` (callers running inside its construction). 

157 """ 

158 if not model: 

159 raise ProviderError( 

160 "No model is configured for this role. Pick one from the catalog " 

161 "or run 'lilbee model pull <model>'.", 

162 provider="llama-server", 

163 kind=ProviderErrorKind.NOT_FOUND, 

164 ) 

165 if registry is None: 

166 # call-time import: keeps the app-layer container off this module's import graph 

167 from lilbee.app.services import get_services 

168 

169 registry = get_services().registry 

170 try: 

171 return registry.resolve(model) 

172 except (KeyError, ValueError): 

173 pass 

174 

175 if is_loose_model_file(model): 

176 return Path(model) 

177 if Path(model).is_absolute(): 

178 raise ProviderError( 

179 f"Model file not found: {model}", 

180 provider="llama-server", 

181 kind=ProviderErrorKind.NOT_FOUND, 

182 ) 

183 

184 raise ProviderError( 

185 f"Model {model!r} is not installed. Run 'lilbee model pull {model}' to download it.", 

186 provider="llama-server", 

187 kind=ProviderErrorKind.NOT_FOUND, 

188 ) 

189 

190 

191def chat_kv_elem_bytes() -> tuple[float, float]: 

192 """Per-element (K, V) byte costs of the KV cache a chat launch allocates. 

193 

194 Reads the same flags the launch passes 

195 (:func:`lilbee.providers.fleet.planning.chat_cache_type_flags`): K carries 

196 ``cfg.kv_cache_type``, while V carries it only when flash attention is 

197 certain to be on, because llama.cpp refuses a quantized V cache without it 

198 and the launch then leaves V at f16. Budgeting from the launch flags keeps 

199 the granted window in step with the cache the engine actually allocates. 

200 """ 

201 # call-time import: planning imports this module at load 

202 from lilbee.providers.fleet.planning import chat_cache_type_flags 

203 

204 def elem_bytes(flag: str | None) -> float: 

205 return KV_CACHE_TYPE_BYTES[KvCacheType(flag) if flag else KvCacheType.F16] 

206 

207 k_flag, v_flag = chat_cache_type_flags() 

208 return elem_bytes(k_flag), elem_bytes(v_flag) 

209 

210 

211def chat_ctx_ceiling(meta: dict[str, str] | None, model_path: Path) -> int: 

212 """Hard upper bound on a chat per-slot n_ctx: trained context, capped by ``cfg.num_ctx_max``.""" 

213 training_ctx = train_ctx_from_meta(meta, fallback=DEFAULT_NUM_CTX, model_path=model_path) 

214 if cfg.num_ctx_max is not None: 

215 return min(training_ctx, cfg.num_ctx_max) 

216 return training_ctx 

217 

218 

219@dataclass(frozen=True) 

220class ChatFit: 

221 """The GPU offload and per-slot window one chat launch runs with. 

222 

223 The pair is inseparable: the window was sized against the memory this 

224 offload leaves free, so serving it at any other offload overruns the card. 

225 """ 

226 

227 gpu_layers: int 

228 ctx: int 

229 

230 

231def resolve_chat_fit( 

232 model_path: Path, meta: dict[str, str] | None, *, available_bytes: int | None = None 

233) -> ChatFit: 

234 """Pick a single-GPU offload and n_ctx aiming for ``cfg.chat_n_ctx_target``, 

235 clamped to model + host. 

236 

237 A gguf-parser fit answers first 

238 (:func:`lilbee.providers.fleet.planning.fit_chat_ctx`), because it prices the 

239 cache each layer of this architecture holds, and it may leave layers in 

240 system memory to free the KV room a usable window needs. Header math takes 

241 over when the estimator cannot answer; it charges every layer as dense 

242 attention over the whole window at the configured offload, which 

243 under-grants linear-attention, sliding-window and MLA models. Either way the 

244 window stops at the smallest of the trained context, ``cfg.num_ctx_max`` and 

245 the target. 

246 

247 A multi-GPU tensor-split chat is sized separately by the fleet against its 

248 per-device headroom (see :func:`lilbee.providers.fleet.ctx.fit_split_ctx`). 

249 ``available_bytes`` overrides the live host-memory read, and every caller 

250 that is sizing a real launch passes it: the fleet and the surfaces that 

251 mirror it hand over 

252 :func:`lilbee.providers.fleet.planning.plan_sizing_budget`, which reports the 

253 memory of the GPU that will run the model and holds a clean-box snapshot so 

254 a reload sizes ctx like the boot did. 

255 """ 

256 # call-time import: planning imports this module at load 

257 from lilbee.providers.fleet.planning import fit_chat_ctx 

258 

259 training_ctx = train_ctx_from_meta(meta, fallback=DEFAULT_NUM_CTX, model_path=model_path) 

260 ceiling = cfg.num_ctx_max if cfg.num_ctx_max is not None else training_ctx 

261 if available_bytes is None: 

262 available_bytes = get_available_memory(cfg.gpu_memory_fraction) 

263 upper = min(training_ctx, ceiling, cfg.chat_n_ctx_target) 

264 

265 try: 

266 return fit_chat_ctx(model_path, meta, available_bytes=available_bytes, ctx_ceiling=upper) 

267 except (ProviderError, OSError, ValueError): 

268 log.debug("gguf-parser ctx fit failed for %s, using header math", model_path, exc_info=True) 

269 return ChatFit( 

270 resolve_n_gpu_layers(embedding=False), 

271 _header_math_chat_ctx( 

272 model_path, 

273 meta, 

274 available_bytes=available_bytes, 

275 training_ctx=training_ctx, 

276 ceiling=ceiling, 

277 ), 

278 ) 

279 

280 

281def _header_math_chat_ctx( 

282 model_path: Path, 

283 meta: dict[str, str] | None, 

284 *, 

285 available_bytes: int, 

286 training_ctx: int, 

287 ceiling: int, 

288) -> int: 

289 """Window from GGUF header arithmetic: weights plus a dense-attention cache.""" 

290 try: 

291 model_bytes = model_path.stat().st_size 

292 kv_per_tok = kv_bytes_per_token(meta, *chat_kv_elem_bytes()) 

293 return compute_dynamic_ctx( 

294 model_bytes=model_bytes, 

295 available_bytes=available_bytes, 

296 training_ctx=training_ctx, 

297 kv_bytes_per_tok=kv_per_tok, 

298 ceiling=ceiling, 

299 target=cfg.chat_n_ctx_target, 

300 ) 

301 except (OSError, ValueError): 

302 log.debug("dynamic ctx sizing failed for %s, using static cap", model_path, exc_info=True) 

303 return min(training_ctx, cfg.chat_n_ctx_target) 

304 

305 

306def resolve_chat_ctx( 

307 model_path: Path, meta: dict[str, str] | None, *, available_bytes: int | None = None 

308) -> int: 

309 """The window half of :func:`resolve_chat_fit`, for callers that only serve 

310 or advertise the context.""" 

311 return resolve_chat_fit(model_path, meta, available_bytes=available_bytes).ctx 

312 

313 

314# Tokens the minimum grounded prompt allows for the question plus the context 

315# template's framing, beyond the system prompt and one retrieved source. 

316_GROUNDED_QUESTION_TOKENS = 128 

317 

318 

319def min_usable_chat_ctx() -> int: 

320 """Smallest chat window that serves one grounded answer: the system prompt, 

321 one retrieved source, the question, and the generation reserve plus margin.""" 

322 return ( 

323 estimate_budget_tokens(cfg.rag_system_prompt) 

324 + cfg.chunk_size 

325 + _GROUNDED_QUESTION_TOKENS 

326 + GENERATION_RESERVE_TOKENS 

327 + CONTEXT_WINDOW_MARGIN_TOKENS 

328 ) 

329 

330 

331def resolve_n_gpu_layers(*, embedding: bool) -> int: 

332 """Resolve ``cfg.n_gpu_layers`` (None=all) to llama.cpp's offload integer. 

333 

334 Zero is honoured for every role. It is not a layer budget but the way a user 

335 says "run this on the CPU", and the search roles used to take the 

336 full-offload sentinel before the setting was read, so embed, rerank and 

337 vision kept loading onto the GPU that had just been excluded. 

338 

339 Any other value is a chat-shaped budget and says nothing useful about a small 

340 embedding model, which still offloads fully. 

341 """ 

342 if cfg.n_gpu_layers == _N_GPU_LAYERS_NONE: 

343 return _N_GPU_LAYERS_NONE 

344 if embedding or cfg.n_gpu_layers is None: 

345 return N_GPU_LAYERS_AUTO 

346 return cfg.n_gpu_layers 

347 

348 

349def resolve_vision_ctx(model_path: Path) -> int: 

350 """Pick n_ctx for a vision OCR load: the model's training context, capped per page. 

351 

352 Uses the model's ``<arch>.context_length`` (not the chat-tuned ``cfg.num_ctx``: a 

353 vision pass packs image-token embeddings plus the prompt, and a small chat ctx 

354 truncates OCR output) but caps it at ``_VISION_PAGE_CTX_CAP``. OCR processes one page 

355 per request, so a single page never exceeds the cap, yet a long-context VLM's full 

356 context would otherwise estimate too large to place alongside a chat giant. 

357 """ 

358 try: 

359 meta = read_gguf_metadata(model_path) 

360 except Exception: 

361 log.debug("read_gguf_metadata failed for vision %s", model_path, exc_info=True) 

362 meta = None 

363 train_ctx = train_ctx_from_meta(meta, fallback=_VISION_FALLBACK_N_CTX, model_path=model_path) 

364 return min(train_ctx, _VISION_PAGE_CTX_CAP)