Coverage for src/lilbee/providers/fleet/binary.py: 100%
91 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
1"""Resolve the bundled engine binaries (llama-server, llama-swap, gguf-parser)."""
3from __future__ import annotations
5import hashlib
6import shutil
7from enum import StrEnum
8from importlib.metadata import version as _pkg_version
9from pathlib import Path
11from lilbee.providers.base import ProviderError, ProviderErrorKind
13# Every index is spelled out rather than pointing at the README. This is read at
14# the moment the engine is needed, often on a remote box. One index per hardware
15# because the builds are not interchangeable: cpu is built with every GPU backend
16# off, metal only ships a macOS arm64 wheel, and vulkan only Linux/Windows ones,
17# so naming a single "default" hands somebody an engine that ignores their GPU.
18_INSTALL_HINT = (
19 "The engine is lilbee's 'engine' extra, published on lilbee.sh rather than "
20 "PyPI, so the index is part of the command:\n"
21 " NVIDIA (CUDA): pip install --pre 'lilbee[engine]' --extra-index-url https://lilbee.sh/cu125/\n"
22 " AMD (ROCm): pip install --pre 'lilbee[engine]' --extra-index-url https://lilbee.sh/rocm/\n"
23 " Apple silicon: pip install --pre 'lilbee[engine]' --extra-index-url https://lilbee.sh/metal/\n"
24 " Other GPUs: pip install --pre 'lilbee[engine]' --extra-index-url https://lilbee.sh/vulkan/\n"
25 " No GPU: pip install --pre 'lilbee[engine]' --extra-index-url https://lilbee.sh/cpu/\n"
26 "A standalone binary from https://github.com/tobocop2/lilbee/releases/latest "
27 "bundles the engine instead. To bring your own, set LILBEE_LLAMA_SERVER_PATH to "
28 "a llama-server binary and put llama-swap / gguf-parser on PATH."
29)
32class EngineTool(StrEnum):
33 """A bundled engine executable resolved from the ``lilbee-engine`` wheel."""
35 LLAMA_SERVER = "llama-server"
36 LLAMA_SWAP = "llama-swap"
37 GGUF_PARSER = "gguf-parser"
40_BUNDLED_ACCESSORS = {
41 EngineTool.LLAMA_SERVER: "get_llama_server_path",
42 EngineTool.LLAMA_SWAP: "get_llama_swap_path",
43 EngineTool.GGUF_PARSER: "get_gguf_parser_path",
44}
47def _bundled_tool(tool: EngineTool) -> Path | None:
48 """Path to *tool* from the ``lilbee-engine`` wheel, or ``None`` if absent."""
49 try:
50 import lilbee_engine
51 except ImportError:
52 return None
53 accessor = getattr(lilbee_engine, _BUNDLED_ACCESSORS[tool], None)
54 if accessor is None: # a wheel that predates this tool lacks the accessor
55 return None
56 path = Path(accessor())
57 return path if path.is_file() else None
60def engine_pin() -> str:
61 """Identity of the engine this lilbee would spawn; sharing keys on it.
63 Two dimensions must match for two processes to share one engine: the engine
64 BUILD (a configured ``LILBEE_LLAMA_SERVER_PATH`` is its own identity so a
65 bring-your-own engine never silently shares with a bundled one) and the
66 load-affecting CONFIG baked into the launch argv (kv-cache type, expert
67 offload, n-gpu-layers, ctx target, ...). A process whose load config differs
68 computes a different pin, so ``contract_matches`` refuses the bind and it
69 overflows to its own engine rather than silently running on the incumbent's
70 flags. Total: never raises, because it runs on every state write.
71 """
72 return f"{engine_build_id()}|{_load_config_signature()}"
75def engine_build_id() -> str:
76 """The engine build's identity: configured path, wheel pin, PATH, or unpinned.
78 A BYO (``custom:``) or PATH-resolved (``path:``) binary is identified by its
79 location AND a cheap build fingerprint (size + mtime), so replacing the binary
80 in place (a brew upgrade, a re-download) changes the pin and never binds a new
81 process to an engine spawned from the old build. The bundled wheel needs no
82 fingerprint: its pin already encodes the build.
83 """
84 from lilbee.core.config import cfg
86 if cfg.llama_server_path:
87 return f"custom:{cfg.llama_server_path}@{_binary_signature(Path(cfg.llama_server_path))}"
88 try:
89 import lilbee_engine
90 except ImportError:
91 lilbee_engine = None
92 if lilbee_engine is not None:
93 try:
94 return str(lilbee_engine.get_engine_pin())
95 except AttributeError: # pre-pin wheels lack the accessor
96 return f"wheel:{_engine_wheel_version()}"
97 found = shutil.which(EngineTool.LLAMA_SERVER.value)
98 if found is not None:
99 return f"path:{found}@{_binary_signature(Path(found))}"
100 return "unpinned"
103def _engine_wheel_version() -> str:
104 """The engine wheel's version, or a marker when it has no distribution metadata.
106 ``lilbee_engine`` can be importable with nothing to look up: an extracted
107 wheel on sys.path, a vendored copy, or a distribution registered under a name
108 that does not normalize to ``lilbee-engine``. Since this feeds the pin, and
109 the pin is computed on every state write, a missing version degrades to a
110 marker rather than raising out of ``engine_pin``.
111 """
112 from importlib.metadata import PackageNotFoundError
114 try:
115 return _pkg_version("lilbee-engine")
116 except PackageNotFoundError:
117 return "unknown"
120def _binary_signature(path: Path) -> str:
121 """A cheap build fingerprint of the binary at *path*: size and mtime.
123 An in-place replacement changes both, so the pin stops matching the old build.
124 Best-effort and total (engine_pin runs on every state write): an unstatable
125 path degrades to a fixed marker rather than raising.
126 """
127 try:
128 st = path.stat()
129 except OSError:
130 return "unstatable"
131 return f"{st.st_size}-{st.st_mtime_ns}"
134def engine_binary_identity(binary: Path) -> str:
135 """Identity of the engine file at *binary*: its location and a digest of its bytes.
137 The file rather than the wheel, because a stub wheel can carry the version of
138 the real one; only the bytes that answered a probe identify what answered it.
139 The bytes rather than the stat fields ``engine_pin`` matches on: an upgrade
140 rewrites the file in place, the rewrite keeps the inode, and a build of the
141 same size written inside one mtime tick moves no stat field at all, so
142 metadata can repeat across a real engine change. ``llama-server`` is a
143 launcher of a few tens of kilobytes that loads the backend libraries at run
144 time, and this is read once per planning pass, so the digest costs far less
145 than the device probe it decides to re-run.
146 """
147 try:
148 with binary.open("rb") as handle:
149 return f"{binary}@{hashlib.file_digest(handle, 'sha256').hexdigest()}"
150 except OSError:
151 return f"{binary}@unreadable"
154# Ctx sizing keys share by window coverage (contract.chat_ctx_covers), not
155# value equality: a running window that covers the demand serves both peers.
156# chat_n_ctx_target in particular defaults per process from its cgroup-capped
157# RAM, so exact equality here restarted a warm engine per co-tenant.
158_CTX_SIZING_KEYS = frozenset({"num_ctx", "num_ctx_max", "chat_n_ctx_target"})
161def _load_config_signature() -> str:
162 """A deterministic digest of the settings an engine bakes in at launch.
164 These decide cross-process sharing, since an engine launched with one set
165 cannot serve a peer that configured another: the ``LOAD_AFFECTING_KEYS`` a
166 single process reloads on (minus the ctx sizing keys, matched by coverage
167 instead), plus the placement keys that fix which devices a launch uses, so
168 a peer with different placement binds its own engine.
169 """
170 from lilbee.core.config import cfg
171 from lilbee.core.config.keys import LOAD_AFFECTING_KEYS, PLACEMENT_PIN_KEYS
173 keys = (LOAD_AFFECTING_KEYS - _CTX_SIZING_KEYS) | PLACEMENT_PIN_KEYS
174 return ";".join(f"{key}={getattr(cfg, key, None)}" for key in sorted(keys))
177def resolve_engine_tool(tool: EngineTool) -> Path:
178 """Resolve *tool*: configured llama-server path, then bundled wheel, then PATH.
180 Never downloads anything; the binaries arrive via the bundled ``lilbee-engine``
181 wheel or bring-your-own. Only llama-server honors ``LILBEE_LLAMA_SERVER_PATH``
182 (an explicit setting beats the bundled wheel); the other tools resolve from the
183 wheel, then ``PATH``.
184 """
185 if tool is EngineTool.LLAMA_SERVER:
186 from lilbee.core.config import cfg
188 if cfg.llama_server_path:
189 configured = Path(cfg.llama_server_path)
190 if not configured.is_file():
191 raise ProviderError(f"LILBEE_LLAMA_SERVER_PATH is not a file: {configured}")
192 return configured
194 bundled = _bundled_tool(tool)
195 if bundled is not None:
196 return bundled
198 found = shutil.which(tool.value)
199 if found is not None:
200 return Path(found)
202 # Only llama-server carries NOT_FOUND: it marks the engine-less host that
203 # legitimately serves nothing. A missing sibling tool (gguf-parser) must not
204 # take that kind, or the sizing fallback would misreport it as a model that
205 # isn't installed.
206 kind = (
207 ProviderErrorKind.NOT_FOUND
208 if tool is EngineTool.LLAMA_SERVER
209 else ProviderErrorKind.UNKNOWN
210 )
211 raise ProviderError(f"{tool.value} binary not found. {_INSTALL_HINT}", kind=kind)
214def resolve_llama_server() -> Path:
215 """Resolve the ``llama-server`` executable."""
216 return resolve_engine_tool(EngineTool.LLAMA_SERVER)
219def resolve_llama_swap() -> Path:
220 """Resolve the ``llama-swap`` executable."""
221 return resolve_engine_tool(EngineTool.LLAMA_SWAP)
224def resolve_gguf_parser() -> Path:
225 """Resolve the ``gguf-parser`` executable."""
226 return resolve_engine_tool(EngineTool.GGUF_PARSER)
229def llama_server_runtime_env() -> dict[str, str]:
230 """Extra environment for a spawned ``llama-server``.
232 The bundled wheel ships its own ggml/llama/mtmd next to the binary with a baked
233 rpath (``@loader_path`` on macOS, ``$ORIGIN`` on Linux), but a CUDA build also
234 links the CUDA 12 runtime, which driver-only GPU images omit. On Linux this
235 adds any installed CUDA-runtime wheel libs to ``LD_LIBRARY_PATH``; elsewhere it
236 is empty.
237 """
238 from lilbee.providers.fleet.cuda_runtime import cuda_runtime_env
240 return cuda_runtime_env()