Coverage for src/lilbee/providers/fleet/binary.py: 100%

91 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""Resolve the bundled engine binaries (llama-server, llama-swap, gguf-parser).""" 

2 

3from __future__ import annotations 

4 

5import hashlib 

6import shutil 

7from enum import StrEnum 

8from importlib.metadata import version as _pkg_version 

9from pathlib import Path 

10 

11from lilbee.providers.base import ProviderError, ProviderErrorKind 

12 

13# Every index is spelled out rather than pointing at the README. This is read at 

14# the moment the engine is needed, often on a remote box. One index per hardware 

15# because the builds are not interchangeable: cpu is built with every GPU backend 

16# off, metal only ships a macOS arm64 wheel, and vulkan only Linux/Windows ones, 

17# so naming a single "default" hands somebody an engine that ignores their GPU. 

18_INSTALL_HINT = ( 

19 "The engine is lilbee's 'engine' extra, published on lilbee.sh rather than " 

20 "PyPI, so the index is part of the command:\n" 

21 " NVIDIA (CUDA): pip install --pre 'lilbee[engine]' --extra-index-url https://lilbee.sh/cu125/\n" 

22 " AMD (ROCm): pip install --pre 'lilbee[engine]' --extra-index-url https://lilbee.sh/rocm/\n" 

23 " Apple silicon: pip install --pre 'lilbee[engine]' --extra-index-url https://lilbee.sh/metal/\n" 

24 " Other GPUs: pip install --pre 'lilbee[engine]' --extra-index-url https://lilbee.sh/vulkan/\n" 

25 " No GPU: pip install --pre 'lilbee[engine]' --extra-index-url https://lilbee.sh/cpu/\n" 

26 "A standalone binary from https://github.com/tobocop2/lilbee/releases/latest " 

27 "bundles the engine instead. To bring your own, set LILBEE_LLAMA_SERVER_PATH to " 

28 "a llama-server binary and put llama-swap / gguf-parser on PATH." 

29) 

30 

31 

32class EngineTool(StrEnum): 

33 """A bundled engine executable resolved from the ``lilbee-engine`` wheel.""" 

34 

35 LLAMA_SERVER = "llama-server" 

36 LLAMA_SWAP = "llama-swap" 

37 GGUF_PARSER = "gguf-parser" 

38 

39 

40_BUNDLED_ACCESSORS = { 

41 EngineTool.LLAMA_SERVER: "get_llama_server_path", 

42 EngineTool.LLAMA_SWAP: "get_llama_swap_path", 

43 EngineTool.GGUF_PARSER: "get_gguf_parser_path", 

44} 

45 

46 

47def _bundled_tool(tool: EngineTool) -> Path | None: 

48 """Path to *tool* from the ``lilbee-engine`` wheel, or ``None`` if absent.""" 

49 try: 

50 import lilbee_engine 

51 except ImportError: 

52 return None 

53 accessor = getattr(lilbee_engine, _BUNDLED_ACCESSORS[tool], None) 

54 if accessor is None: # a wheel that predates this tool lacks the accessor 

55 return None 

56 path = Path(accessor()) 

57 return path if path.is_file() else None 

58 

59 

60def engine_pin() -> str: 

61 """Identity of the engine this lilbee would spawn; sharing keys on it. 

62 

63 Two dimensions must match for two processes to share one engine: the engine 

64 BUILD (a configured ``LILBEE_LLAMA_SERVER_PATH`` is its own identity so a 

65 bring-your-own engine never silently shares with a bundled one) and the 

66 load-affecting CONFIG baked into the launch argv (kv-cache type, expert 

67 offload, n-gpu-layers, ctx target, ...). A process whose load config differs 

68 computes a different pin, so ``contract_matches`` refuses the bind and it 

69 overflows to its own engine rather than silently running on the incumbent's 

70 flags. Total: never raises, because it runs on every state write. 

71 """ 

72 return f"{engine_build_id()}|{_load_config_signature()}" 

73 

74 

75def engine_build_id() -> str: 

76 """The engine build's identity: configured path, wheel pin, PATH, or unpinned. 

77 

78 A BYO (``custom:``) or PATH-resolved (``path:``) binary is identified by its 

79 location AND a cheap build fingerprint (size + mtime), so replacing the binary 

80 in place (a brew upgrade, a re-download) changes the pin and never binds a new 

81 process to an engine spawned from the old build. The bundled wheel needs no 

82 fingerprint: its pin already encodes the build. 

83 """ 

84 from lilbee.core.config import cfg 

85 

86 if cfg.llama_server_path: 

87 return f"custom:{cfg.llama_server_path}@{_binary_signature(Path(cfg.llama_server_path))}" 

88 try: 

89 import lilbee_engine 

90 except ImportError: 

91 lilbee_engine = None 

92 if lilbee_engine is not None: 

93 try: 

94 return str(lilbee_engine.get_engine_pin()) 

95 except AttributeError: # pre-pin wheels lack the accessor 

96 return f"wheel:{_engine_wheel_version()}" 

97 found = shutil.which(EngineTool.LLAMA_SERVER.value) 

98 if found is not None: 

99 return f"path:{found}@{_binary_signature(Path(found))}" 

100 return "unpinned" 

101 

102 

103def _engine_wheel_version() -> str: 

104 """The engine wheel's version, or a marker when it has no distribution metadata. 

105 

106 ``lilbee_engine`` can be importable with nothing to look up: an extracted 

107 wheel on sys.path, a vendored copy, or a distribution registered under a name 

108 that does not normalize to ``lilbee-engine``. Since this feeds the pin, and 

109 the pin is computed on every state write, a missing version degrades to a 

110 marker rather than raising out of ``engine_pin``. 

111 """ 

112 from importlib.metadata import PackageNotFoundError 

113 

114 try: 

115 return _pkg_version("lilbee-engine") 

116 except PackageNotFoundError: 

117 return "unknown" 

118 

119 

120def _binary_signature(path: Path) -> str: 

121 """A cheap build fingerprint of the binary at *path*: size and mtime. 

122 

123 An in-place replacement changes both, so the pin stops matching the old build. 

124 Best-effort and total (engine_pin runs on every state write): an unstatable 

125 path degrades to a fixed marker rather than raising. 

126 """ 

127 try: 

128 st = path.stat() 

129 except OSError: 

130 return "unstatable" 

131 return f"{st.st_size}-{st.st_mtime_ns}" 

132 

133 

134def engine_binary_identity(binary: Path) -> str: 

135 """Identity of the engine file at *binary*: its location and a digest of its bytes. 

136 

137 The file rather than the wheel, because a stub wheel can carry the version of 

138 the real one; only the bytes that answered a probe identify what answered it. 

139 The bytes rather than the stat fields ``engine_pin`` matches on: an upgrade 

140 rewrites the file in place, the rewrite keeps the inode, and a build of the 

141 same size written inside one mtime tick moves no stat field at all, so 

142 metadata can repeat across a real engine change. ``llama-server`` is a 

143 launcher of a few tens of kilobytes that loads the backend libraries at run 

144 time, and this is read once per planning pass, so the digest costs far less 

145 than the device probe it decides to re-run. 

146 """ 

147 try: 

148 with binary.open("rb") as handle: 

149 return f"{binary}@{hashlib.file_digest(handle, 'sha256').hexdigest()}" 

150 except OSError: 

151 return f"{binary}@unreadable" 

152 

153 

154# Ctx sizing keys share by window coverage (contract.chat_ctx_covers), not 

155# value equality: a running window that covers the demand serves both peers. 

156# chat_n_ctx_target in particular defaults per process from its cgroup-capped 

157# RAM, so exact equality here restarted a warm engine per co-tenant. 

158_CTX_SIZING_KEYS = frozenset({"num_ctx", "num_ctx_max", "chat_n_ctx_target"}) 

159 

160 

161def _load_config_signature() -> str: 

162 """A deterministic digest of the settings an engine bakes in at launch. 

163 

164 These decide cross-process sharing, since an engine launched with one set 

165 cannot serve a peer that configured another: the ``LOAD_AFFECTING_KEYS`` a 

166 single process reloads on (minus the ctx sizing keys, matched by coverage 

167 instead), plus the placement keys that fix which devices a launch uses, so 

168 a peer with different placement binds its own engine. 

169 """ 

170 from lilbee.core.config import cfg 

171 from lilbee.core.config.keys import LOAD_AFFECTING_KEYS, PLACEMENT_PIN_KEYS 

172 

173 keys = (LOAD_AFFECTING_KEYS - _CTX_SIZING_KEYS) | PLACEMENT_PIN_KEYS 

174 return ";".join(f"{key}={getattr(cfg, key, None)}" for key in sorted(keys)) 

175 

176 

177def resolve_engine_tool(tool: EngineTool) -> Path: 

178 """Resolve *tool*: configured llama-server path, then bundled wheel, then PATH. 

179 

180 Never downloads anything; the binaries arrive via the bundled ``lilbee-engine`` 

181 wheel or bring-your-own. Only llama-server honors ``LILBEE_LLAMA_SERVER_PATH`` 

182 (an explicit setting beats the bundled wheel); the other tools resolve from the 

183 wheel, then ``PATH``. 

184 """ 

185 if tool is EngineTool.LLAMA_SERVER: 

186 from lilbee.core.config import cfg 

187 

188 if cfg.llama_server_path: 

189 configured = Path(cfg.llama_server_path) 

190 if not configured.is_file(): 

191 raise ProviderError(f"LILBEE_LLAMA_SERVER_PATH is not a file: {configured}") 

192 return configured 

193 

194 bundled = _bundled_tool(tool) 

195 if bundled is not None: 

196 return bundled 

197 

198 found = shutil.which(tool.value) 

199 if found is not None: 

200 return Path(found) 

201 

202 # Only llama-server carries NOT_FOUND: it marks the engine-less host that 

203 # legitimately serves nothing. A missing sibling tool (gguf-parser) must not 

204 # take that kind, or the sizing fallback would misreport it as a model that 

205 # isn't installed. 

206 kind = ( 

207 ProviderErrorKind.NOT_FOUND 

208 if tool is EngineTool.LLAMA_SERVER 

209 else ProviderErrorKind.UNKNOWN 

210 ) 

211 raise ProviderError(f"{tool.value} binary not found. {_INSTALL_HINT}", kind=kind) 

212 

213 

214def resolve_llama_server() -> Path: 

215 """Resolve the ``llama-server`` executable.""" 

216 return resolve_engine_tool(EngineTool.LLAMA_SERVER) 

217 

218 

219def resolve_llama_swap() -> Path: 

220 """Resolve the ``llama-swap`` executable.""" 

221 return resolve_engine_tool(EngineTool.LLAMA_SWAP) 

222 

223 

224def resolve_gguf_parser() -> Path: 

225 """Resolve the ``gguf-parser`` executable.""" 

226 return resolve_engine_tool(EngineTool.GGUF_PARSER) 

227 

228 

229def llama_server_runtime_env() -> dict[str, str]: 

230 """Extra environment for a spawned ``llama-server``. 

231 

232 The bundled wheel ships its own ggml/llama/mtmd next to the binary with a baked 

233 rpath (``@loader_path`` on macOS, ``$ORIGIN`` on Linux), but a CUDA build also 

234 links the CUDA 12 runtime, which driver-only GPU images omit. On Linux this 

235 adds any installed CUDA-runtime wheel libs to ``LD_LIBRARY_PATH``; elsewhere it 

236 is empty. 

237 """ 

238 from lilbee.providers.fleet.cuda_runtime import cuda_runtime_env 

239 

240 return cuda_runtime_env()