Coverage for src/lilbee/app/placement.py: 100%

140 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""Surface-agnostic placement use-cases: inspect, preview, and set GPU placement.""" 

2 

3from __future__ import annotations 

4 

5import time 

6from collections.abc import Callable 

7from dataclasses import dataclass, replace 

8 

9from lilbee.app.services import peek_services 

10from lilbee.core import settings 

11from lilbee.core.config import cfg 

12from lilbee.providers.fleet.placement_spec import PlacementSpec 

13from lilbee.providers.fleet.planning import ( 

14 ResolvedPlacement, 

15 clear_read_device_cache, 

16 resolve_placement_plan, 

17) 

18from lilbee.providers.roles import EngineBackend, WorkerRole 

19from lilbee.providers.warm_progress import WarmPhase, WarmProgress, is_active_warm 

20 

21_PLACEMENT_KEY = "placement" 

22 

23# Ceiling and cadence for waiting out the post-reload chat warm: a cold 

24# tensor-split giant off a slow filesystem takes minutes, and the wait stops 

25# early when nothing is warming or the warm failed. The grace covers the gap 

26# between reload_placement returning and its off-thread warm stamping a phase. 

27_CHAT_READY_TIMEOUT_S = 1800.0 

28_CHAT_READY_POLL_S = 0.5 

29_CHAT_READY_GRACE_S = 3.0 

30 

31 

32@dataclass(frozen=True) 

33class GpuInfo: 

34 """One detected GPU as a surface can render it.""" 

35 

36 index: int 

37 backend: str 

38 label: str 

39 name: str 

40 total_bytes: int 

41 free_bytes: int 

42 

43 

44@dataclass(frozen=True) 

45class RolePlacementView: 

46 """Where one role's model is placed in the resolved plan.""" 

47 

48 role: WorkerRole 

49 model: str 

50 devices: tuple[int, ...] 

51 tensor_split: tuple[int, ...] | None 

52 replicas: int 

53 

54 

55@dataclass(frozen=True) 

56class SkippedRole: 

57 """A configured role left unplaced because its model isn't downloaded.""" 

58 

59 role: WorkerRole 

60 model: str 

61 

62 

63@dataclass(frozen=True) 

64class TightRole: 

65 """A placed role whose estimate exceeds the memory on the card it landed on.""" 

66 

67 role: WorkerRole 

68 shortfall_bytes: int 

69 

70 

71@dataclass(frozen=True) 

72class PlacementView: 

73 """The full placement picture: GPUs, per-role placement, and whether manual.""" 

74 

75 gpus: tuple[GpuInfo, ...] 

76 roles: tuple[RolePlacementView, ...] 

77 unplaceable: tuple[WorkerRole, ...] 

78 manual: bool 

79 spec_json: str | None 

80 # Configured roles absent from the plan because their model isn't installed, 

81 # so a surface can show "not downloaded" instead of an unexplained empty table. 

82 skipped_not_installed: tuple[SkippedRole, ...] = () 

83 # Roles sharing one swap group: each is placed, but only one is resident at a 

84 # time, so their footprints do not sum against the card they name. 

85 co_tenants: tuple[WorkerRole, ...] = () 

86 # A saved spec this hardware no longer satisfies. The auto plan is what runs, 

87 # but the spec stays in config.toml and reapplies once it fits again, so a 

88 # surface has to say it is there rather than report placement as plain auto. 

89 rejected_spec_json: str | None = None 

90 # Roles placed on a card that cannot hold them, with the shortfall in bytes. 

91 # They load on demand and may fail; a view that omits this shows them as 

92 # comfortably placed right up until they do. 

93 tight: tuple[TightRole, ...] = () 

94 # The backend the engine selected. A surface reports this rather than reading 

95 # it off ``gpus``: an empty device list is a CPU host and a failed probe 

96 # alike, and only UNKNOWN says which one this is. 

97 engine_backend: EngineBackend = EngineBackend.UNKNOWN 

98 

99 

100def _active_spec() -> PlacementSpec | None: 

101 raw = cfg.placement 

102 return PlacementSpec.from_json(raw) if raw else None 

103 

104 

105def _view( 

106 resolved: ResolvedPlacement, 

107 *, 

108 manual: bool, 

109 spec_json: str | None, 

110 rejected_spec_json: str | None = None, 

111) -> PlacementView: 

112 gpus = tuple( 

113 GpuInfo( 

114 index=d.index, 

115 backend=d.backend, 

116 label=f"{d.backend}{d.index}", 

117 name=d.name, 

118 total_bytes=d.total_bytes, 

119 free_bytes=d.free_bytes, 

120 ) 

121 for d in resolved.devices 

122 ) 

123 by_role: dict[WorkerRole, RolePlacementView] = {} 

124 for plan in resolved.instances: 

125 existing = by_role.get(plan.role) 

126 if existing is not None: 

127 devices = tuple(sorted(set(existing.devices) | set(plan.devices))) 

128 by_role[plan.role] = replace(existing, devices=devices, replicas=existing.replicas + 1) 

129 else: 

130 by_role[plan.role] = RolePlacementView( 

131 role=plan.role, 

132 model=resolved.model_refs.get(plan.role, ""), 

133 devices=plan.devices, 

134 tensor_split=plan.tensor_split or None, 

135 replicas=1, 

136 ) 

137 return PlacementView( 

138 gpus=gpus, 

139 roles=tuple(by_role.values()), 

140 unplaceable=resolved.unplaceable_roles, 

141 manual=manual, 

142 spec_json=spec_json, 

143 tight=tuple( 

144 TightRole(role=role, shortfall_bytes=shortfall) 

145 for role, shortfall in sorted(resolved.tight_roles.items(), key=lambda kv: kv[0].value) 

146 ), 

147 skipped_not_installed=tuple( 

148 SkippedRole(role=role, model=ref) 

149 for role, ref in resolved.skipped_not_installed.items() 

150 ), 

151 co_tenants=tuple(sorted(resolved.co_tenants, key=lambda role: role.value)), 

152 rejected_spec_json=rejected_spec_json, 

153 engine_backend=resolved.engine_backend, 

154 ) 

155 

156 

157def get_placement() -> PlacementView: 

158 """The current effective placement (manual if a spec is set, else auto). 

159 

160 A saved spec that no longer fits the hardware is not the effective placement: 

161 the fleet runs the auto plan, and this reports that rather than a manual layout 

162 nothing is using. 

163 """ 

164 spec = _active_spec() 

165 resolved = resolve_placement_plan(spec, fall_back_to_auto=True) 

166 if spec is None: 

167 return _view(resolved, manual=False, spec_json=None) 

168 if not resolved.spec_applied: 

169 return _view(resolved, manual=False, spec_json=None, rejected_spec_json=spec.to_json()) 

170 return _view(resolved, manual=True, spec_json=spec.to_json()) 

171 

172 

173def preview_placement(spec: PlacementSpec | None = None) -> PlacementView: 

174 """Dry-run: what spec (or auto, when None) would place. No persistence or reload.""" 

175 resolved = resolve_placement_plan(spec) 

176 return _view(resolved, manual=spec is not None, spec_json=spec.to_json() if spec else None) 

177 

178 

179def placement_refused_message() -> str: 

180 """Shared refusal for placement changes on the shared HTTP server. 

181 

182 Kept in one place so the REST routes and the HTTP-mounted MCP tools 

183 cannot drift apart. 

184 """ 

185 return ( 

186 "Changing placement on the HTTP server is unavailable: it rebuilds the shared " 

187 "fleet for every connected client. Enable allow_http_placement " 

188 "(LILBEE_ALLOW_HTTP_PLACEMENT) on a single-client deployment, or change it " 

189 "from the CLI or TUI." 

190 ) 

191 

192 

193def set_placement(spec: PlacementSpec | None) -> PlacementView: 

194 """Validate, persist to config.toml, apply to the live fleet, and return the new view. 

195 

196 Raises PlacementError before any write when the spec does not fit the hardware. 

197 The live fleet applies the change surgically (``reload_placement`` restarts 

198 only the roles whose placement moved), so an untouched role's loaded model 

199 stays resident; with no services built there is nothing running and the next 

200 use plans fresh. On the live path the planner re-plans against its clean-box 

201 plan snapshot (see ``planning.capture_plan_probe``): probing under a loaded 

202 fleet would report our own residency as unavailable and poison the chat 

203 context sizing, while charging stays against total capacity (bb-a8f). 

204 """ 

205 resolved = resolve_placement_plan(spec) 

206 if spec is None: 

207 settings.delete_values(cfg.data_root, [_PLACEMENT_KEY]) 

208 cfg.placement = None 

209 else: 

210 spec_json = spec.to_json() 

211 settings.update_values(cfg.data_root, {_PLACEMENT_KEY: spec_json}) 

212 cfg.placement = spec_json 

213 services = peek_services() 

214 if services is None: 

215 clear_read_device_cache() # nothing running; let the next boot probe fresh 

216 else: 

217 services.provider.reload_placement(wait=True) 

218 return _view(resolved, manual=spec is not None, spec_json=spec.to_json() if spec else None) 

219 

220 

221def wait_chat_ready( 

222 timeout_s: float = _CHAT_READY_TIMEOUT_S, 

223 *, 

224 on_progress: Callable[[WarmProgress], None] | None = None, 

225 should_abort: Callable[[], bool] | None = None, 

226) -> bool: 

227 """Block while a chat warm is in flight; True once a prompt can be served. 

228 

229 ``reload_placement(wait=True)`` returns once the proxies are healthy while the 

230 restarted model still warms off-thread, so a chat request sent right after an 

231 apply hits the busy 429 path. Callers that gate user input on the reload call 

232 this to hold until the model actually serves. Waits only while a warm is 

233 actively in flight: with no fleet, no warm, or a failed/finished warm it 

234 returns at once, so a change that never restarts chat cannot stall the caller. 

235 The brief grace covers the reload kicking its warm on a separate thread. 

236 

237 ``on_progress`` receives each actively-reporting warm snapshot so the caller 

238 can render the load. ``should_abort`` is polled every cycle; True ends the 

239 wait at once, so a cancelled prompt never pins its worker thread. 

240 """ 

241 services = peek_services() 

242 if services is None: 

243 return False 

244 provider = services.provider 

245 started = time.monotonic() 

246 deadline = started + timeout_s 

247 grace_deadline = started + _CHAT_READY_GRACE_S 

248 while time.monotonic() < deadline: 

249 if provider.role_ready(WorkerRole.CHAT): 

250 return True 

251 if should_abort is not None and should_abort(): 

252 return False 

253 snapshot = provider.warm_progress() 

254 # A requested warm counts as in flight before it stamps a phase: the fleet 

255 # spawns and health-checks llama-swap first, which takes seconds. 

256 if is_active_warm(snapshot): 

257 if on_progress is not None and snapshot is not None: 

258 on_progress(snapshot) 

259 grace_deadline = time.monotonic() + _CHAT_READY_GRACE_S 

260 elif provider.warm_pending(): 

261 grace_deadline = time.monotonic() + _CHAT_READY_GRACE_S 

262 elif time.monotonic() > grace_deadline: 

263 return False 

264 time.sleep(_CHAT_READY_POLL_S) 

265 return False 

266 

267 

268def request_engine_warm() -> None: 

269 """Kick the provider's warm-up when nothing is loaded or loading. 

270 

271 ``warm_up_pool`` is idempotent (a no-op while a warm is in flight or the 

272 fleet is up), so a prompt sent after a failed boot warm drives a fresh 

273 engine start instead of bouncing for the rest of the session. 

274 """ 

275 services = peek_services() 

276 if services is None: 

277 return 

278 services.provider.warm_up_pool() 

279 

280 

281def chat_engine_ready() -> bool: 

282 """Whether a chat prompt can be served right now. 

283 

284 Positive readiness, not absence-of-warm: before the services container is 

285 built nothing is loading yet and nothing can answer, which a warm snapshot 

286 cannot distinguish from a finished load. 

287 """ 

288 services = peek_services() 

289 if services is None: 

290 return False 

291 return services.provider.role_ready(WorkerRole.CHAT) 

292 

293 

294def active_chat_warm_progress() -> WarmProgress | None: 

295 """The chat warm snapshot while a cold load is genuinely in flight, else None. 

296 

297 A surface gates interactive input on this: ``None`` covers ready, no fleet, a 

298 missing model, and a finished or failed warm, so nothing traps the input in a 

299 locked state. Non-``None`` carries the phase and byte progress to render. 

300 

301 The in-process warm snapshot is checked before ``role_ready`` because the task 

302 bar polls this on every tick: the snapshot is a free attribute read, while 

303 ``role_ready`` is an HTTP probe of the engine. Ordering it this way keeps the 

304 probe to the seconds a load is actually in flight instead of firing forever on 

305 an idle TUI. The two orders return the same answer. 

306 """ 

307 services = peek_services() 

308 if services is None: 

309 return None 

310 provider = services.provider 

311 snapshot = provider.warm_progress() 

312 if not is_active_warm(snapshot): 

313 return None 

314 return None if provider.role_ready(WorkerRole.CHAT) else snapshot 

315 

316 

317def chat_warm_error() -> str | None: 

318 """The failed chat warm's error text, or None when no failure is on record.""" 

319 services = peek_services() 

320 if services is None: 

321 return None 

322 snapshot = services.provider.warm_progress() 

323 if snapshot is not None and snapshot.phase is WarmPhase.ERROR: 

324 return snapshot.error or "" 

325 return None