Coverage for src/lilbee/app/placement.py: 100%
140 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
1"""Surface-agnostic placement use-cases: inspect, preview, and set GPU placement."""
3from __future__ import annotations
5import time
6from collections.abc import Callable
7from dataclasses import dataclass, replace
9from lilbee.app.services import peek_services
10from lilbee.core import settings
11from lilbee.core.config import cfg
12from lilbee.providers.fleet.placement_spec import PlacementSpec
13from lilbee.providers.fleet.planning import (
14 ResolvedPlacement,
15 clear_read_device_cache,
16 resolve_placement_plan,
17)
18from lilbee.providers.roles import EngineBackend, WorkerRole
19from lilbee.providers.warm_progress import WarmPhase, WarmProgress, is_active_warm
21_PLACEMENT_KEY = "placement"
23# Ceiling and cadence for waiting out the post-reload chat warm: a cold
24# tensor-split giant off a slow filesystem takes minutes, and the wait stops
25# early when nothing is warming or the warm failed. The grace covers the gap
26# between reload_placement returning and its off-thread warm stamping a phase.
27_CHAT_READY_TIMEOUT_S = 1800.0
28_CHAT_READY_POLL_S = 0.5
29_CHAT_READY_GRACE_S = 3.0
32@dataclass(frozen=True)
33class GpuInfo:
34 """One detected GPU as a surface can render it."""
36 index: int
37 backend: str
38 label: str
39 name: str
40 total_bytes: int
41 free_bytes: int
44@dataclass(frozen=True)
45class RolePlacementView:
46 """Where one role's model is placed in the resolved plan."""
48 role: WorkerRole
49 model: str
50 devices: tuple[int, ...]
51 tensor_split: tuple[int, ...] | None
52 replicas: int
55@dataclass(frozen=True)
56class SkippedRole:
57 """A configured role left unplaced because its model isn't downloaded."""
59 role: WorkerRole
60 model: str
63@dataclass(frozen=True)
64class TightRole:
65 """A placed role whose estimate exceeds the memory on the card it landed on."""
67 role: WorkerRole
68 shortfall_bytes: int
71@dataclass(frozen=True)
72class PlacementView:
73 """The full placement picture: GPUs, per-role placement, and whether manual."""
75 gpus: tuple[GpuInfo, ...]
76 roles: tuple[RolePlacementView, ...]
77 unplaceable: tuple[WorkerRole, ...]
78 manual: bool
79 spec_json: str | None
80 # Configured roles absent from the plan because their model isn't installed,
81 # so a surface can show "not downloaded" instead of an unexplained empty table.
82 skipped_not_installed: tuple[SkippedRole, ...] = ()
83 # Roles sharing one swap group: each is placed, but only one is resident at a
84 # time, so their footprints do not sum against the card they name.
85 co_tenants: tuple[WorkerRole, ...] = ()
86 # A saved spec this hardware no longer satisfies. The auto plan is what runs,
87 # but the spec stays in config.toml and reapplies once it fits again, so a
88 # surface has to say it is there rather than report placement as plain auto.
89 rejected_spec_json: str | None = None
90 # Roles placed on a card that cannot hold them, with the shortfall in bytes.
91 # They load on demand and may fail; a view that omits this shows them as
92 # comfortably placed right up until they do.
93 tight: tuple[TightRole, ...] = ()
94 # The backend the engine selected. A surface reports this rather than reading
95 # it off ``gpus``: an empty device list is a CPU host and a failed probe
96 # alike, and only UNKNOWN says which one this is.
97 engine_backend: EngineBackend = EngineBackend.UNKNOWN
100def _active_spec() -> PlacementSpec | None:
101 raw = cfg.placement
102 return PlacementSpec.from_json(raw) if raw else None
105def _view(
106 resolved: ResolvedPlacement,
107 *,
108 manual: bool,
109 spec_json: str | None,
110 rejected_spec_json: str | None = None,
111) -> PlacementView:
112 gpus = tuple(
113 GpuInfo(
114 index=d.index,
115 backend=d.backend,
116 label=f"{d.backend}{d.index}",
117 name=d.name,
118 total_bytes=d.total_bytes,
119 free_bytes=d.free_bytes,
120 )
121 for d in resolved.devices
122 )
123 by_role: dict[WorkerRole, RolePlacementView] = {}
124 for plan in resolved.instances:
125 existing = by_role.get(plan.role)
126 if existing is not None:
127 devices = tuple(sorted(set(existing.devices) | set(plan.devices)))
128 by_role[plan.role] = replace(existing, devices=devices, replicas=existing.replicas + 1)
129 else:
130 by_role[plan.role] = RolePlacementView(
131 role=plan.role,
132 model=resolved.model_refs.get(plan.role, ""),
133 devices=plan.devices,
134 tensor_split=plan.tensor_split or None,
135 replicas=1,
136 )
137 return PlacementView(
138 gpus=gpus,
139 roles=tuple(by_role.values()),
140 unplaceable=resolved.unplaceable_roles,
141 manual=manual,
142 spec_json=spec_json,
143 tight=tuple(
144 TightRole(role=role, shortfall_bytes=shortfall)
145 for role, shortfall in sorted(resolved.tight_roles.items(), key=lambda kv: kv[0].value)
146 ),
147 skipped_not_installed=tuple(
148 SkippedRole(role=role, model=ref)
149 for role, ref in resolved.skipped_not_installed.items()
150 ),
151 co_tenants=tuple(sorted(resolved.co_tenants, key=lambda role: role.value)),
152 rejected_spec_json=rejected_spec_json,
153 engine_backend=resolved.engine_backend,
154 )
157def get_placement() -> PlacementView:
158 """The current effective placement (manual if a spec is set, else auto).
160 A saved spec that no longer fits the hardware is not the effective placement:
161 the fleet runs the auto plan, and this reports that rather than a manual layout
162 nothing is using.
163 """
164 spec = _active_spec()
165 resolved = resolve_placement_plan(spec, fall_back_to_auto=True)
166 if spec is None:
167 return _view(resolved, manual=False, spec_json=None)
168 if not resolved.spec_applied:
169 return _view(resolved, manual=False, spec_json=None, rejected_spec_json=spec.to_json())
170 return _view(resolved, manual=True, spec_json=spec.to_json())
173def preview_placement(spec: PlacementSpec | None = None) -> PlacementView:
174 """Dry-run: what spec (or auto, when None) would place. No persistence or reload."""
175 resolved = resolve_placement_plan(spec)
176 return _view(resolved, manual=spec is not None, spec_json=spec.to_json() if spec else None)
179def placement_refused_message() -> str:
180 """Shared refusal for placement changes on the shared HTTP server.
182 Kept in one place so the REST routes and the HTTP-mounted MCP tools
183 cannot drift apart.
184 """
185 return (
186 "Changing placement on the HTTP server is unavailable: it rebuilds the shared "
187 "fleet for every connected client. Enable allow_http_placement "
188 "(LILBEE_ALLOW_HTTP_PLACEMENT) on a single-client deployment, or change it "
189 "from the CLI or TUI."
190 )
193def set_placement(spec: PlacementSpec | None) -> PlacementView:
194 """Validate, persist to config.toml, apply to the live fleet, and return the new view.
196 Raises PlacementError before any write when the spec does not fit the hardware.
197 The live fleet applies the change surgically (``reload_placement`` restarts
198 only the roles whose placement moved), so an untouched role's loaded model
199 stays resident; with no services built there is nothing running and the next
200 use plans fresh. On the live path the planner re-plans against its clean-box
201 plan snapshot (see ``planning.capture_plan_probe``): probing under a loaded
202 fleet would report our own residency as unavailable and poison the chat
203 context sizing, while charging stays against total capacity (bb-a8f).
204 """
205 resolved = resolve_placement_plan(spec)
206 if spec is None:
207 settings.delete_values(cfg.data_root, [_PLACEMENT_KEY])
208 cfg.placement = None
209 else:
210 spec_json = spec.to_json()
211 settings.update_values(cfg.data_root, {_PLACEMENT_KEY: spec_json})
212 cfg.placement = spec_json
213 services = peek_services()
214 if services is None:
215 clear_read_device_cache() # nothing running; let the next boot probe fresh
216 else:
217 services.provider.reload_placement(wait=True)
218 return _view(resolved, manual=spec is not None, spec_json=spec.to_json() if spec else None)
221def wait_chat_ready(
222 timeout_s: float = _CHAT_READY_TIMEOUT_S,
223 *,
224 on_progress: Callable[[WarmProgress], None] | None = None,
225 should_abort: Callable[[], bool] | None = None,
226) -> bool:
227 """Block while a chat warm is in flight; True once a prompt can be served.
229 ``reload_placement(wait=True)`` returns once the proxies are healthy while the
230 restarted model still warms off-thread, so a chat request sent right after an
231 apply hits the busy 429 path. Callers that gate user input on the reload call
232 this to hold until the model actually serves. Waits only while a warm is
233 actively in flight: with no fleet, no warm, or a failed/finished warm it
234 returns at once, so a change that never restarts chat cannot stall the caller.
235 The brief grace covers the reload kicking its warm on a separate thread.
237 ``on_progress`` receives each actively-reporting warm snapshot so the caller
238 can render the load. ``should_abort`` is polled every cycle; True ends the
239 wait at once, so a cancelled prompt never pins its worker thread.
240 """
241 services = peek_services()
242 if services is None:
243 return False
244 provider = services.provider
245 started = time.monotonic()
246 deadline = started + timeout_s
247 grace_deadline = started + _CHAT_READY_GRACE_S
248 while time.monotonic() < deadline:
249 if provider.role_ready(WorkerRole.CHAT):
250 return True
251 if should_abort is not None and should_abort():
252 return False
253 snapshot = provider.warm_progress()
254 # A requested warm counts as in flight before it stamps a phase: the fleet
255 # spawns and health-checks llama-swap first, which takes seconds.
256 if is_active_warm(snapshot):
257 if on_progress is not None and snapshot is not None:
258 on_progress(snapshot)
259 grace_deadline = time.monotonic() + _CHAT_READY_GRACE_S
260 elif provider.warm_pending():
261 grace_deadline = time.monotonic() + _CHAT_READY_GRACE_S
262 elif time.monotonic() > grace_deadline:
263 return False
264 time.sleep(_CHAT_READY_POLL_S)
265 return False
268def request_engine_warm() -> None:
269 """Kick the provider's warm-up when nothing is loaded or loading.
271 ``warm_up_pool`` is idempotent (a no-op while a warm is in flight or the
272 fleet is up), so a prompt sent after a failed boot warm drives a fresh
273 engine start instead of bouncing for the rest of the session.
274 """
275 services = peek_services()
276 if services is None:
277 return
278 services.provider.warm_up_pool()
281def chat_engine_ready() -> bool:
282 """Whether a chat prompt can be served right now.
284 Positive readiness, not absence-of-warm: before the services container is
285 built nothing is loading yet and nothing can answer, which a warm snapshot
286 cannot distinguish from a finished load.
287 """
288 services = peek_services()
289 if services is None:
290 return False
291 return services.provider.role_ready(WorkerRole.CHAT)
294def active_chat_warm_progress() -> WarmProgress | None:
295 """The chat warm snapshot while a cold load is genuinely in flight, else None.
297 A surface gates interactive input on this: ``None`` covers ready, no fleet, a
298 missing model, and a finished or failed warm, so nothing traps the input in a
299 locked state. Non-``None`` carries the phase and byte progress to render.
301 The in-process warm snapshot is checked before ``role_ready`` because the task
302 bar polls this on every tick: the snapshot is a free attribute read, while
303 ``role_ready`` is an HTTP probe of the engine. Ordering it this way keeps the
304 probe to the seconds a load is actually in flight instead of firing forever on
305 an idle TUI. The two orders return the same answer.
306 """
307 services = peek_services()
308 if services is None:
309 return None
310 provider = services.provider
311 snapshot = provider.warm_progress()
312 if not is_active_warm(snapshot):
313 return None
314 return None if provider.role_ready(WorkerRole.CHAT) else snapshot
317def chat_warm_error() -> str | None:
318 """The failed chat warm's error text, or None when no failure is on record."""
319 services = peek_services()
320 if services is None:
321 return None
322 snapshot = services.provider.warm_progress()
323 if snapshot is not None and snapshot.phase is WarmPhase.ERROR:
324 return snapshot.error or ""
325 return None