Coverage for src/lilbee/providers/fleet/devices.py: 100%

209 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""Enumerate and pin GPUs using the llama-server binary's own device view. 

2 

3The hazard this avoids: a device index from one API (Vulkan) is meaningless to 

4another (CUDA); the same ordinal can be a different physical card. So both 

5enumeration and pinning go through the binary's native backend index space, 

6obtained from ``llama-server --list-devices``. The Vulkan VRAM probe is only a 

7fallback when the binary can't enumerate. See docs/architecture.md. 

8""" 

9 

10from __future__ import annotations 

11 

12import logging 

13import os 

14import re 

15import subprocess 

16from dataclasses import dataclass 

17from pathlib import Path 

18 

19from lilbee.providers.base import ProviderError, ProviderErrorKind 

20from lilbee.providers.fleet.gpu_select import USABLE_VULKAN_TYPES, VkDeviceType 

21from lilbee.providers.fleet.proc import run_bounded 

22from lilbee.providers.roles import EngineBackend 

23 

24log = logging.getLogger(__name__) 

25 

26_PROVIDER = "llama-server" 

27_LIST_DEVICES_TIMEOUT_S = 60.0 

28# How long to wait for a killed probe to be reaped before abandoning it: a child 

29# wedged in uninterruptible GPU-driver I/O ignores even SIGKILL. 

30_PROBE_KILL_WAIT_S = 5.0 

31# How much of the probe's own output to quote in a diagnostic. Enough to carry 

32# the driver's error line, short enough to stay a readable message. 

33_PROBE_TAIL_CHARS = 400 

34_TOPO_TIMEOUT_S = 15.0 

35_GPU_LABEL_RE = re.compile(r"^GPU(\d+)$") 

36# llama-server prints this before the device loop, so a run that lists no GPUs 

37# still prints it. Its absence means the binary never got as far as enumerating. 

38_DEVICE_LIST_HEADER = "Available devices:" 

39# nvidia-smi emits SGR escapes (e.g. an underlined header) even when stdout is 

40# not a tty; strip them or the header's GPU labels never match. 

41_ANSI_SGR_RE = re.compile(r"\x1b\[[0-9;]*m") 

42# A topo-matrix header is 2+ leading GPU labels; a data row has exactly one. And 

43# a link needs at least two GPUs to exist between. 

44_TOPO_MIN_GPUS = 2 

45MIB = 1024 * 1024 

46# Per-backend visible-devices env vars (the probe inherits them; the children 

47# re-emit them, composed through any parent restriction). 

48_CUDA_VISIBLE_VAR = "CUDA_VISIBLE_DEVICES" 

49_CUDA_ORDER_VAR = "CUDA_DEVICE_ORDER" 

50_PCI_BUS_ID_ORDER = "PCI_BUS_ID" 

51_ROCR_VISIBLE_VAR = "ROCR_VISIBLE_DEVICES" 

52_HIP_VISIBLE_VAR = "HIP_VISIBLE_DEVICES" 

53# ROCm's third numeric visibility variable, filtering exactly as the other two do. 

54_GPU_DEVICE_ORDINAL_VAR = "GPU_DEVICE_ORDINAL" 

55_VK_VISIBLE_VAR = "GGML_VK_VISIBLE_DEVICES" 

56# " CUDA0: NVIDIA GeForce RTX 3090 (24268 MiB, 23500 MiB free)" 

57_DEVICE_RE = re.compile( 

58 r"^\s*([A-Za-z]+)(\d+):\s*(.+?)\s*\((\d+)\s*MiB(?:,\s*(\d+)\s*MiB\s*free)?\)\s*$" 

59) 

60# The engine's own name for the backend. Vendor-agnostic, so several rules key 

61# on it: a Vulkan device's type has to be asked of the loader, and its util 

62# source is chosen by the vendor in its device name rather than by the backend. 

63VULKAN_BACKEND = "Vulkan" 

64# One row per backend string the engine prints: its pin priority, and the name a 

65# client reports for it. Several engine names mean one backend (HIP is ROCm, MTL 

66# is Metal), so a surface that printed the raw name told two hosts apart that are 

67# the same machine class. 

68# 

69# The two tables below are derived from this one rather than transcribed beside 

70# it. Adding a backend is what makes a new GPU class usable at all, and with two 

71# hand-written tables that edit ranked the new backend while leaving it nameless, 

72# so a working GPU host reported "unknown". 

73_BACKENDS: dict[str, tuple[int, EngineBackend]] = { 

74 "CUDA": (3, EngineBackend.CUDA), 

75 "ROCm": (3, EngineBackend.ROCM), 

76 "HIP": (3, EngineBackend.ROCM), 

77 "MTL": (3, EngineBackend.METAL), 

78 "Metal": (3, EngineBackend.METAL), 

79 "SYCL": (2, EngineBackend.SYCL), 

80 VULKAN_BACKEND: (1, EngineBackend.VULKAN), 

81} 

82# Pin priority when a build reports more than one GPU backend: a real GPU backend 

83# always wins over Vulkan, which wins over CPU. 

84_BACKEND_RANK = {name: rank for name, (rank, _) in _BACKENDS.items()} 

85# The engine's own backend names mapped to the name a client reports. 

86_REPORTED_BACKEND = {name: reported for name, (_, reported) in _BACKENDS.items()} 

87# Backends whose memory is always the host's: Apple Silicon reports a working-set 

88# slice of system RAM, never a dedicated pool. 

89_UNIFIED_BACKENDS = frozenset({"MTL", "Metal"}) 

90# Below this, a reported total is a BIOS carveout rather than a card's own pool. 

91# An APU hands out a fixed slice of system RAM as "VRAM", often a few hundred 

92# MiB, and planned as a dedicated device that size it refuses every role while 

93# the machine has the whole system's memory to share. No real discrete GPU worth 

94# serving from ships with less. 

95_DEDICATED_VRAM_FLOOR = 2 * 1024 * MIB 

96 

97 

98@dataclass(frozen=True) 

99class FleetDevice: 

100 """One GPU as the binary's backend enumerates it (native index space).""" 

101 

102 backend: str 

103 index: int 

104 name: str 

105 total_bytes: int 

106 free_bytes: int 

107 # Whether this device's memory is the host's memory. An integrated GPU or an 

108 # Apple Silicon Mac has no dedicated VRAM, so its reported total is a slice 

109 # of the same RAM the OS and every other process is using, and placement 

110 # must stay inside the system budget rather than treating it as headroom. 

111 unified: bool = False 

112 # Whether this device came from the host's Vulkan loader rather than from the 

113 # engine's own listing. Its index is then a raw loader ordinal, which is a 

114 # different space from the one the engine names its devices in, so it can be 

115 # sized against but never pinned by. 

116 from_loader: bool = False 

117 

118 

119@dataclass(frozen=True) 

120class DeviceProbe: 

121 """The device probe's parsed devices plus its raw output for diagnostics.""" 

122 

123 devices: list[FleetDevice] 

124 output: str 

125 # Whether the engine answered --list-devices at all: exited cleanly and 

126 # printed the header it always prints. False means the binary does not speak 

127 # this protocol (a build predating the flag prints usage text and exits 

128 # non-zero), so its silence about devices is not a statement that there are 

129 # none. Defaults False so a probe that never ran is never mistaken for one 

130 # that ran and found nothing. 

131 spoke_protocol: bool = False 

132 # Whether the engine listed GPU devices and every one was rejected. Distinct 

133 # from a host that simply has none: the engine will still pick one of those 

134 # devices at launch unless it is told not to. 

135 refused_all: bool = False 

136 # Which backend the engine selected. Defaults to UNKNOWN so a probe that never 

137 # ran reports no claim rather than CPU. 

138 backend: EngineBackend = EngineBackend.UNKNOWN 

139 

140 

141def _parse_topo_matrix(topo_text: str) -> tuple[set[int], set[frozenset[int]]]: 

142 """GPU row indices and NVLink-joined pairs from ``nvidia-smi topo -m`` output. 

143 

144 The matrix header row labels the GPU columns; each ``GPU<r>`` row lists the 

145 link type to each column (``NV#`` is NVLink; ``PIX``/``PHB``/``SYS`` are PCIe). 

146 """ 

147 header_cols: list[int] = [] 

148 gpu_rows: set[int] = set() 

149 pairs: set[frozenset[int]] = set() 

150 for line in _ANSI_SGR_RE.sub("", topo_text).splitlines(): 

151 tokens = line.split() 

152 # Leading run of GPU-label tokens: the header is all labels (>=2), a data 

153 # row is one label ("GPU3") followed by link-type cells. split() strips the 

154 # header's leading whitespace, so this run length is what tells them apart. 

155 leading_labels: list[int] = [] 

156 for token in tokens: 

157 match = _GPU_LABEL_RE.match(token) 

158 if match is None: 

159 break 

160 leading_labels.append(int(match.group(1))) 

161 if len(leading_labels) >= _TOPO_MIN_GPUS: 

162 header_cols = leading_labels 

163 elif len(leading_labels) == 1: 

164 row_idx = leading_labels[0] 

165 gpu_rows.add(row_idx) 

166 for col_idx, cell in zip(header_cols, tokens[1:], strict=False): 

167 if row_idx != col_idx and cell.startswith("NV"): 

168 pairs.add(frozenset({row_idx, col_idx})) 

169 return gpu_rows, pairs 

170 

171 

172def host_lacks_nvlink() -> bool: 

173 """Whether this host's GPUs are joined only by PCIe (no NVLink anywhere). 

174 

175 Tensor-splitting a large model across PCIe-only cards is all-reduce bound and 

176 much slower than over NVLink. Deliberately a host-level claim: the fleet's 

177 device indices live in the serving binary's backend index space, which does 

178 not map onto ``nvidia-smi``'s physical numbering under a visible-devices 

179 restriction (the very hazard this module exists to avoid), so per-pair 

180 verdicts against plan indices would be unreliable. Returns False (no claim) 

181 when the probe fails or reports fewer than two GPUs, so a non-NVIDIA or 

182 single-card host stays silent rather than warning wrongly. 

183 """ 

184 try: 

185 stdout, _ = run_bounded( 

186 ["nvidia-smi", "topo", "-m"], 

187 timeout_s=_TOPO_TIMEOUT_S, 

188 kill_wait_s=_PROBE_KILL_WAIT_S, 

189 label="nvidia-smi topo", 

190 ) 

191 except (OSError, subprocess.SubprocessError): 

192 return False 

193 gpu_rows, pairs = _parse_topo_matrix(stdout) 

194 return len(gpu_rows) >= _TOPO_MIN_GPUS and not pairs 

195 

196 

197def _probe_env() -> dict[str, str]: 

198 """Env for the probe: stable PCI ordering so CUDA indices match what we pin. 

199 

200 A preset ``CUDA_DEVICE_ORDER`` is respected; ``visible_env`` re-emits the same 

201 order var, so the probe and the spawned servers see one device ordering. 

202 """ 

203 env = dict(os.environ) 

204 env.setdefault(_CUDA_ORDER_VAR, _PCI_BUS_ID_ORDER) 

205 return env 

206 

207 

208def probe_devices(binary: Path, *, timeout_s: float = _LIST_DEVICES_TIMEOUT_S) -> DeviceProbe: 

209 """Parse ``<binary> --list-devices``; empty devices when unavailable/unparseable. 

210 

211 Filtered to a single GPU backend (the highest-ranked one present) so device 

212 indices are unambiguous when a build exposes several backends. A probe that 

213 does not respond within *timeout_s* raises a ``ProviderError`` naming the 

214 stuck probe: that is a wedged GPU driver, not a GPU-less host, and treating 

215 it as "no devices" would silently plan a CPU fleet on a GPU box. 

216 """ 

217 try: 

218 output, returncode = _run_list_devices(binary, timeout_s) 

219 except (OSError, subprocess.SubprocessError) as exc: 

220 # Silently returning an empty probe here made an unrunnable binary look 

221 # exactly like a host with no GPU, and the fleet planned for CPU with 

222 # nothing said. The reason is the whole diagnosis: a wrong architecture, 

223 # a missing loader, a permission denial. 

224 log.warning( 

225 "Could not run the GPU device probe (%s --list-devices): %s. Continuing " 

226 "as though this host has no GPU; check that the engine binary is " 

227 "executable and built for this machine.", 

228 binary.name, 

229 exc, 

230 ) 

231 return DeviceProbe([], "") 

232 parsed = _parse_devices(output) 

233 selected = _select_backend(parsed) 

234 offered = [d for d in parsed if d.backend in _BACKEND_RANK] 

235 answered = _DEVICE_LIST_HEADER in output 

236 spoke = returncode == 0 and answered 

237 if not spoke and answered: 

238 # It knew the flag and started answering, then died. Blaming the flag 

239 # here sent the reader looking for the wrong engine build, when what 

240 # they have is a crash partway through enumeration. 

241 log.warning( 

242 "%s --list-devices printed its device header then crashed (exit %d), so the " 

243 "device list may be incomplete. This is usually a GPU driver or ICD fault " 

244 "during enumeration. The probe reported: %s", 

245 binary.name, 

246 returncode, 

247 _probe_tail(output), 

248 ) 

249 elif not spoke: 

250 log.warning( 

251 "%s --list-devices exited %d without printing its device header, so it " 

252 "does not appear to support the flag. Falling back to the host's Vulkan " 

253 "loader to find GPUs; set %s if this is not the engine you meant to use.", 

254 binary.name, 

255 returncode, 

256 "LILBEE_ENGINE_DIR", 

257 ) 

258 return DeviceProbe( 

259 selected, 

260 output, 

261 spoke_protocol=spoke, 

262 refused_all=bool(offered) and not selected, 

263 backend=_selected_backend(selected, spoke=spoke), 

264 ) 

265 

266 

267def _selected_backend(selected: list[FleetDevice], *, spoke: bool) -> EngineBackend: 

268 """Which backend the engine selected, or UNKNOWN when it did not answer. 

269 

270 An empty device list means CPU only when the engine answered the question. A 

271 binary that never spoke the protocol said nothing about its backend, and 

272 calling that CPU labels a GPU host as a CPU one in every diagnostic. 

273 

274 A selected device's backend is always a row of ``_BACKENDS``, because that is 

275 what the selector filters on, so this indexes rather than defaulting. A 

276 default here would turn a table the selector and the reporter disagree about 

277 into a working GPU host that reports "unknown", which is this function's own 

278 defect one backend along. 

279 """ 

280 if selected: 

281 return _REPORTED_BACKEND[selected[0].backend] 

282 return EngineBackend.CPU if spoke else EngineBackend.UNKNOWN 

283 

284 

285def _run_list_devices(binary: Path, timeout_s: float) -> tuple[str, int]: 

286 """Run the probe with a bounded reap; raise on timeout. 

287 

288 A probe wedged in uninterruptible GPU-driver I/O would otherwise hang the 

289 caller forever, since ``subprocess.run``'s timeout waits unbounded for the 

290 reap; ``run_bounded`` abandons an unkillable child after a short wait. 

291 

292 The probe holds a device context and writes no state file, so nothing can reap 

293 it later by record. It is the one caller that opts into the lifetime binding, 

294 where the kernel offers one, and it is killed on the way out of every abort, 

295 not just the timeout. 

296 """ 

297 try: 

298 return run_bounded( 

299 [str(binary), "--list-devices"], 

300 timeout_s=timeout_s, 

301 kill_wait_s=_PROBE_KILL_WAIT_S, 

302 env=_probe_env(), 

303 merge_stderr=True, 

304 label=f"{binary.name} --list-devices", 

305 bind_lifetime=True, 

306 ) 

307 except subprocess.TimeoutExpired as exc: 

308 # Whatever the probe managed to print before it wedged says more than any 

309 # fixed advice can, and the fixed advice named one vendor's tool at a host 

310 # that may have neither that vendor nor that tool. 

311 raise ProviderError( 

312 f"The GPU device probe ({binary.name} --list-devices) did not respond " 

313 f"within {timeout_s:.0f}s, so the engine cannot start. The GPU driver is " 

314 "most likely wedged; check that your vendor's tool responds (nvidia-smi, " 

315 "rocm-smi, xpu-smi) and reboot the host if it hangs.\n" 

316 f"The probe reported: {_probe_tail(_decoded_output(exc.output))}", 

317 provider=_PROVIDER, 

318 kind=ProviderErrorKind.SERVER, 

319 ) from None 

320 

321 

322def _decoded_output(output: object) -> str: 

323 """Partial child output from a timeout, which arrives as bytes even under text mode.""" 

324 if isinstance(output, bytes): 

325 return output.decode(errors="replace") 

326 return output if isinstance(output, str) else "" 

327 

328 

329def _probe_tail(output: str) -> str: 

330 """The tail of what the probe printed, for a message that has to stay readable.""" 

331 text = output.strip() 

332 return text[-_PROBE_TAIL_CHARS:] if text else "(nothing)" 

333 

334 

335def _parse_devices(text: str) -> list[FleetDevice]: 

336 devices: list[FleetDevice] = [] 

337 # Sampled at most once per parse, and only when a line actually needs it: 

338 # free memory is live, so it is read fresh here rather than cached, and the 

339 # loader must not be opened once per device line to answer the same question. 

340 loader_free: dict[str, int] | None = None 

341 for line in text.splitlines(): 

342 match = _DEVICE_RE.match(line) 

343 if match is None: 

344 continue 

345 backend, index, name, total_mib, free_mib = match.groups() 

346 total = int(total_mib) * MIB 

347 if total == 0: 

348 # No memory is not a small GPU, it is one that cannot hold a model: 

349 # a driver listing an adapter before its memory is queryable. Kept, it 

350 # is the smallest card in the fleet and collapses every budget sized 

351 # against the smallest, while the non-empty list switches off the 

352 # shared-memory budget a host with no usable GPU depends on. 

353 log.warning( 

354 "Ignoring GPU %s%s (%s): it reports no memory, so nothing can be " 

355 "placed on it. Check the GPU driver if this device should be usable.", 

356 backend, 

357 index, 

358 name.strip(), 

359 ) 

360 continue 

361 if free_mib: 

362 free = int(free_mib) * MIB 

363 else: 

364 if loader_free is None: 

365 loader_free = _loader_free_bytes(backend) 

366 free = loader_free.get(name.strip(), total) 

367 devices.append( 

368 FleetDevice( 

369 backend, 

370 int(index), 

371 name.strip(), 

372 total, 

373 free, 

374 unified=_is_unified(backend, name.strip()) or total < _DEDICATED_VRAM_FLOOR, 

375 ) 

376 ) 

377 return devices 

378 

379 

380# Mesa and friends expose CPU rasterizers through the Vulkan loader, and 

381# llama.cpp's Vulkan backend enumerates them exactly like a GPU: same 

382# "VulkanN: <name> (<total> MiB, <free> MiB free)" shape, with system RAM 

383# reported as VRAM. Planning against one is worse than having no GPU at all, 

384# because the "VRAM" looks enormous: a host with a real iGPU beside lavapipe 

385# can be planned as a two-GPU machine and tensor-split across a real adapter 

386# and a software renderer, which runs orders of magnitude slower than either 

387# CPU inference or the iGPU alone. 

388_SOFTWARE_RENDERER_MARKERS = ("llvmpipe", "lavapipe", "softpipe", "swiftshader") 

389 

390 

391def _is_software_renderer(device: FleetDevice) -> bool: 

392 """Whether *device* is a CPU rasterizer masquerading as a GPU. 

393 

394 A name test, so it only recognizes the rasterizers it already knows, and a 

395 renamed or newly written one walks past it. It stays as the answer for hosts 

396 where the Vulkan loader can't be opened from this process and the device 

397 type is therefore unavailable; where the type is available, 

398 ``_is_unusable_vulkan`` decides and this never gets the chance to be wrong. 

399 """ 

400 name = device.name.casefold() 

401 return any(marker in name for marker in _SOFTWARE_RENDERER_MARKERS) 

402 

403 

404def _loader_free_bytes(backend: str) -> dict[str, int]: 

405 """Live free memory per device name, for a listing that printed no free figure. 

406 

407 ggml omits the figure when the driver has no ``VK_EXT_memory_budget``, and 

408 treating the omission as "all of it" is how a desktop holding gigabytes of 

409 compositor and browser VRAM was planned as an empty card. The loader exposes 

410 that extension to this process even when the engine build cannot use it, so 

411 it is asked directly; a name it cannot speak for keeps the heap size. 

412 

413 Empty for any other backend: the Vulkan loader knows nothing about the 

414 devices a CUDA or ROCm listing names. 

415 """ 

416 if backend != VULKAN_BACKEND: 

417 return {} 

418 from lilbee.providers.fleet.gpu_select import vulkan_free_bytes_by_name 

419 

420 return vulkan_free_bytes_by_name() 

421 

422 

423def _vulkan_device_type(name: str) -> VkDeviceType | None: 

424 """The loader's type for the Vulkan adapter the engine printed as *name*. 

425 

426 ``None`` when the loader can't be reached or reports no adapter by that 

427 name, which reads as "no opinion": the device is kept and assumed dedicated, 

428 preserving the behaviour of hosts that never had a type to consult. 

429 """ 

430 from lilbee.providers.fleet.gpu_select import vulkan_device_types_by_name 

431 

432 return vulkan_device_types_by_name().get(name) 

433 

434 

435def _is_unusable_vulkan(device: FleetDevice) -> bool: 

436 """Whether *device* is a Vulkan adapter ggml would not choose to run on. 

437 

438 ggml's Vulkan backend builds its device pool from discrete and integrated 

439 adapters only, and falls back to the first non-CPU adapter when it finds 

440 neither. In a VM that fallback is a paravirtual adapter (VMware SVGA, 

441 VirtIO-GPU Venus, QXL), which reports guest RAM as VRAM and is typically 

442 compute-incomplete or fails at allocation. Planning a fleet onto one costs 

443 more than planning no GPU at all, since a non-empty device list also turns 

444 off the shared-RAM budget. 

445 

446 Only a positive claim counts. VIRTUAL_GPU and CPU are the loader naming what 

447 the adapter is; OTHER is it declining to, and refusing on a shrug took the 

448 GPU away from real hardware whose driver simply does not classify itself. 

449 """ 

450 if device.backend != VULKAN_BACKEND: 

451 return False 

452 device_type = _vulkan_device_type(device.name) 

453 if device_type is None or device_type in USABLE_VULKAN_TYPES: 

454 return False 

455 # OTHER is the loader shrugging, not an accusation. The spec's own wording is 

456 # "does not match any other available types", which a driver reaches for when 

457 # it cannot classify itself, and some real adapters do. Refusing on it took a 

458 # working GPU away from a machine the engine had already listed one for. 

459 # VIRTUAL_GPU and CPU are positive claims and keep their veto. 

460 return device_type is not VkDeviceType.OTHER 

461 

462 

463def _is_unified(backend: str, name: str) -> bool: 

464 """Whether the device *backend* printed as *name* shares its memory with the host. 

465 

466 Metal is unified by construction on Apple Silicon: the figure it reports is 

467 ``recommendedMaxWorkingSetSize``, a slice of system RAM rather than a 

468 separate pool. For Vulkan the loader knows the device type, so the type is 

469 asked for rather than guessed; a size heuristic cannot work here, since a 

470 24 GB discrete card in a 32 GB host and an Apple GPU reporting two thirds of 

471 RAM are indistinguishable by proportion. 

472 

473 CUDA, ROCm and SYCL print no type at all, and an AMD APU or a Jetson looks 

474 exactly like a discrete card there while reporting system RAM as its memory. 

475 Those fall back to a question about the machine rather than the device: a 

476 host whose Vulkan loader sees adapters but no discrete one has no discrete 

477 GPU for another backend to be enumerating. 

478 """ 

479 if backend in _UNIFIED_BACKENDS: 

480 return True 

481 if backend == VULKAN_BACKEND: 

482 return _vulkan_device_type(name) is VkDeviceType.INTEGRATED_GPU 

483 from lilbee.providers.fleet.gpu_select import host_has_no_discrete_gpu 

484 

485 return host_has_no_discrete_gpu() 

486 

487 

488def _select_backend(devices: list[FleetDevice]) -> list[FleetDevice]: 

489 """Keep one GPU backend's devices: highest rank, then most memory. 

490 

491 Returns a single backend so pinning is unambiguous: ``visible_env`` keys off 

492 one backend, and mixing index spaces is the very hazard this module avoids. 

493 

494 CUDA, ROCm, HIP and Metal all rank alike, and a build that loads several 

495 backends (``ggml_backend_load_all`` does) makes the tie real. Breaking it on 

496 the backend's name meant a host with a 4090 beside an RX 6600 planned onto 

497 the AMD card because "ROCm" sorts after "CUDA", and the NVIDIA card idled 

498 with nothing said. Total memory decides instead; the name is only the last 

499 resort that keeps the choice deterministic. 

500 """ 

501 ranked = [ 

502 d 

503 for d in devices 

504 if d.backend in _BACKEND_RANK 

505 and not _is_software_renderer(d) 

506 and not _is_unusable_vulkan(d) 

507 ] 

508 if not ranked: 

509 return [] 

510 by_backend: dict[str, list[FleetDevice]] = {} 

511 for device in ranked: 

512 by_backend.setdefault(device.backend, []).append(device) 

513 backend, chosen = max(by_backend.items(), key=_backend_preference) 

514 for other, group in by_backend.items(): 

515 if other != backend: 

516 log.info( 

517 "Engine reports %d %s device(s) beside %d %s device(s); planning onto %s, " 

518 "which has more memory. Backends cannot be mixed: their device indexes " 

519 "name different cards.", 

520 len(group), 

521 other, 

522 len(chosen), 

523 backend, 

524 backend, 

525 ) 

526 return chosen 

527 

528 

529def _backend_preference(item: tuple[str, list[FleetDevice]]) -> tuple[int, int, int, str]: 

530 """Sort key for choosing one backend's devices: rank, dedicated bytes, size. 

531 

532 Dedicated bytes come before raw size because the discrete backends all tie at 

533 the same rank, and a shared-heap carveout reports a total that is host RAM 

534 the host budget already counts. Left on raw size, an APU advertising a large 

535 carveout beat a discrete card, which was then discarded and left idle while 

536 the plan double-promised memory it did not have. 

537 """ 

538 backend, group = item 

539 dedicated = sum(d.total_bytes for d in group if not d.unified) 

540 return _BACKEND_RANK[backend], dedicated, sum(d.total_bytes for d in group), backend 

541 

542 

543def _compose_visible(indices: list[int], parent_value: str | None) -> str: 

544 """Visible-devices value naming the same physical devices the probe saw. 

545 

546 When the parent env already restricts the var, the probe's indices are 

547 relative to that comma-separated list (integer or UUID entries), so each 

548 index maps through it; the child's value then names the same physical 

549 devices instead of being re-interpreted as absolute. 

550 """ 

551 if parent_value is None: 

552 return ",".join(str(i) for i in indices) 

553 entries = [entry.strip() for entry in parent_value.split(",") if entry.strip()] 

554 out: list[str] = [] 

555 for i in indices: 

556 if i >= len(entries): 

557 # The probe enumerates devices under the parent restriction, so every 

558 # index must map into it. An out-of-range index is an invariant 

559 # violation; emitting a bare ``str(i)`` would pin an absolute integer 

560 # into a possibly UUID-namespaced list, silently selecting the wrong 

561 # GPU. Fail loudly instead. 

562 raise ValueError( 

563 f"device index {i} is outside the parent visible-devices list " 

564 f"{parent_value!r}; cannot compose a child pin without selecting the wrong GPU" 

565 ) 

566 out.append(entries[i]) 

567 return ",".join(out) 

568 

569 

570def visible_env(devices: tuple[FleetDevice, ...]) -> dict[str, str]: 

571 """Env that pins a child to *devices* via the right var for their backend. 

572 

573 Indices are the backend-native ones from ``probe_devices``, composed through 

574 any parent visible-devices restriction so the child names the same physical 

575 devices the probe enumerated; no cross-API index translation occurs. 

576 """ 

577 if not devices: 

578 return {} 

579 backend = devices[0].backend 

580 indices = [d.index for d in devices] 

581 if backend == "CUDA": 

582 return { 

583 _CUDA_VISIBLE_VAR: _compose_visible(indices, os.environ.get(_CUDA_VISIBLE_VAR)), 

584 _CUDA_ORDER_VAR: os.environ.get(_CUDA_ORDER_VAR, _PCI_BUS_ID_ORDER), 

585 } 

586 if backend in ("ROCm", "HIP"): 

587 return _amd_visible_env(indices) 

588 if backend == VULKAN_BACKEND: 

589 # Deliberately not GGML_VK_VISIBLE_DEVICES. That variable indexes the raw 

590 # loader enumeration, while these indices come from the engine's own 

591 # filtered list, so the two disagree wherever ggml drops or merges a 

592 # device -- two ICDs for one card being the clear case. Setting it also 

593 # disables ggml's type filter, support check and dedup. Vulkan is pinned 

594 # with --device instead, in the same space the names were parsed from. 

595 return {} 

596 if backend == "SYCL": 

597 # Deliberately no ONEAPI_DEVICE_SELECTOR. It is a selector grammar over a 

598 # backend runtime, not the index space --list-devices numbers, so a 

599 # composed level_zero ordinal can name a different physical card than the 

600 # one the probe enumerated. SYCL pins by --device instead, in the space 

601 # the indices were read from. An inherited parent selector still applies: 

602 # the engine enumerated behind it, so its names are already relative to it. 

603 return {} 

604 return {} 

605 

606 

607def amd_visible_var() -> str: 

608 """The one AMD visibility var an index list may be written to. 

609 

610 ``ROCR_VISIBLE_DEVICES`` and ``HIP_VISIBLE_DEVICES`` are applied sequentially: 

611 ROCr filters first, then HIP re-indexes within the survivors. Writing the same 

612 indices to both double-filters and selects the wrong cards, or none at all. 

613 ``GPU_DEVICE_ORDINAL`` is the third and filters the same way, so writing HIP 

614 over an ordinal mask both overrides it and re-exposes cards it had hidden. 

615 

616 So exactly one is ever written: whichever the environment already restricts, 

617 in the runtime's precedence (HIP, then the ordinal, then ROCr), or HIP when 

618 nothing restricts. An empty value means "no devices" rather than "this is the 

619 variable in use", so it does not claim precedence. Every caller writing an AMD 

620 pin asks here; two callers each picking their own would put the pair back. 

621 """ 

622 for name in (_HIP_VISIBLE_VAR, _GPU_DEVICE_ORDINAL_VAR, _ROCR_VISIBLE_VAR): 

623 if os.environ.get(name, "").strip(): 

624 return name 

625 return _HIP_VISIBLE_VAR 

626 

627 

628def _amd_visible_env(indices: list[int]) -> dict[str, str]: 

629 """Pin an AMD ROCm/HIP child to the probe's *indices* with one visibility var. 

630 

631 The probe enumerated a single index space already filtered by whichever var 

632 the parent set, so the chosen var is composed against that parent value and 

633 the other is left inherited untouched. The child inherits the parent env, so 

634 an unset override keeps any inherited sibling var in force. 

635 """ 

636 var = amd_visible_var() 

637 return {var: _compose_visible(indices, os.environ.get(var))}