Coverage for src/lilbee/providers/fleet/devices.py: 100%
209 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
1"""Enumerate and pin GPUs using the llama-server binary's own device view.
3The hazard this avoids: a device index from one API (Vulkan) is meaningless to
4another (CUDA); the same ordinal can be a different physical card. So both
5enumeration and pinning go through the binary's native backend index space,
6obtained from ``llama-server --list-devices``. The Vulkan VRAM probe is only a
7fallback when the binary can't enumerate. See docs/architecture.md.
8"""
10from __future__ import annotations
12import logging
13import os
14import re
15import subprocess
16from dataclasses import dataclass
17from pathlib import Path
19from lilbee.providers.base import ProviderError, ProviderErrorKind
20from lilbee.providers.fleet.gpu_select import USABLE_VULKAN_TYPES, VkDeviceType
21from lilbee.providers.fleet.proc import run_bounded
22from lilbee.providers.roles import EngineBackend
24log = logging.getLogger(__name__)
26_PROVIDER = "llama-server"
27_LIST_DEVICES_TIMEOUT_S = 60.0
28# How long to wait for a killed probe to be reaped before abandoning it: a child
29# wedged in uninterruptible GPU-driver I/O ignores even SIGKILL.
30_PROBE_KILL_WAIT_S = 5.0
31# How much of the probe's own output to quote in a diagnostic. Enough to carry
32# the driver's error line, short enough to stay a readable message.
33_PROBE_TAIL_CHARS = 400
34_TOPO_TIMEOUT_S = 15.0
35_GPU_LABEL_RE = re.compile(r"^GPU(\d+)$")
36# llama-server prints this before the device loop, so a run that lists no GPUs
37# still prints it. Its absence means the binary never got as far as enumerating.
38_DEVICE_LIST_HEADER = "Available devices:"
39# nvidia-smi emits SGR escapes (e.g. an underlined header) even when stdout is
40# not a tty; strip them or the header's GPU labels never match.
41_ANSI_SGR_RE = re.compile(r"\x1b\[[0-9;]*m")
42# A topo-matrix header is 2+ leading GPU labels; a data row has exactly one. And
43# a link needs at least two GPUs to exist between.
44_TOPO_MIN_GPUS = 2
45MIB = 1024 * 1024
46# Per-backend visible-devices env vars (the probe inherits them; the children
47# re-emit them, composed through any parent restriction).
48_CUDA_VISIBLE_VAR = "CUDA_VISIBLE_DEVICES"
49_CUDA_ORDER_VAR = "CUDA_DEVICE_ORDER"
50_PCI_BUS_ID_ORDER = "PCI_BUS_ID"
51_ROCR_VISIBLE_VAR = "ROCR_VISIBLE_DEVICES"
52_HIP_VISIBLE_VAR = "HIP_VISIBLE_DEVICES"
53# ROCm's third numeric visibility variable, filtering exactly as the other two do.
54_GPU_DEVICE_ORDINAL_VAR = "GPU_DEVICE_ORDINAL"
55_VK_VISIBLE_VAR = "GGML_VK_VISIBLE_DEVICES"
56# " CUDA0: NVIDIA GeForce RTX 3090 (24268 MiB, 23500 MiB free)"
57_DEVICE_RE = re.compile(
58 r"^\s*([A-Za-z]+)(\d+):\s*(.+?)\s*\((\d+)\s*MiB(?:,\s*(\d+)\s*MiB\s*free)?\)\s*$"
59)
60# The engine's own name for the backend. Vendor-agnostic, so several rules key
61# on it: a Vulkan device's type has to be asked of the loader, and its util
62# source is chosen by the vendor in its device name rather than by the backend.
63VULKAN_BACKEND = "Vulkan"
64# One row per backend string the engine prints: its pin priority, and the name a
65# client reports for it. Several engine names mean one backend (HIP is ROCm, MTL
66# is Metal), so a surface that printed the raw name told two hosts apart that are
67# the same machine class.
68#
69# The two tables below are derived from this one rather than transcribed beside
70# it. Adding a backend is what makes a new GPU class usable at all, and with two
71# hand-written tables that edit ranked the new backend while leaving it nameless,
72# so a working GPU host reported "unknown".
73_BACKENDS: dict[str, tuple[int, EngineBackend]] = {
74 "CUDA": (3, EngineBackend.CUDA),
75 "ROCm": (3, EngineBackend.ROCM),
76 "HIP": (3, EngineBackend.ROCM),
77 "MTL": (3, EngineBackend.METAL),
78 "Metal": (3, EngineBackend.METAL),
79 "SYCL": (2, EngineBackend.SYCL),
80 VULKAN_BACKEND: (1, EngineBackend.VULKAN),
81}
82# Pin priority when a build reports more than one GPU backend: a real GPU backend
83# always wins over Vulkan, which wins over CPU.
84_BACKEND_RANK = {name: rank for name, (rank, _) in _BACKENDS.items()}
85# The engine's own backend names mapped to the name a client reports.
86_REPORTED_BACKEND = {name: reported for name, (_, reported) in _BACKENDS.items()}
87# Backends whose memory is always the host's: Apple Silicon reports a working-set
88# slice of system RAM, never a dedicated pool.
89_UNIFIED_BACKENDS = frozenset({"MTL", "Metal"})
90# Below this, a reported total is a BIOS carveout rather than a card's own pool.
91# An APU hands out a fixed slice of system RAM as "VRAM", often a few hundred
92# MiB, and planned as a dedicated device that size it refuses every role while
93# the machine has the whole system's memory to share. No real discrete GPU worth
94# serving from ships with less.
95_DEDICATED_VRAM_FLOOR = 2 * 1024 * MIB
98@dataclass(frozen=True)
99class FleetDevice:
100 """One GPU as the binary's backend enumerates it (native index space)."""
102 backend: str
103 index: int
104 name: str
105 total_bytes: int
106 free_bytes: int
107 # Whether this device's memory is the host's memory. An integrated GPU or an
108 # Apple Silicon Mac has no dedicated VRAM, so its reported total is a slice
109 # of the same RAM the OS and every other process is using, and placement
110 # must stay inside the system budget rather than treating it as headroom.
111 unified: bool = False
112 # Whether this device came from the host's Vulkan loader rather than from the
113 # engine's own listing. Its index is then a raw loader ordinal, which is a
114 # different space from the one the engine names its devices in, so it can be
115 # sized against but never pinned by.
116 from_loader: bool = False
119@dataclass(frozen=True)
120class DeviceProbe:
121 """The device probe's parsed devices plus its raw output for diagnostics."""
123 devices: list[FleetDevice]
124 output: str
125 # Whether the engine answered --list-devices at all: exited cleanly and
126 # printed the header it always prints. False means the binary does not speak
127 # this protocol (a build predating the flag prints usage text and exits
128 # non-zero), so its silence about devices is not a statement that there are
129 # none. Defaults False so a probe that never ran is never mistaken for one
130 # that ran and found nothing.
131 spoke_protocol: bool = False
132 # Whether the engine listed GPU devices and every one was rejected. Distinct
133 # from a host that simply has none: the engine will still pick one of those
134 # devices at launch unless it is told not to.
135 refused_all: bool = False
136 # Which backend the engine selected. Defaults to UNKNOWN so a probe that never
137 # ran reports no claim rather than CPU.
138 backend: EngineBackend = EngineBackend.UNKNOWN
141def _parse_topo_matrix(topo_text: str) -> tuple[set[int], set[frozenset[int]]]:
142 """GPU row indices and NVLink-joined pairs from ``nvidia-smi topo -m`` output.
144 The matrix header row labels the GPU columns; each ``GPU<r>`` row lists the
145 link type to each column (``NV#`` is NVLink; ``PIX``/``PHB``/``SYS`` are PCIe).
146 """
147 header_cols: list[int] = []
148 gpu_rows: set[int] = set()
149 pairs: set[frozenset[int]] = set()
150 for line in _ANSI_SGR_RE.sub("", topo_text).splitlines():
151 tokens = line.split()
152 # Leading run of GPU-label tokens: the header is all labels (>=2), a data
153 # row is one label ("GPU3") followed by link-type cells. split() strips the
154 # header's leading whitespace, so this run length is what tells them apart.
155 leading_labels: list[int] = []
156 for token in tokens:
157 match = _GPU_LABEL_RE.match(token)
158 if match is None:
159 break
160 leading_labels.append(int(match.group(1)))
161 if len(leading_labels) >= _TOPO_MIN_GPUS:
162 header_cols = leading_labels
163 elif len(leading_labels) == 1:
164 row_idx = leading_labels[0]
165 gpu_rows.add(row_idx)
166 for col_idx, cell in zip(header_cols, tokens[1:], strict=False):
167 if row_idx != col_idx and cell.startswith("NV"):
168 pairs.add(frozenset({row_idx, col_idx}))
169 return gpu_rows, pairs
172def host_lacks_nvlink() -> bool:
173 """Whether this host's GPUs are joined only by PCIe (no NVLink anywhere).
175 Tensor-splitting a large model across PCIe-only cards is all-reduce bound and
176 much slower than over NVLink. Deliberately a host-level claim: the fleet's
177 device indices live in the serving binary's backend index space, which does
178 not map onto ``nvidia-smi``'s physical numbering under a visible-devices
179 restriction (the very hazard this module exists to avoid), so per-pair
180 verdicts against plan indices would be unreliable. Returns False (no claim)
181 when the probe fails or reports fewer than two GPUs, so a non-NVIDIA or
182 single-card host stays silent rather than warning wrongly.
183 """
184 try:
185 stdout, _ = run_bounded(
186 ["nvidia-smi", "topo", "-m"],
187 timeout_s=_TOPO_TIMEOUT_S,
188 kill_wait_s=_PROBE_KILL_WAIT_S,
189 label="nvidia-smi topo",
190 )
191 except (OSError, subprocess.SubprocessError):
192 return False
193 gpu_rows, pairs = _parse_topo_matrix(stdout)
194 return len(gpu_rows) >= _TOPO_MIN_GPUS and not pairs
197def _probe_env() -> dict[str, str]:
198 """Env for the probe: stable PCI ordering so CUDA indices match what we pin.
200 A preset ``CUDA_DEVICE_ORDER`` is respected; ``visible_env`` re-emits the same
201 order var, so the probe and the spawned servers see one device ordering.
202 """
203 env = dict(os.environ)
204 env.setdefault(_CUDA_ORDER_VAR, _PCI_BUS_ID_ORDER)
205 return env
208def probe_devices(binary: Path, *, timeout_s: float = _LIST_DEVICES_TIMEOUT_S) -> DeviceProbe:
209 """Parse ``<binary> --list-devices``; empty devices when unavailable/unparseable.
211 Filtered to a single GPU backend (the highest-ranked one present) so device
212 indices are unambiguous when a build exposes several backends. A probe that
213 does not respond within *timeout_s* raises a ``ProviderError`` naming the
214 stuck probe: that is a wedged GPU driver, not a GPU-less host, and treating
215 it as "no devices" would silently plan a CPU fleet on a GPU box.
216 """
217 try:
218 output, returncode = _run_list_devices(binary, timeout_s)
219 except (OSError, subprocess.SubprocessError) as exc:
220 # Silently returning an empty probe here made an unrunnable binary look
221 # exactly like a host with no GPU, and the fleet planned for CPU with
222 # nothing said. The reason is the whole diagnosis: a wrong architecture,
223 # a missing loader, a permission denial.
224 log.warning(
225 "Could not run the GPU device probe (%s --list-devices): %s. Continuing "
226 "as though this host has no GPU; check that the engine binary is "
227 "executable and built for this machine.",
228 binary.name,
229 exc,
230 )
231 return DeviceProbe([], "")
232 parsed = _parse_devices(output)
233 selected = _select_backend(parsed)
234 offered = [d for d in parsed if d.backend in _BACKEND_RANK]
235 answered = _DEVICE_LIST_HEADER in output
236 spoke = returncode == 0 and answered
237 if not spoke and answered:
238 # It knew the flag and started answering, then died. Blaming the flag
239 # here sent the reader looking for the wrong engine build, when what
240 # they have is a crash partway through enumeration.
241 log.warning(
242 "%s --list-devices printed its device header then crashed (exit %d), so the "
243 "device list may be incomplete. This is usually a GPU driver or ICD fault "
244 "during enumeration. The probe reported: %s",
245 binary.name,
246 returncode,
247 _probe_tail(output),
248 )
249 elif not spoke:
250 log.warning(
251 "%s --list-devices exited %d without printing its device header, so it "
252 "does not appear to support the flag. Falling back to the host's Vulkan "
253 "loader to find GPUs; set %s if this is not the engine you meant to use.",
254 binary.name,
255 returncode,
256 "LILBEE_ENGINE_DIR",
257 )
258 return DeviceProbe(
259 selected,
260 output,
261 spoke_protocol=spoke,
262 refused_all=bool(offered) and not selected,
263 backend=_selected_backend(selected, spoke=spoke),
264 )
267def _selected_backend(selected: list[FleetDevice], *, spoke: bool) -> EngineBackend:
268 """Which backend the engine selected, or UNKNOWN when it did not answer.
270 An empty device list means CPU only when the engine answered the question. A
271 binary that never spoke the protocol said nothing about its backend, and
272 calling that CPU labels a GPU host as a CPU one in every diagnostic.
274 A selected device's backend is always a row of ``_BACKENDS``, because that is
275 what the selector filters on, so this indexes rather than defaulting. A
276 default here would turn a table the selector and the reporter disagree about
277 into a working GPU host that reports "unknown", which is this function's own
278 defect one backend along.
279 """
280 if selected:
281 return _REPORTED_BACKEND[selected[0].backend]
282 return EngineBackend.CPU if spoke else EngineBackend.UNKNOWN
285def _run_list_devices(binary: Path, timeout_s: float) -> tuple[str, int]:
286 """Run the probe with a bounded reap; raise on timeout.
288 A probe wedged in uninterruptible GPU-driver I/O would otherwise hang the
289 caller forever, since ``subprocess.run``'s timeout waits unbounded for the
290 reap; ``run_bounded`` abandons an unkillable child after a short wait.
292 The probe holds a device context and writes no state file, so nothing can reap
293 it later by record. It is the one caller that opts into the lifetime binding,
294 where the kernel offers one, and it is killed on the way out of every abort,
295 not just the timeout.
296 """
297 try:
298 return run_bounded(
299 [str(binary), "--list-devices"],
300 timeout_s=timeout_s,
301 kill_wait_s=_PROBE_KILL_WAIT_S,
302 env=_probe_env(),
303 merge_stderr=True,
304 label=f"{binary.name} --list-devices",
305 bind_lifetime=True,
306 )
307 except subprocess.TimeoutExpired as exc:
308 # Whatever the probe managed to print before it wedged says more than any
309 # fixed advice can, and the fixed advice named one vendor's tool at a host
310 # that may have neither that vendor nor that tool.
311 raise ProviderError(
312 f"The GPU device probe ({binary.name} --list-devices) did not respond "
313 f"within {timeout_s:.0f}s, so the engine cannot start. The GPU driver is "
314 "most likely wedged; check that your vendor's tool responds (nvidia-smi, "
315 "rocm-smi, xpu-smi) and reboot the host if it hangs.\n"
316 f"The probe reported: {_probe_tail(_decoded_output(exc.output))}",
317 provider=_PROVIDER,
318 kind=ProviderErrorKind.SERVER,
319 ) from None
322def _decoded_output(output: object) -> str:
323 """Partial child output from a timeout, which arrives as bytes even under text mode."""
324 if isinstance(output, bytes):
325 return output.decode(errors="replace")
326 return output if isinstance(output, str) else ""
329def _probe_tail(output: str) -> str:
330 """The tail of what the probe printed, for a message that has to stay readable."""
331 text = output.strip()
332 return text[-_PROBE_TAIL_CHARS:] if text else "(nothing)"
335def _parse_devices(text: str) -> list[FleetDevice]:
336 devices: list[FleetDevice] = []
337 # Sampled at most once per parse, and only when a line actually needs it:
338 # free memory is live, so it is read fresh here rather than cached, and the
339 # loader must not be opened once per device line to answer the same question.
340 loader_free: dict[str, int] | None = None
341 for line in text.splitlines():
342 match = _DEVICE_RE.match(line)
343 if match is None:
344 continue
345 backend, index, name, total_mib, free_mib = match.groups()
346 total = int(total_mib) * MIB
347 if total == 0:
348 # No memory is not a small GPU, it is one that cannot hold a model:
349 # a driver listing an adapter before its memory is queryable. Kept, it
350 # is the smallest card in the fleet and collapses every budget sized
351 # against the smallest, while the non-empty list switches off the
352 # shared-memory budget a host with no usable GPU depends on.
353 log.warning(
354 "Ignoring GPU %s%s (%s): it reports no memory, so nothing can be "
355 "placed on it. Check the GPU driver if this device should be usable.",
356 backend,
357 index,
358 name.strip(),
359 )
360 continue
361 if free_mib:
362 free = int(free_mib) * MIB
363 else:
364 if loader_free is None:
365 loader_free = _loader_free_bytes(backend)
366 free = loader_free.get(name.strip(), total)
367 devices.append(
368 FleetDevice(
369 backend,
370 int(index),
371 name.strip(),
372 total,
373 free,
374 unified=_is_unified(backend, name.strip()) or total < _DEDICATED_VRAM_FLOOR,
375 )
376 )
377 return devices
380# Mesa and friends expose CPU rasterizers through the Vulkan loader, and
381# llama.cpp's Vulkan backend enumerates them exactly like a GPU: same
382# "VulkanN: <name> (<total> MiB, <free> MiB free)" shape, with system RAM
383# reported as VRAM. Planning against one is worse than having no GPU at all,
384# because the "VRAM" looks enormous: a host with a real iGPU beside lavapipe
385# can be planned as a two-GPU machine and tensor-split across a real adapter
386# and a software renderer, which runs orders of magnitude slower than either
387# CPU inference or the iGPU alone.
388_SOFTWARE_RENDERER_MARKERS = ("llvmpipe", "lavapipe", "softpipe", "swiftshader")
391def _is_software_renderer(device: FleetDevice) -> bool:
392 """Whether *device* is a CPU rasterizer masquerading as a GPU.
394 A name test, so it only recognizes the rasterizers it already knows, and a
395 renamed or newly written one walks past it. It stays as the answer for hosts
396 where the Vulkan loader can't be opened from this process and the device
397 type is therefore unavailable; where the type is available,
398 ``_is_unusable_vulkan`` decides and this never gets the chance to be wrong.
399 """
400 name = device.name.casefold()
401 return any(marker in name for marker in _SOFTWARE_RENDERER_MARKERS)
404def _loader_free_bytes(backend: str) -> dict[str, int]:
405 """Live free memory per device name, for a listing that printed no free figure.
407 ggml omits the figure when the driver has no ``VK_EXT_memory_budget``, and
408 treating the omission as "all of it" is how a desktop holding gigabytes of
409 compositor and browser VRAM was planned as an empty card. The loader exposes
410 that extension to this process even when the engine build cannot use it, so
411 it is asked directly; a name it cannot speak for keeps the heap size.
413 Empty for any other backend: the Vulkan loader knows nothing about the
414 devices a CUDA or ROCm listing names.
415 """
416 if backend != VULKAN_BACKEND:
417 return {}
418 from lilbee.providers.fleet.gpu_select import vulkan_free_bytes_by_name
420 return vulkan_free_bytes_by_name()
423def _vulkan_device_type(name: str) -> VkDeviceType | None:
424 """The loader's type for the Vulkan adapter the engine printed as *name*.
426 ``None`` when the loader can't be reached or reports no adapter by that
427 name, which reads as "no opinion": the device is kept and assumed dedicated,
428 preserving the behaviour of hosts that never had a type to consult.
429 """
430 from lilbee.providers.fleet.gpu_select import vulkan_device_types_by_name
432 return vulkan_device_types_by_name().get(name)
435def _is_unusable_vulkan(device: FleetDevice) -> bool:
436 """Whether *device* is a Vulkan adapter ggml would not choose to run on.
438 ggml's Vulkan backend builds its device pool from discrete and integrated
439 adapters only, and falls back to the first non-CPU adapter when it finds
440 neither. In a VM that fallback is a paravirtual adapter (VMware SVGA,
441 VirtIO-GPU Venus, QXL), which reports guest RAM as VRAM and is typically
442 compute-incomplete or fails at allocation. Planning a fleet onto one costs
443 more than planning no GPU at all, since a non-empty device list also turns
444 off the shared-RAM budget.
446 Only a positive claim counts. VIRTUAL_GPU and CPU are the loader naming what
447 the adapter is; OTHER is it declining to, and refusing on a shrug took the
448 GPU away from real hardware whose driver simply does not classify itself.
449 """
450 if device.backend != VULKAN_BACKEND:
451 return False
452 device_type = _vulkan_device_type(device.name)
453 if device_type is None or device_type in USABLE_VULKAN_TYPES:
454 return False
455 # OTHER is the loader shrugging, not an accusation. The spec's own wording is
456 # "does not match any other available types", which a driver reaches for when
457 # it cannot classify itself, and some real adapters do. Refusing on it took a
458 # working GPU away from a machine the engine had already listed one for.
459 # VIRTUAL_GPU and CPU are positive claims and keep their veto.
460 return device_type is not VkDeviceType.OTHER
463def _is_unified(backend: str, name: str) -> bool:
464 """Whether the device *backend* printed as *name* shares its memory with the host.
466 Metal is unified by construction on Apple Silicon: the figure it reports is
467 ``recommendedMaxWorkingSetSize``, a slice of system RAM rather than a
468 separate pool. For Vulkan the loader knows the device type, so the type is
469 asked for rather than guessed; a size heuristic cannot work here, since a
470 24 GB discrete card in a 32 GB host and an Apple GPU reporting two thirds of
471 RAM are indistinguishable by proportion.
473 CUDA, ROCm and SYCL print no type at all, and an AMD APU or a Jetson looks
474 exactly like a discrete card there while reporting system RAM as its memory.
475 Those fall back to a question about the machine rather than the device: a
476 host whose Vulkan loader sees adapters but no discrete one has no discrete
477 GPU for another backend to be enumerating.
478 """
479 if backend in _UNIFIED_BACKENDS:
480 return True
481 if backend == VULKAN_BACKEND:
482 return _vulkan_device_type(name) is VkDeviceType.INTEGRATED_GPU
483 from lilbee.providers.fleet.gpu_select import host_has_no_discrete_gpu
485 return host_has_no_discrete_gpu()
488def _select_backend(devices: list[FleetDevice]) -> list[FleetDevice]:
489 """Keep one GPU backend's devices: highest rank, then most memory.
491 Returns a single backend so pinning is unambiguous: ``visible_env`` keys off
492 one backend, and mixing index spaces is the very hazard this module avoids.
494 CUDA, ROCm, HIP and Metal all rank alike, and a build that loads several
495 backends (``ggml_backend_load_all`` does) makes the tie real. Breaking it on
496 the backend's name meant a host with a 4090 beside an RX 6600 planned onto
497 the AMD card because "ROCm" sorts after "CUDA", and the NVIDIA card idled
498 with nothing said. Total memory decides instead; the name is only the last
499 resort that keeps the choice deterministic.
500 """
501 ranked = [
502 d
503 for d in devices
504 if d.backend in _BACKEND_RANK
505 and not _is_software_renderer(d)
506 and not _is_unusable_vulkan(d)
507 ]
508 if not ranked:
509 return []
510 by_backend: dict[str, list[FleetDevice]] = {}
511 for device in ranked:
512 by_backend.setdefault(device.backend, []).append(device)
513 backend, chosen = max(by_backend.items(), key=_backend_preference)
514 for other, group in by_backend.items():
515 if other != backend:
516 log.info(
517 "Engine reports %d %s device(s) beside %d %s device(s); planning onto %s, "
518 "which has more memory. Backends cannot be mixed: their device indexes "
519 "name different cards.",
520 len(group),
521 other,
522 len(chosen),
523 backend,
524 backend,
525 )
526 return chosen
529def _backend_preference(item: tuple[str, list[FleetDevice]]) -> tuple[int, int, int, str]:
530 """Sort key for choosing one backend's devices: rank, dedicated bytes, size.
532 Dedicated bytes come before raw size because the discrete backends all tie at
533 the same rank, and a shared-heap carveout reports a total that is host RAM
534 the host budget already counts. Left on raw size, an APU advertising a large
535 carveout beat a discrete card, which was then discarded and left idle while
536 the plan double-promised memory it did not have.
537 """
538 backend, group = item
539 dedicated = sum(d.total_bytes for d in group if not d.unified)
540 return _BACKEND_RANK[backend], dedicated, sum(d.total_bytes for d in group), backend
543def _compose_visible(indices: list[int], parent_value: str | None) -> str:
544 """Visible-devices value naming the same physical devices the probe saw.
546 When the parent env already restricts the var, the probe's indices are
547 relative to that comma-separated list (integer or UUID entries), so each
548 index maps through it; the child's value then names the same physical
549 devices instead of being re-interpreted as absolute.
550 """
551 if parent_value is None:
552 return ",".join(str(i) for i in indices)
553 entries = [entry.strip() for entry in parent_value.split(",") if entry.strip()]
554 out: list[str] = []
555 for i in indices:
556 if i >= len(entries):
557 # The probe enumerates devices under the parent restriction, so every
558 # index must map into it. An out-of-range index is an invariant
559 # violation; emitting a bare ``str(i)`` would pin an absolute integer
560 # into a possibly UUID-namespaced list, silently selecting the wrong
561 # GPU. Fail loudly instead.
562 raise ValueError(
563 f"device index {i} is outside the parent visible-devices list "
564 f"{parent_value!r}; cannot compose a child pin without selecting the wrong GPU"
565 )
566 out.append(entries[i])
567 return ",".join(out)
570def visible_env(devices: tuple[FleetDevice, ...]) -> dict[str, str]:
571 """Env that pins a child to *devices* via the right var for their backend.
573 Indices are the backend-native ones from ``probe_devices``, composed through
574 any parent visible-devices restriction so the child names the same physical
575 devices the probe enumerated; no cross-API index translation occurs.
576 """
577 if not devices:
578 return {}
579 backend = devices[0].backend
580 indices = [d.index for d in devices]
581 if backend == "CUDA":
582 return {
583 _CUDA_VISIBLE_VAR: _compose_visible(indices, os.environ.get(_CUDA_VISIBLE_VAR)),
584 _CUDA_ORDER_VAR: os.environ.get(_CUDA_ORDER_VAR, _PCI_BUS_ID_ORDER),
585 }
586 if backend in ("ROCm", "HIP"):
587 return _amd_visible_env(indices)
588 if backend == VULKAN_BACKEND:
589 # Deliberately not GGML_VK_VISIBLE_DEVICES. That variable indexes the raw
590 # loader enumeration, while these indices come from the engine's own
591 # filtered list, so the two disagree wherever ggml drops or merges a
592 # device -- two ICDs for one card being the clear case. Setting it also
593 # disables ggml's type filter, support check and dedup. Vulkan is pinned
594 # with --device instead, in the same space the names were parsed from.
595 return {}
596 if backend == "SYCL":
597 # Deliberately no ONEAPI_DEVICE_SELECTOR. It is a selector grammar over a
598 # backend runtime, not the index space --list-devices numbers, so a
599 # composed level_zero ordinal can name a different physical card than the
600 # one the probe enumerated. SYCL pins by --device instead, in the space
601 # the indices were read from. An inherited parent selector still applies:
602 # the engine enumerated behind it, so its names are already relative to it.
603 return {}
604 return {}
607def amd_visible_var() -> str:
608 """The one AMD visibility var an index list may be written to.
610 ``ROCR_VISIBLE_DEVICES`` and ``HIP_VISIBLE_DEVICES`` are applied sequentially:
611 ROCr filters first, then HIP re-indexes within the survivors. Writing the same
612 indices to both double-filters and selects the wrong cards, or none at all.
613 ``GPU_DEVICE_ORDINAL`` is the third and filters the same way, so writing HIP
614 over an ordinal mask both overrides it and re-exposes cards it had hidden.
616 So exactly one is ever written: whichever the environment already restricts,
617 in the runtime's precedence (HIP, then the ordinal, then ROCr), or HIP when
618 nothing restricts. An empty value means "no devices" rather than "this is the
619 variable in use", so it does not claim precedence. Every caller writing an AMD
620 pin asks here; two callers each picking their own would put the pair back.
621 """
622 for name in (_HIP_VISIBLE_VAR, _GPU_DEVICE_ORDINAL_VAR, _ROCR_VISIBLE_VAR):
623 if os.environ.get(name, "").strip():
624 return name
625 return _HIP_VISIBLE_VAR
628def _amd_visible_env(indices: list[int]) -> dict[str, str]:
629 """Pin an AMD ROCm/HIP child to the probe's *indices* with one visibility var.
631 The probe enumerated a single index space already filtered by whichever var
632 the parent set, so the chosen var is composed against that parent value and
633 the other is left inherited untouched. The child inherits the parent env, so
634 an unset override keeps any inherited sibling var in force.
635 """
636 var = amd_visible_var()
637 return {var: _compose_visible(indices, os.environ.get(var))}