Coverage for src/lilbee/runtime/hardware.py: 100%
79 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
1"""Hardware-fit signaling and per-row size-variant grouping for the catalog."""
3from __future__ import annotations
5import threading
6from collections.abc import Callable
7from dataclasses import dataclass
8from enum import StrEnum
9from typing import Any
11from cachetools import TTLCache, cached
12from pydantic import BaseModel
14from lilbee.catalog.models import CatalogModel, ModelFamily
15from lilbee.core.config import cfg
17_BYTES_PER_GB = 1024**3
18_FITS_HEADROOM_BYTES = 1 * _BYTES_PER_GB
19_MEMORY_PROBE_TTL_S = 60.0
21# The probe is an nvidia-smi subprocess without pynvml; the catalog stamps a fit
22# chip on every page, so repeated requests share one probe per fraction.
23_available_memory_cache: TTLCache[Any, int] = TTLCache(maxsize=8, ttl=_MEMORY_PROBE_TTL_S)
24_available_memory_lock = threading.Lock()
27class FitLevel(StrEnum):
28 FITS = "fits"
29 TIGHT = "tight"
30 WONT_RUN = "wont_run"
33# Fit levels in rank order, best first. Derived from the enum so a new level
34# cannot miss the map.
35FIT_RANK: dict[FitLevel, int] = {level: rank for rank, level in enumerate(FitLevel)}
38@dataclass(frozen=True)
39class FitChip:
40 level: FitLevel
41 headroom_gb: float
44def compute_fit(model_size_bytes: int, available_bytes: int) -> FitChip:
45 """Classify how a model footprint fits the available memory budget.
47 Headroom_gb is positive when the model fits and negative when it
48 won't. The 1 GB band between FITS and TIGHT leaves room for the
49 inference runtime, KV cache, and OS overhead beyond the raw weight
50 file.
51 """
52 headroom_bytes = available_bytes - model_size_bytes
53 headroom_gb = headroom_bytes / _BYTES_PER_GB
54 if headroom_bytes >= _FITS_HEADROOM_BYTES:
55 level = FitLevel.FITS
56 elif headroom_bytes >= 0:
57 level = FitLevel.TIGHT
58 else:
59 level = FitLevel.WONT_RUN
60 return FitChip(level=level, headroom_gb=headroom_gb)
63def chip_for_size(size_gb: float, available_bytes: int | None) -> FitChip | None:
64 """Fit chip for a *size_gb* footprint, or None when it cannot be measured."""
65 if available_bytes is None or size_gb <= 0:
66 return None
67 return compute_fit(int(size_gb * _BYTES_PER_GB), available_bytes)
70def fit_for_size(size_gb: float, available_bytes: int | None) -> FitLevel | None:
71 """Fit level for a *size_gb* footprint, or None when it cannot be measured."""
72 chip = chip_for_size(size_gb, available_bytes)
73 return None if chip is None else chip.level
76def make_fit_filter(
77 worst: FitLevel | None, available_bytes: int | None
78) -> Callable[[CatalogModel], bool] | None:
79 """Row predicate for a *worst* acceptable fit, or None when no fit was asked for.
81 A row whose fit cannot be measured is kept.
82 """
83 if worst is None:
84 return None
85 worst_rank = FIT_RANK[worst]
87 def keep(model: CatalogModel) -> bool:
88 level = fit_for_size(model.size_gb, available_bytes)
89 return level is None or FIT_RANK[level] <= worst_rank
91 return keep
94def available_memory_for_fit() -> int | None:
95 """Bytes available to a model after ``cfg.gpu_memory_fraction``, or None on probe failure.
97 Sums every GPU's memory (``total=True``) because lilbee tensor-splits a model
98 too large for one card across the whole fleet; sizing the fit chip against a
99 single card would wrongly mark a runnable split model "won't run". The actual
100 per-card placement is decided precisely by the fleet planner at load time.
102 Single entry point so the TUI and the HTTP catalog handler classify fit
103 against the same number; otherwise the same model would chip differently in
104 each surface.
105 """
106 try:
107 budget = _probe_available_memory(cfg.gpu_memory_fraction)
108 except Exception:
109 return None
110 return budget + _expert_offload_headroom()
113@cached(_available_memory_cache, lock=_available_memory_lock)
114def _probe_available_memory(fraction: float) -> int:
115 """Whole-fleet memory budget after *fraction*, one probe per fraction per TTL."""
116 from lilbee.providers.model_cache import get_available_memory
118 return get_available_memory(fraction, total=True)
121def _expert_offload_headroom() -> int:
122 """System memory the fit budget may borrow when expert offload is configured.
124 A sparse model's experts live in system RAM under offload, so a host whose
125 budget is discrete VRAM can run a model larger than that VRAM and must not
126 be told otherwise. Zero unless the budget really is device memory: every
127 other path (Apple unified memory, a non-NVIDIA or CPU-only host) already
128 reports system RAM, and adding it twice would invent capacity. Zero too for a
129 non-positive ``n_cpu_moe``, which offloads nothing. The chip is per-family and
130 this budget is global, so it reads optimistically for a dense model pulled on
131 an offload-enabled host (a sparse model gains the room, a dense one still
132 fails to place); the planner sizes the real placement at load time.
134 Scaled from installed RAM, not from what is free this instant, to match the
135 capacity basis of the VRAM budget it is added to. Mixing the two made a
136 catalog entry fit or not fit depending on whatever else the machine happened
137 to be doing when the page was drawn, and shrank the budget exactly when
138 another model was already resident.
139 """
140 from lilbee.providers.model_cache import has_nvidia_gpu, total_system_memory
142 if not (cfg.cpu_moe or (cfg.n_cpu_moe is not None and cfg.n_cpu_moe >= 1)):
143 return 0
144 try:
145 if not has_nvidia_gpu():
146 return 0
147 return int(total_system_memory() * cfg.gpu_memory_fraction)
148 except Exception:
149 return 0
152class SizeVariantInfo(BaseModel):
153 """One size/quant of a model family, serialised for HTTP responses."""
155 size_label: str
156 params: str
157 size_gb: float
158 ref: str
161def family_size_variants(family: ModelFamily) -> list[SizeVariantInfo]:
162 """Build the per-row size-variant strip for a featured ModelFamily, smallest first."""
163 variants = sorted(family.variants, key=lambda v: v.size_mb)
164 return [
165 SizeVariantInfo(
166 size_label=_size_variant_label(v.param_count, v.quant),
167 params=v.param_count,
168 size_gb=v.size_mb / 1024,
169 ref=v.hf_repo,
170 )
171 for v in variants
172 ]
175def _size_variant_label(param_count: str, quant: str) -> str:
176 """Render the compact label for one size variant (``8B Q4_K_M``)."""
177 pieces = [p for p in (param_count, quant) if p]
178 return " ".join(pieces) if pieces else "--"