Coverage for src/lilbee/runtime/hardware.py: 100%

79 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""Hardware-fit signaling and per-row size-variant grouping for the catalog.""" 

2 

3from __future__ import annotations 

4 

5import threading 

6from collections.abc import Callable 

7from dataclasses import dataclass 

8from enum import StrEnum 

9from typing import Any 

10 

11from cachetools import TTLCache, cached 

12from pydantic import BaseModel 

13 

14from lilbee.catalog.models import CatalogModel, ModelFamily 

15from lilbee.core.config import cfg 

16 

17_BYTES_PER_GB = 1024**3 

18_FITS_HEADROOM_BYTES = 1 * _BYTES_PER_GB 

19_MEMORY_PROBE_TTL_S = 60.0 

20 

21# The probe is an nvidia-smi subprocess without pynvml; the catalog stamps a fit 

22# chip on every page, so repeated requests share one probe per fraction. 

23_available_memory_cache: TTLCache[Any, int] = TTLCache(maxsize=8, ttl=_MEMORY_PROBE_TTL_S) 

24_available_memory_lock = threading.Lock() 

25 

26 

27class FitLevel(StrEnum): 

28 FITS = "fits" 

29 TIGHT = "tight" 

30 WONT_RUN = "wont_run" 

31 

32 

33# Fit levels in rank order, best first. Derived from the enum so a new level 

34# cannot miss the map. 

35FIT_RANK: dict[FitLevel, int] = {level: rank for rank, level in enumerate(FitLevel)} 

36 

37 

38@dataclass(frozen=True) 

39class FitChip: 

40 level: FitLevel 

41 headroom_gb: float 

42 

43 

44def compute_fit(model_size_bytes: int, available_bytes: int) -> FitChip: 

45 """Classify how a model footprint fits the available memory budget. 

46 

47 Headroom_gb is positive when the model fits and negative when it 

48 won't. The 1 GB band between FITS and TIGHT leaves room for the 

49 inference runtime, KV cache, and OS overhead beyond the raw weight 

50 file. 

51 """ 

52 headroom_bytes = available_bytes - model_size_bytes 

53 headroom_gb = headroom_bytes / _BYTES_PER_GB 

54 if headroom_bytes >= _FITS_HEADROOM_BYTES: 

55 level = FitLevel.FITS 

56 elif headroom_bytes >= 0: 

57 level = FitLevel.TIGHT 

58 else: 

59 level = FitLevel.WONT_RUN 

60 return FitChip(level=level, headroom_gb=headroom_gb) 

61 

62 

63def chip_for_size(size_gb: float, available_bytes: int | None) -> FitChip | None: 

64 """Fit chip for a *size_gb* footprint, or None when it cannot be measured.""" 

65 if available_bytes is None or size_gb <= 0: 

66 return None 

67 return compute_fit(int(size_gb * _BYTES_PER_GB), available_bytes) 

68 

69 

70def fit_for_size(size_gb: float, available_bytes: int | None) -> FitLevel | None: 

71 """Fit level for a *size_gb* footprint, or None when it cannot be measured.""" 

72 chip = chip_for_size(size_gb, available_bytes) 

73 return None if chip is None else chip.level 

74 

75 

76def make_fit_filter( 

77 worst: FitLevel | None, available_bytes: int | None 

78) -> Callable[[CatalogModel], bool] | None: 

79 """Row predicate for a *worst* acceptable fit, or None when no fit was asked for. 

80 

81 A row whose fit cannot be measured is kept. 

82 """ 

83 if worst is None: 

84 return None 

85 worst_rank = FIT_RANK[worst] 

86 

87 def keep(model: CatalogModel) -> bool: 

88 level = fit_for_size(model.size_gb, available_bytes) 

89 return level is None or FIT_RANK[level] <= worst_rank 

90 

91 return keep 

92 

93 

94def available_memory_for_fit() -> int | None: 

95 """Bytes available to a model after ``cfg.gpu_memory_fraction``, or None on probe failure. 

96 

97 Sums every GPU's memory (``total=True``) because lilbee tensor-splits a model 

98 too large for one card across the whole fleet; sizing the fit chip against a 

99 single card would wrongly mark a runnable split model "won't run". The actual 

100 per-card placement is decided precisely by the fleet planner at load time. 

101 

102 Single entry point so the TUI and the HTTP catalog handler classify fit 

103 against the same number; otherwise the same model would chip differently in 

104 each surface. 

105 """ 

106 try: 

107 budget = _probe_available_memory(cfg.gpu_memory_fraction) 

108 except Exception: 

109 return None 

110 return budget + _expert_offload_headroom() 

111 

112 

113@cached(_available_memory_cache, lock=_available_memory_lock) 

114def _probe_available_memory(fraction: float) -> int: 

115 """Whole-fleet memory budget after *fraction*, one probe per fraction per TTL.""" 

116 from lilbee.providers.model_cache import get_available_memory 

117 

118 return get_available_memory(fraction, total=True) 

119 

120 

121def _expert_offload_headroom() -> int: 

122 """System memory the fit budget may borrow when expert offload is configured. 

123 

124 A sparse model's experts live in system RAM under offload, so a host whose 

125 budget is discrete VRAM can run a model larger than that VRAM and must not 

126 be told otherwise. Zero unless the budget really is device memory: every 

127 other path (Apple unified memory, a non-NVIDIA or CPU-only host) already 

128 reports system RAM, and adding it twice would invent capacity. Zero too for a 

129 non-positive ``n_cpu_moe``, which offloads nothing. The chip is per-family and 

130 this budget is global, so it reads optimistically for a dense model pulled on 

131 an offload-enabled host (a sparse model gains the room, a dense one still 

132 fails to place); the planner sizes the real placement at load time. 

133 

134 Scaled from installed RAM, not from what is free this instant, to match the 

135 capacity basis of the VRAM budget it is added to. Mixing the two made a 

136 catalog entry fit or not fit depending on whatever else the machine happened 

137 to be doing when the page was drawn, and shrank the budget exactly when 

138 another model was already resident. 

139 """ 

140 from lilbee.providers.model_cache import has_nvidia_gpu, total_system_memory 

141 

142 if not (cfg.cpu_moe or (cfg.n_cpu_moe is not None and cfg.n_cpu_moe >= 1)): 

143 return 0 

144 try: 

145 if not has_nvidia_gpu(): 

146 return 0 

147 return int(total_system_memory() * cfg.gpu_memory_fraction) 

148 except Exception: 

149 return 0 

150 

151 

152class SizeVariantInfo(BaseModel): 

153 """One size/quant of a model family, serialised for HTTP responses.""" 

154 

155 size_label: str 

156 params: str 

157 size_gb: float 

158 ref: str 

159 

160 

161def family_size_variants(family: ModelFamily) -> list[SizeVariantInfo]: 

162 """Build the per-row size-variant strip for a featured ModelFamily, smallest first.""" 

163 variants = sorted(family.variants, key=lambda v: v.size_mb) 

164 return [ 

165 SizeVariantInfo( 

166 size_label=_size_variant_label(v.param_count, v.quant), 

167 params=v.param_count, 

168 size_gb=v.size_mb / 1024, 

169 ref=v.hf_repo, 

170 ) 

171 for v in variants 

172 ] 

173 

174 

175def _size_variant_label(param_count: str, quant: str) -> str: 

176 """Render the compact label for one size variant (``8B Q4_K_M``).""" 

177 pieces = [p for p in (param_count, quant) if p] 

178 return " ".join(pieces) if pieces else "--"