Coverage for src/lilbee/catalog/models.py: 100%

112 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""Catalog dataclasses and pydantic types. Imports only the catalog's leaf modules.""" 

2 

3import functools 

4import re 

5from dataclasses import dataclass 

6 

7from pydantic import BaseModel 

8 

9from lilbee.catalog.refs import ggml_bytes_per_param, ggml_quant_block_sizes, quant_label 

10from lilbee.catalog.types import ModelCompat, ModelTask 

11 

12# Minimum recommended floor so a tiny model still reports a sane RAM ask. 

13_MIN_RAM_FLOOR_GB = 2.0 

14# Working-set multiple over the on-disk size (weights + KV cache + overhead). 

15_RAM_OVER_SIZE_FACTOR = 1.5 

16 

17_BYTES_PER_GB = 1024**3 

18 

19 

20@functools.cache 

21def _default_bytes_per_param() -> float: 

22 """Bytes per weight of Q4_K, the type a filename naming no quant is sized as. 

23 

24 Q4_K heads the pull path's quant preference, so it is the type a pull would 

25 most likely land on. 

26 """ 

27 block, type_size = ggml_quant_block_sizes()["Q4_K"] 

28 return type_size / block 

29 

30 

31# A quant ggml does not name still says how many bits it packs. One fp16 scale 

32# per group costs an eighth on top, whatever the width, because a group is sized 

33# to the width: 1-bit in groups of 128, 2-bit in 64, 4-bit in 32 all carry two 

34# bytes per group. Reading the width beats falling back to Q4_K_M, which reports 

35# a 2-bit file at more than twice its size. 

36_SCALE_OVERHEAD = 1.125 

37_BITS_PER_BYTE = 8 

38 

39_WIDTH_RE = re.compile(r"I?Q(\d)") 

40 

41 

42def _width_bytes_per_param(quant: str) -> float | None: 

43 """Bytes per weight from the bit width *quant* names, or None if it names none.""" 

44 match = _WIDTH_RE.match(quant) 

45 if match is None: 

46 return None 

47 return int(match.group(1)) / _BITS_PER_BYTE * _SCALE_OVERHEAD 

48 

49 

50def _quant_bytes_per_param(gguf_filename: str) -> float: 

51 """Bytes per weight of the ggml type *gguf_filename* names, or Q4_K when it names none. 

52 

53 The type's own block arithmetic answers first, so a label cannot read under 

54 what its tensors physically cost. The bit width is the last resort, for a 

55 publisher's own naming that ggml has no type for. 

56 """ 

57 quant = quant_label(gguf_filename) 

58 rate = ggml_bytes_per_param(quant) 

59 if rate is not None: 

60 return rate 

61 width = _width_bytes_per_param(quant) 

62 return width if width is not None else _default_bytes_per_param() 

63 

64 

65def estimate_min_ram_gb(size_gb: float) -> float: 

66 """Estimate the RAM a model needs from its on-disk size (single source).""" 

67 return round(max(_MIN_RAM_FLOOR_GB, size_gb * _RAM_OVER_SIZE_FACTOR), 1) 

68 

69 

70def estimate_size_gb(params: int, gguf_filename: str) -> float: 

71 """Approximate the on-disk GB of *gguf_filename* from a model's parameter count. 

72 

73 A lower bound, and the browse list renders it as approximate. llama.cpp 

74 promotes ``output.weight`` and an untied ``token_embd`` above the ftype and 

75 leaves the norms in F32, so a published file costs more per weight than the 

76 type its name carries; how much more needs the header's vocabulary and 

77 embedding lengths, which a listing row does not have. 

78 

79 This is the one place a size cannot be read: the HF listing API reports a 

80 parameter count (``gguf.total``) and no per-file bytes, and getting the real 

81 figure for a 50-row page means 50 more requests. Every path that acts on a 

82 size resolves the exact one for the single file in play. 

83 """ 

84 if params <= 0: 

85 return 0.0 # unknown: display as "?" in UI 

86 return round(params * _quant_bytes_per_param(gguf_filename) / _BYTES_PER_GB, 1) 

87 

88 

89class HfGgufMeta(BaseModel): 

90 """GGUF metadata returned by the HF API when expand=gguf is requested. 

91 

92 ModelInfo.gguf is typed as ``dict | None`` upstream, so we validate it ourselves. 

93 

94 ``total`` is the model's parameter count, not a byte size; ``totalFileSize`` 

95 holds bytes. Verified against repos that name their own parameter count: 

96 Qwen3-8B-GGUF reports ``total=8_190_000_000`` against 4.7 GB of files. 

97 """ 

98 

99 total: int = 0 

100 architecture: str = "" 

101 context_length: int = 0 

102 

103 

104@dataclass 

105class DownloadProgress: 

106 """Human-readable snapshot of download progress. 

107 

108 ``percent`` is a float (0.0 to 100.0) so the ProgressBar renders smooth 

109 fractional movement during multi-GB downloads. Call sites that need 

110 an integer for display format it themselves. 

111 """ 

112 

113 percent: float 

114 detail: str 

115 is_cache_hit: bool 

116 

117 

118@dataclass(frozen=True) 

119class CatalogModel: 

120 """One catalog entry, keyed by HuggingFace repo. ``gguf_filename`` may be a glob.""" 

121 

122 hf_repo: str 

123 gguf_filename: str 

124 size_gb: float 

125 min_ram_gb: float 

126 description: str 

127 featured: bool 

128 downloads: int 

129 task: ModelTask 

130 architecture: str = "" 

131 compat: ModelCompat = ModelCompat.UNKNOWN 

132 # Parameter count. Size buckets key off this rather than on-disk bytes so a 

133 # model keeps its bucket across quants. 0 when the repo publishes no GGUF 

134 # metadata. 

135 params: int = 0 

136 # HuggingFace trending rank. 0 when the listing omits it. 

137 trending_score: int = 0 

138 # Safety-stripped (abliterated/uncensored) per the repo's HF tags. Browse 

139 # rows carry it; recommendation rails exclude rows that set it. 

140 safety_stripped: bool = False 

141 

142 @property 

143 def ref(self) -> str: 

144 """Browse-time ref (the HF repo); concrete filename is resolved at install.""" 

145 return self.hf_repo 

146 

147 @property 

148 def display_name(self) -> str: 

149 """Human-readable label derived from the HuggingFace repo id.""" 

150 # circular: models -> formatting via clean_display_name 

151 from lilbee.catalog.formatting import clean_display_name 

152 

153 return clean_display_name(self.hf_repo) 

154 

155 

156@dataclass(frozen=True) 

157class CatalogResult: 

158 """Paginated catalog result. 

159 

160 ``truncated`` is True when the HuggingFace scan stopped at its bound with 

161 rows left unread, so matches past them are unreachable at any offset. 

162 """ 

163 

164 total: int | None 

165 limit: int 

166 offset: int 

167 models: list[CatalogModel] 

168 has_more: bool = False 

169 truncated: bool = False 

170 

171 

172@dataclass(frozen=True) 

173class PageWindow: 

174 """The part of a page left for the rows paged after the ones held locally.""" 

175 

176 rest_offset: int 

177 rest_limit: int 

178 

179 

180def page_window(leading_count: int, offset: int, limit: int) -> PageWindow: 

181 """The window left of ``[offset, offset + limit)`` after *leading_count* local rows.""" 

182 covered = min(offset + limit, leading_count) - min(offset, leading_count) 

183 return PageWindow(rest_offset=max(0, offset - leading_count), rest_limit=limit - covered) 

184 

185 

186@dataclass(frozen=True) 

187class HfPage: 

188 """One page of HuggingFace API results and the cursor of the page after it.""" 

189 

190 models: list[CatalogModel] 

191 next_cursor: str | None = None 

192 

193 @property 

194 def has_more(self) -> bool: 

195 """True when HuggingFace lists a next page.""" 

196 return self.next_cursor is not None 

197 

198 

199def dedupe_models(models: list[CatalogModel]) -> list[CatalogModel]: 

200 """Models in first-seen order, later repeats dropped.""" 

201 seen: set[str] = set() 

202 unique: list[CatalogModel] = [] 

203 for model in models: 

204 if model.hf_repo not in seen: 

205 seen.add(model.hf_repo) 

206 unique.append(model) 

207 return unique 

208 

209 

210@dataclass(frozen=True) 

211class ModelVariant: 

212 """One quantization within a model family. ``filename`` may be a glob.""" 

213 

214 hf_repo: str 

215 filename: str 

216 param_count: str 

217 quant: str 

218 size_mb: int 

219 mmproj_filename: str = "" 

220 compat: ModelCompat = ModelCompat.UNKNOWN 

221 safety_stripped: bool = False 

222 

223 

224@dataclass(frozen=True) 

225class ModelFamily: 

226 """A group of related model variants (e.g. Qwen3 in multiple sizes).""" 

227 

228 slug: str # family slug for building refs: "qwen3" 

229 name: str # display name: "Qwen3" 

230 task: ModelTask 

231 description: str 

232 variants: tuple[ModelVariant, ...]