Coverage for src/lilbee/catalog/models.py: 100%
112 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
1"""Catalog dataclasses and pydantic types. Imports only the catalog's leaf modules."""
3import functools
4import re
5from dataclasses import dataclass
7from pydantic import BaseModel
9from lilbee.catalog.refs import ggml_bytes_per_param, ggml_quant_block_sizes, quant_label
10from lilbee.catalog.types import ModelCompat, ModelTask
12# Minimum recommended floor so a tiny model still reports a sane RAM ask.
13_MIN_RAM_FLOOR_GB = 2.0
14# Working-set multiple over the on-disk size (weights + KV cache + overhead).
15_RAM_OVER_SIZE_FACTOR = 1.5
17_BYTES_PER_GB = 1024**3
20@functools.cache
21def _default_bytes_per_param() -> float:
22 """Bytes per weight of Q4_K, the type a filename naming no quant is sized as.
24 Q4_K heads the pull path's quant preference, so it is the type a pull would
25 most likely land on.
26 """
27 block, type_size = ggml_quant_block_sizes()["Q4_K"]
28 return type_size / block
31# A quant ggml does not name still says how many bits it packs. One fp16 scale
32# per group costs an eighth on top, whatever the width, because a group is sized
33# to the width: 1-bit in groups of 128, 2-bit in 64, 4-bit in 32 all carry two
34# bytes per group. Reading the width beats falling back to Q4_K_M, which reports
35# a 2-bit file at more than twice its size.
36_SCALE_OVERHEAD = 1.125
37_BITS_PER_BYTE = 8
39_WIDTH_RE = re.compile(r"I?Q(\d)")
42def _width_bytes_per_param(quant: str) -> float | None:
43 """Bytes per weight from the bit width *quant* names, or None if it names none."""
44 match = _WIDTH_RE.match(quant)
45 if match is None:
46 return None
47 return int(match.group(1)) / _BITS_PER_BYTE * _SCALE_OVERHEAD
50def _quant_bytes_per_param(gguf_filename: str) -> float:
51 """Bytes per weight of the ggml type *gguf_filename* names, or Q4_K when it names none.
53 The type's own block arithmetic answers first, so a label cannot read under
54 what its tensors physically cost. The bit width is the last resort, for a
55 publisher's own naming that ggml has no type for.
56 """
57 quant = quant_label(gguf_filename)
58 rate = ggml_bytes_per_param(quant)
59 if rate is not None:
60 return rate
61 width = _width_bytes_per_param(quant)
62 return width if width is not None else _default_bytes_per_param()
65def estimate_min_ram_gb(size_gb: float) -> float:
66 """Estimate the RAM a model needs from its on-disk size (single source)."""
67 return round(max(_MIN_RAM_FLOOR_GB, size_gb * _RAM_OVER_SIZE_FACTOR), 1)
70def estimate_size_gb(params: int, gguf_filename: str) -> float:
71 """Approximate the on-disk GB of *gguf_filename* from a model's parameter count.
73 A lower bound, and the browse list renders it as approximate. llama.cpp
74 promotes ``output.weight`` and an untied ``token_embd`` above the ftype and
75 leaves the norms in F32, so a published file costs more per weight than the
76 type its name carries; how much more needs the header's vocabulary and
77 embedding lengths, which a listing row does not have.
79 This is the one place a size cannot be read: the HF listing API reports a
80 parameter count (``gguf.total``) and no per-file bytes, and getting the real
81 figure for a 50-row page means 50 more requests. Every path that acts on a
82 size resolves the exact one for the single file in play.
83 """
84 if params <= 0:
85 return 0.0 # unknown: display as "?" in UI
86 return round(params * _quant_bytes_per_param(gguf_filename) / _BYTES_PER_GB, 1)
89class HfGgufMeta(BaseModel):
90 """GGUF metadata returned by the HF API when expand=gguf is requested.
92 ModelInfo.gguf is typed as ``dict | None`` upstream, so we validate it ourselves.
94 ``total`` is the model's parameter count, not a byte size; ``totalFileSize``
95 holds bytes. Verified against repos that name their own parameter count:
96 Qwen3-8B-GGUF reports ``total=8_190_000_000`` against 4.7 GB of files.
97 """
99 total: int = 0
100 architecture: str = ""
101 context_length: int = 0
104@dataclass
105class DownloadProgress:
106 """Human-readable snapshot of download progress.
108 ``percent`` is a float (0.0 to 100.0) so the ProgressBar renders smooth
109 fractional movement during multi-GB downloads. Call sites that need
110 an integer for display format it themselves.
111 """
113 percent: float
114 detail: str
115 is_cache_hit: bool
118@dataclass(frozen=True)
119class CatalogModel:
120 """One catalog entry, keyed by HuggingFace repo. ``gguf_filename`` may be a glob."""
122 hf_repo: str
123 gguf_filename: str
124 size_gb: float
125 min_ram_gb: float
126 description: str
127 featured: bool
128 downloads: int
129 task: ModelTask
130 architecture: str = ""
131 compat: ModelCompat = ModelCompat.UNKNOWN
132 # Parameter count. Size buckets key off this rather than on-disk bytes so a
133 # model keeps its bucket across quants. 0 when the repo publishes no GGUF
134 # metadata.
135 params: int = 0
136 # HuggingFace trending rank. 0 when the listing omits it.
137 trending_score: int = 0
138 # Safety-stripped (abliterated/uncensored) per the repo's HF tags. Browse
139 # rows carry it; recommendation rails exclude rows that set it.
140 safety_stripped: bool = False
142 @property
143 def ref(self) -> str:
144 """Browse-time ref (the HF repo); concrete filename is resolved at install."""
145 return self.hf_repo
147 @property
148 def display_name(self) -> str:
149 """Human-readable label derived from the HuggingFace repo id."""
150 # circular: models -> formatting via clean_display_name
151 from lilbee.catalog.formatting import clean_display_name
153 return clean_display_name(self.hf_repo)
156@dataclass(frozen=True)
157class CatalogResult:
158 """Paginated catalog result.
160 ``truncated`` is True when the HuggingFace scan stopped at its bound with
161 rows left unread, so matches past them are unreachable at any offset.
162 """
164 total: int | None
165 limit: int
166 offset: int
167 models: list[CatalogModel]
168 has_more: bool = False
169 truncated: bool = False
172@dataclass(frozen=True)
173class PageWindow:
174 """The part of a page left for the rows paged after the ones held locally."""
176 rest_offset: int
177 rest_limit: int
180def page_window(leading_count: int, offset: int, limit: int) -> PageWindow:
181 """The window left of ``[offset, offset + limit)`` after *leading_count* local rows."""
182 covered = min(offset + limit, leading_count) - min(offset, leading_count)
183 return PageWindow(rest_offset=max(0, offset - leading_count), rest_limit=limit - covered)
186@dataclass(frozen=True)
187class HfPage:
188 """One page of HuggingFace API results and the cursor of the page after it."""
190 models: list[CatalogModel]
191 next_cursor: str | None = None
193 @property
194 def has_more(self) -> bool:
195 """True when HuggingFace lists a next page."""
196 return self.next_cursor is not None
199def dedupe_models(models: list[CatalogModel]) -> list[CatalogModel]:
200 """Models in first-seen order, later repeats dropped."""
201 seen: set[str] = set()
202 unique: list[CatalogModel] = []
203 for model in models:
204 if model.hf_repo not in seen:
205 seen.add(model.hf_repo)
206 unique.append(model)
207 return unique
210@dataclass(frozen=True)
211class ModelVariant:
212 """One quantization within a model family. ``filename`` may be a glob."""
214 hf_repo: str
215 filename: str
216 param_count: str
217 quant: str
218 size_mb: int
219 mmproj_filename: str = ""
220 compat: ModelCompat = ModelCompat.UNKNOWN
221 safety_stripped: bool = False
224@dataclass(frozen=True)
225class ModelFamily:
226 """A group of related model variants (e.g. Qwen3 in multiple sizes)."""
228 slug: str # family slug for building refs: "qwen3"
229 name: str # display name: "Qwen3"
230 task: ModelTask
231 description: str
232 variants: tuple[ModelVariant, ...]