Coverage for src/lilbee/modelhub/registry.py: 100%

379 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""Manifest store keyed by ``(hf_repo, gguf_filename)`` over the HF cache. 

2 

3Canonical ref: ``<hf_repo>/<gguf_filename>``. Two quants of the same 

4repo are two distinct installations. Manifests live at 

5``manifests/<repo--repo>/<filename>.json``; blobs at 

6``models--<repo--repo>/blobs/<sha>``. 

7""" 

8 

9from __future__ import annotations 

10 

11import contextlib 

12import hashlib 

13import json 

14import logging 

15import os 

16import re 

17import shutil 

18import tempfile 

19from dataclasses import asdict, dataclass, field 

20from datetime import UTC, datetime 

21from pathlib import Path 

22from typing import TYPE_CHECKING 

23 

24from lilbee.catalog.query import reclassify_by_name 

25from lilbee.catalog.refs import ( 

26 GGUF_SUFFIX, 

27 NATIVE_GGUF_REF_MIN_SLASHES, 

28 format_native_gguf_ref, 

29 is_bare_hf_repo, 

30 split_shard_filenames, 

31) 

32from lilbee.catalog.types import ModelTask 

33from lilbee.core.config.model import cfg 

34from lilbee.core.security import validate_path_within 

35 

36if TYPE_CHECKING: 

37 from lilbee.catalog.models import CatalogModel 

38 

39log = logging.getLogger(__name__) 

40 

41_HASH_CHUNK_SIZE = 8192 # bytes read per iteration when hashing 

42_REPO_SEGMENT_RE = re.compile(r"^[a-zA-Z0-9._-]+/[a-zA-Z0-9._-]+$") 

43# A GGUF filename, optionally under repo subdirectories (unsloth stores quants 

44# in e.g. ``Q4_K_M/Model-...-00001-of-00003.gguf``). Path separators are allowed; 

45# ``..`` and absolute paths are rejected in the validator to stay inside the repo. 

46_FILENAME_RE = re.compile(r"^[a-zA-Z0-9._/-]+\.gguf$") 

47 

48REPO_DIR_SEPARATOR = "--" 

49_MANIFEST_SUFFIX = f"{GGUF_SUFFIX}.json" 

50 

51 

52def _validate_hf_repo(hf_repo: str) -> str: 

53 """Validate that a HuggingFace repo id has the form ``org/name``.""" 

54 if not hf_repo or not _REPO_SEGMENT_RE.match(hf_repo) or ".." in hf_repo: 

55 raise ValueError(f"Invalid hf_repo: {hf_repo!r}") 

56 return hf_repo 

57 

58 

59def _validate_gguf_filename(filename: str) -> str: 

60 """Validate a ``.gguf`` filename, allowing repo subdirectories but no traversal.""" 

61 if ( 

62 not filename 

63 or not _FILENAME_RE.match(filename) 

64 or ".." in filename 

65 or filename.startswith("/") 

66 ): 

67 raise ValueError(f"Invalid gguf_filename: {filename!r}") 

68 return filename 

69 

70 

71_REF_SHAPE_HINT = "Use '<org>/<repo>/<filename>.gguf'." 

72 

73 

74def parse_hf_ref(ref: str) -> tuple[str, str]: 

75 """Split ``<org>/<repo>/<file>.gguf`` into ``(hf_repo, gguf_filename)``. 

76 

77 The repo is always the first two segments (``<org>/<repo>``); everything 

78 after is the filename, which may include repo subdirectories (unsloth stores 

79 quants under e.g. ``Q4_K_M/Model-...-00001-of-00003.gguf``). 

80 """ 

81 if not ref.endswith(GGUF_SUFFIX) or ref.count("/") < NATIVE_GGUF_REF_MIN_SLASHES: 

82 raise ValueError(f"Model ref {ref!r} is not a HuggingFace ref. {_REF_SHAPE_HINT}") 

83 parts = ref.split("/") 

84 hf_repo = "/".join(parts[:NATIVE_GGUF_REF_MIN_SLASHES]) 

85 gguf_filename = "/".join(parts[NATIVE_GGUF_REF_MIN_SLASHES:]) 

86 return _validate_hf_repo(hf_repo), _validate_gguf_filename(gguf_filename) 

87 

88 

89def repo_to_dir(hf_repo: str) -> str: 

90 """Encode an HF repo for use as a directory name (HF cache convention).""" 

91 return hf_repo.replace("/", REPO_DIR_SEPARATOR) 

92 

93 

94@dataclass 

95class ModelManifest: 

96 """One installed model's metadata. Identity: ``(hf_repo, gguf_filename)``.""" 

97 

98 hf_repo: str 

99 gguf_filename: str 

100 size_bytes: int # primary (first-shard) blob size; validated against the blob on disk 

101 task: ModelTask 

102 downloaded_at: str # ISO 8601 

103 blob: str | None = None # SHA-256 hex of the blob in the HF cache; None pre-install 

104 # A split GGUF has further shard blobs beyond ``blob``. ``total_size_bytes`` 

105 # is the sum across every shard (None = single file, use ``size_bytes``); 

106 # ``shard_blobs`` are the non-primary shard digests, so removal frees them all. 

107 total_size_bytes: int | None = None 

108 shard_blobs: list[str] = field(default_factory=list) 

109 

110 @property 

111 def ref(self) -> str: 

112 return format_native_gguf_ref(self.hf_repo, self.gguf_filename) 

113 

114 @property 

115 def disk_size_bytes(self) -> int: 

116 """Total bytes this model occupies on disk, across all shards.""" 

117 return self.total_size_bytes if self.total_size_bytes is not None else self.size_bytes 

118 

119 

120def _copy_atomic(source_path: Path, blob_path: Path) -> None: 

121 """Copy *source_path* to *blob_path* via a temp file + atomic rename. 

122 

123 A crash mid-copy leaves only the temp file, never a partial blob at 

124 the final digest path that callers would treat as complete. 

125 """ 

126 fd, tmp_name = tempfile.mkstemp(dir=str(blob_path.parent), suffix=".part") 

127 tmp_path = Path(tmp_name) 

128 try: 

129 with os.fdopen(fd, "wb") as dst, source_path.open("rb") as src: 

130 shutil.copyfileobj(src, dst) 

131 os.replace(tmp_path, blob_path) 

132 except BaseException: 

133 tmp_path.unlink(missing_ok=True) 

134 raise 

135 

136 

137def _blob_size_matches(blob_file: Path, expected_size: int) -> bool: 

138 """True iff *blob_file* exists and its byte size equals *expected_size*. 

139 

140 A blob shorter than the manifest's recorded size is a truncated / 

141 interrupted download and must not count as installed. 

142 """ 

143 try: 

144 return blob_file.stat().st_size == expected_size 

145 except OSError: 

146 return False 

147 

148 

149def _sha256_file(path: Path) -> str: 

150 """Compute SHA-256 hex digest of a file.""" 

151 h = hashlib.sha256() 

152 with path.open("rb") as f: 

153 while True: 

154 chunk = f.read(_HASH_CHUNK_SIZE) 

155 if not chunk: 

156 break 

157 h.update(chunk) 

158 return h.hexdigest() 

159 

160 

161_SHA256_HEX = re.compile(r"[0-9a-f]{64}") 

162 

163 

164def _blob_digest(source_path: Path) -> str: 

165 """Digest for *source_path*, reusing the HF-cache blob name when possible. 

166 

167 huggingface_hub names cache blobs by their sha256, so a snapshot path that 

168 resolves into a ``blobs/`` dir already carries its digest. Re-hashing a 

169 multi-GB GGUF only to recompute that name is slow and, on network volumes, 

170 I/O-fragile enough to fail registration outright. Plain files still hash. 

171 """ 

172 real = source_path.resolve() 

173 if real.parent.name == "blobs" and _SHA256_HEX.fullmatch(real.name): 

174 return real.name 

175 return _sha256_file(source_path) 

176 

177 

178def _reraise_walk_error(error: OSError) -> None: 

179 """Re-raise a directory-read fault; a directory that vanished is not one.""" 

180 if not isinstance(error, FileNotFoundError): 

181 raise error 

182 

183 

184def _manifest_files(repo_dir: Path) -> list[Path]: 

185 """Sorted manifest paths under *repo_dir*, quant subdirectories included. 

186 

187 Raises ``OSError`` when a directory under *repo_dir* cannot be read, so an 

188 unreadable tree never reads as an empty one. 

189 """ 

190 found: list[Path] = [] 

191 for dirpath, _dirnames, filenames in os.walk(repo_dir, onerror=_reraise_walk_error): 

192 found.extend(Path(dirpath) / name for name in filenames if name.endswith(_MANIFEST_SUFFIX)) 

193 return sorted(found) 

194 

195 

196class ModelRegistry: 

197 """Read/write manifests and resolve refs to blobs in the HF cache.""" 

198 

199 def __init__(self, models_dir: Path) -> None: 

200 self._root = models_dir 

201 self._manifests_dir = models_dir / "manifests" 

202 

203 def _repo_cache_dir(self, hf_repo: str) -> Path: 

204 """The HuggingFace cache directory for *hf_repo* under this registry root.""" 

205 return self._root / f"models--{repo_to_dir(hf_repo)}" 

206 

207 def resolve(self, ref: str) -> Path: 

208 """Return the loadable GGUF path for *ref*; ``KeyError`` if not installed. 

209 

210 A single-file GGUF resolves to its content-hashed blob (the snapshot 

211 file on a cache without symlinks); a split GGUF to its first shard's 

212 snapshot symlink (so llama.cpp finds the sibling shards). 

213 

214 The canonical *ref* is ``<org>/<repo>/<file>.gguf`` resolved via the 

215 lilbee manifest. Two other shapes are accepted as a backwards-compat 

216 concession for builds already published (whose on-disk layout differs), 

217 not as the intended contract: a bare ``<org>/<repo>`` (older builds 

218 persisted these into ``config.toml``) resolves to the one quant of that 

219 repo that's installed, and a manifest that's missing / unparseable / 

220 blob-less falls back to whatever GGUF ``huggingface_hub`` reports the 

221 cache holds for that ref. The HF cache layout is stable, so this lets an 

222 upgrade keep working without anyone purging their lilbee data dir; it is 

223 deliberately the exception here, not a pattern to follow elsewhere. 

224 """ 

225 if is_bare_hf_repo(ref): 

226 return self._resolve_repo_only(_validate_hf_repo(ref)) 

227 hf_repo, gguf_filename = parse_hf_ref(ref) 

228 shards = split_shard_filenames(gguf_filename) 

229 if len(shards) > 1: 

230 return self._resolve_split(ref, hf_repo, shards) 

231 manifest = self._read_manifest(hf_repo, gguf_filename) 

232 if manifest is not None: 

233 backing = self._manifest_backing_file(manifest) 

234 if backing is not None: 

235 return backing 

236 recovered = self._find_cached_gguf(hf_repo, gguf_filename) 

237 if recovered is not None: 

238 self._reregister_from_cache(hf_repo, gguf_filename, recovered) 

239 return recovered 

240 if manifest is None: 

241 raise KeyError(f"Model {ref} not installed") 

242 # Manifest present but neither it nor the cache yields a blob; keep the 

243 # specific diagnostic so a corrupted cache stays debuggable. 

244 cache_path = self._repo_cache_dir(manifest.hf_repo) 

245 if not cache_path.exists(): 

246 raise KeyError(f"Cache folder missing for {ref}: {cache_path.name}") 

247 if manifest.blob is None: 

248 raise KeyError(f"Manifest for {ref} has no blob hash; install incomplete") 

249 blob_file = cache_path / "blobs" / manifest.blob 

250 if blob_file.exists(): 

251 raise KeyError( 

252 f"Blob for {ref} is truncated: {blob_file.stat().st_size} of " 

253 f"{manifest.size_bytes} bytes; re-download required" 

254 ) 

255 raise KeyError(f"Blob file missing for {ref}: {manifest.blob}") 

256 

257 def _resolve_split(self, ref: str, hf_repo: str, shards: list[str]) -> Path: 

258 """Resolve a split GGUF to its first shard's snapshot symlink. 

259 

260 llama.cpp loads the whole set from the first shard, locating the siblings 

261 by filename next to it. Only the snapshot dir co-locates the shards under 

262 their real names (the blobs dir names them by hash), so hand back the 

263 symlink, not the blob. Every shard must be present first: the first shard 

264 alone used to read as installed, registering an unloadable model that a 

265 re-pull then skipped. 

266 """ 

267 if not self._split_shards_present(hf_repo, shards[0]): 

268 raise KeyError(f"Split GGUF {ref} is missing shards; re-pull to fetch the full set") 

269 first_shard = self._snapshot_gguf_path(hf_repo, shards[0]) 

270 if first_shard is None: 

271 raise KeyError(f"Model {ref} not installed") 

272 if self._read_manifest(hf_repo, shards[0]) is None: 

273 # Same cache recovery as the single-file path; resolve the symlink so 

274 # the manifest records the content-hashed blob, not the link, and pass 

275 # the snapshot path so the shard accounting is recovered too. 

276 self._reregister_from_cache( 

277 hf_repo, shards[0], first_shard.resolve(), snapshot_path=first_shard 

278 ) 

279 return first_shard 

280 

281 def _resolve_repo_only(self, hf_repo: str) -> Path: 

282 """Resolve a bare ``<org>/<repo>`` ref to the GGUF of that repo on disk. 

283 

284 Older builds persisted bare repo refs for the chat / embedding model. 

285 Prefers a current-format manifest under the repo; otherwise asks 

286 ``huggingface_hub`` what GGUFs the cache holds for the repo and returns 

287 the first one (alphabetical for determinism if more than one quant is 

288 installed). 

289 """ 

290 manifest_dir = self._manifests_dir / repo_to_dir(hf_repo) 

291 if manifest_dir.is_dir(): 

292 for mf in _manifest_files(manifest_dir): 

293 manifest = self._load_manifest_file(mf) 

294 if manifest is None: 

295 continue 

296 backing = self._manifest_backing_file(manifest) 

297 if backing is not None: 

298 return backing 

299 for filename in sorted(self._cached_gguf_names(hf_repo)): 

300 shards = split_shard_filenames(filename) 

301 if len(shards) > 1: 

302 # A split set in the cache: skip its non-first shards and resolve 

303 # the whole set from shard 1 so we hand back the snapshot symlink 

304 # (siblings co-located, loadable) with shard accounting, not 

305 # shard 1's blob as an unloadable single file. 

306 if filename != shards[0]: 

307 continue 

308 with contextlib.suppress(KeyError): 

309 return self._resolve_split( 

310 format_native_gguf_ref(hf_repo, filename), hf_repo, shards 

311 ) 

312 continue 

313 recovered = self._find_cached_gguf(hf_repo, filename) 

314 if recovered is not None: 

315 self._reregister_from_cache(hf_repo, filename, recovered) 

316 return recovered 

317 raise KeyError(f"Model {hf_repo} not installed") 

318 

319 def _cached_gguf_names(self, hf_repo: str) -> set[str]: 

320 """``.gguf`` filenames the HuggingFace cache holds for *hf_repo*.""" 

321 if not self._root.is_dir(): 

322 return set() 

323 from huggingface_hub import scan_cache_dir 

324 

325 info = scan_cache_dir(self._root) 

326 return { 

327 f.file_name 

328 for repo in info.repos 

329 if repo.repo_id == hf_repo 

330 for rev in repo.revisions 

331 for f in rev.files 

332 if f.file_name.endswith(GGUF_SUFFIX) 

333 } 

334 

335 def _snapshot_gguf_path(self, hf_repo: str, gguf_filename: str) -> Path | None: 

336 """Return the snapshot *symlink* path for a cached GGUF, or None. 

337 

338 Returns the symlink, not the blob, so a split GGUF loads from a dir where 

339 its sibling shards are co-located under their real names. 

340 """ 

341 from huggingface_hub import try_to_load_from_cache 

342 

343 hit = try_to_load_from_cache( 

344 repo_id=hf_repo, filename=gguf_filename, cache_dir=str(self._root) 

345 ) 

346 candidate: Path | None = None 

347 if isinstance(hit, str): # exact repo-relative match 

348 candidate = Path(hit) 

349 else: # None or the _CACHED_NO_EXIST sentinel: locate the basename instead 

350 snapshots = self._repo_cache_dir(hf_repo) / "snapshots" 

351 if snapshots.is_dir(): 

352 basename = Path(gguf_filename).name 

353 # Several cached revisions can hold the basename; prefer the most 

354 # recently materialized one over an arbitrary lexicographic pick. 

355 candidate = max( 

356 snapshots.rglob(basename), key=lambda p: p.lstat().st_mtime, default=None 

357 ) 

358 if candidate is None: 

359 return None 

360 try: 

361 validate_path_within(candidate.resolve(), self._root) 

362 except ValueError: 

363 return None 

364 return candidate 

365 

366 def _find_cached_gguf(self, hf_repo: str, gguf_filename: str) -> Path | None: 

367 """Return the cached blob path for ``hf_repo``/``gguf_filename``, or None. 

368 

369 Locates the snapshot symlink (subdir-aware) and resolves it to its blob, 

370 bounded to the cache directory. 

371 """ 

372 symlink = self._snapshot_gguf_path(hf_repo, gguf_filename) 

373 return symlink.resolve() if symlink is not None else None 

374 

375 def _split_shards_present(self, hf_repo: str, gguf_filename: str) -> bool: 

376 """True unless *gguf_filename* is a split GGUF missing one of its shards. 

377 

378 A single-file GGUF is always present here. For a split set 

379 (``<base>-0000N-of-0000M.gguf``) every shard must be cached, since 

380 llama.cpp loads the whole set from the first shard but needs them all. 

381 """ 

382 shards = split_shard_filenames(gguf_filename) 

383 if len(shards) == 1: 

384 return True 

385 return all(self._find_cached_gguf(hf_repo, shard) is not None for shard in shards) 

386 

387 def shard_paths(self, ref: str) -> list[Path]: 

388 """On-disk paths of *ref*'s GGUF shards that exist next to its resolved path. 

389 

390 A split GGUF resolves to its first shard's snapshot symlink with the 

391 siblings co-located, so every shard is returned; a single-file GGUF 

392 resolves to its content-hashed blob, where no sibling exists under the 

393 real filename. Raises ``KeyError`` / ``ValueError`` like :meth:`resolve`. 

394 """ 

395 first = self.resolve(ref) 

396 _repo, filename = parse_hf_ref(ref) 

397 candidates = ( 

398 first.parent / Path(shard).name for shard in split_shard_filenames(Path(filename).name) 

399 ) 

400 return [path for path in candidates if path.exists()] 

401 

402 def _reregister_from_cache( 

403 self, 

404 hf_repo: str, 

405 gguf_filename: str, 

406 blob_path: Path, 

407 snapshot_path: Path | None = None, 

408 ) -> None: 

409 """Best-effort manifest write for a cache-recovered model so listings see it. 

410 

411 *snapshot_path* is the first shard's snapshot path (siblings co-located); 

412 when given, the split-shard accounting is recovered too, so a cache-only 

413 split GGUF still frees every shard and reports its full size. 

414 """ 

415 ref = format_native_gguf_ref(hf_repo, gguf_filename) 

416 try: 

417 task = ModelTask(reclassify_by_name(ref, ModelTask.CHAT)) 

418 total_size, shard_blobs = ( 

419 _shard_accounting(snapshot_path) if snapshot_path is not None else (None, []) 

420 ) 

421 self._write_manifest( 

422 ModelManifest( 

423 hf_repo=hf_repo, 

424 gguf_filename=gguf_filename, 

425 size_bytes=blob_path.stat().st_size, 

426 task=task, 

427 downloaded_at=datetime.now(UTC).isoformat(), 

428 # Only a file under blobs/ is named by its sha; a symlink-less 

429 # snapshot file is not, so no digest is recorded for it. 

430 blob=blob_path.name if blob_path.parent.name == "blobs" else None, 

431 total_size_bytes=total_size, 

432 shard_blobs=shard_blobs, 

433 ) 

434 ) 

435 log.info("Recovered manifest for %s from the model cache", ref) 

436 except Exception: # cache-warming write; the resolve already returned a path 

437 log.debug("Could not re-register %s from the model cache", ref, exc_info=True) 

438 

439 def is_installed(self, ref: str) -> bool: 

440 """Return True if a model is installed and its blob is present.""" 

441 try: 

442 self.resolve(ref) 

443 return True 

444 except (KeyError, ValueError): 

445 return False 

446 

447 def install( 

448 self, 

449 hf_repo: str, 

450 gguf_filename: str, 

451 source_path: Path, 

452 manifest: ModelManifest, 

453 ) -> Path: 

454 """Write a manifest, copying *source_path* into the HF cache if needed.""" 

455 digest = _blob_digest(source_path) 

456 cache_path = self._repo_cache_dir(hf_repo) 

457 blobs_dir = cache_path / "blobs" 

458 blob_path = blobs_dir / digest 

459 if not blob_path.exists(): 

460 blobs_dir.mkdir(parents=True, exist_ok=True) 

461 _copy_atomic(source_path, blob_path) 

462 

463 updated = ModelManifest( 

464 hf_repo=hf_repo, 

465 gguf_filename=gguf_filename, 

466 # Record the size install actually wrote, not the caller's claim, 

467 # so the on-disk size check has a trustworthy reference. 

468 size_bytes=source_path.stat().st_size, 

469 task=manifest.task, 

470 downloaded_at=manifest.downloaded_at, 

471 blob=digest, 

472 # Carry the split-shard accounting through unchanged (computed by the 

473 # caller from the full shard set); install only rewrites the primary. 

474 total_size_bytes=manifest.total_size_bytes, 

475 shard_blobs=manifest.shard_blobs, 

476 ) 

477 self._write_manifest(updated) 

478 return blob_path 

479 

480 def remove(self, ref: str) -> bool: 

481 """Remove a manifest and its backing blob. 

482 

483 The blob is shared via SHA-256 digest, so it only goes away 

484 when no other installed manifest references the same digest. 

485 Empty cache directories (``blobs/``, the per-repo ``models--`` 

486 folder, and the per-repo manifest folder) are pruned so a 

487 deleted model leaves no orphan bytes behind. 

488 """ 

489 try: 

490 hf_repo, gguf_filename = parse_hf_ref(ref) 

491 except ValueError: 

492 return False 

493 manifest = self._read_manifest(hf_repo, gguf_filename) 

494 if manifest is None: 

495 return False 

496 # Manifests written before shard accounting existed have no shard_blobs, so 

497 # recover them from the cache *before* unlinking (resolve needs the manifest). 

498 shard_blobs = manifest.shard_blobs or self._recover_legacy_shard_blobs(ref) 

499 manifest_path = self._manifest_path(hf_repo, gguf_filename) 

500 manifest_path.unlink() 

501 repo_dir = manifest_path.parent 

502 if repo_dir.exists() and not any(repo_dir.iterdir()): 

503 repo_dir.rmdir() 

504 self._unlink_snapshot_entries(manifest) 

505 # Free the primary blob and every extra shard blob; a split GGUF has more 

506 # than one, and leaving the others orphans them when a sibling quant keeps 

507 # the repo cache dir alive. The surviving manifests for this repo are read 

508 # once here rather than per digest (list_installed walks the whole tree). 

509 # A symlink-less recovery records no digest; GC still runs once so an 

510 # unused repo cache dir is pruned. 

511 siblings = [m for m in self.list_installed() if m.hf_repo == manifest.hf_repo] 

512 digests: list[str | None] = [d for d in [manifest.blob, *shard_blobs] if d is not None] 

513 for digest in digests or [None]: 

514 self._gc_blob(manifest.hf_repo, digest, siblings=siblings) 

515 log.info("Removed model %s", ref) 

516 return True 

517 

518 def _unlink_snapshot_entries(self, manifest: ModelManifest) -> None: 

519 """Drop the snapshot entries for *manifest*'s shards. 

520 

521 On a symlink-less cache they hold the bytes and would resurrect the 

522 model via cache recovery; on a symlinked cache they would dangle once 

523 the blob is gc'd. 

524 """ 

525 for shard in split_shard_filenames(manifest.gguf_filename): 

526 snapshot_entry = self._snapshot_gguf_path(manifest.hf_repo, shard) 

527 if snapshot_entry is not None: 

528 snapshot_entry.unlink(missing_ok=True) 

529 

530 def _recover_legacy_shard_blobs(self, ref: str) -> list[str]: 

531 """Extra shard blob digests for a pre-accounting split GGUF, best-effort. 

532 

533 Older manifests recorded only the first shard, so removing them would 

534 orphan the rest. Derive the sibling shards from the cache; empty on any 

535 failure or for a single-file model, so removal never breaks. 

536 """ 

537 with contextlib.suppress(Exception): 

538 shards = self.shard_paths(ref) 

539 return [_blob_digest(path) for path in shards[1:]] 

540 return [] 

541 

542 def _gc_blob( 

543 self, hf_repo: str, digest: str | None, *, siblings: list[ModelManifest] | None = None 

544 ) -> None: 

545 """Drop blob bytes and HuggingFace cache cruft now that *digest* 

546 and possibly the whole repo are unused. A None *digest* (a manifest 

547 recovered on a symlink-less cache records no digest) only prunes the 

548 repo dir when no installed manifest is left. 

549 

550 When the per-repo ``models--<repo>/`` directory has no installed 

551 manifests left, the whole directory is wiped so HF's ``refs/``, 

552 ``snapshots/``, and stale ``blobs/`` all go with it. Otherwise 

553 only the specific blob file is removed when no remaining 

554 manifest still references its digest. 

555 

556 ``siblings`` is the surviving-manifest list for *hf_repo*; callers 

557 freeing several blobs at once pass it in so the manifest tree is walked 

558 once instead of per digest. Defaults to reading it when omitted. 

559 """ 

560 cache_path = self._repo_cache_dir(hf_repo) 

561 try: 

562 validate_path_within(cache_path, self._root) 

563 except ValueError: 

564 log.warning("Refusing to remove cache outside models_dir: %s", cache_path) 

565 return 

566 if siblings is None: 

567 siblings = [m for m in self.list_installed() if m.hf_repo == hf_repo] 

568 if not siblings: 

569 if cache_path.exists(): 

570 shutil.rmtree(cache_path) 

571 return 

572 if digest is None: # no digest recorded (symlink-less recovery): dir-prune only 

573 return 

574 if any(digest == m.blob or digest in m.shard_blobs for m in siblings): 

575 return 

576 blob_file = cache_path / "blobs" / digest 

577 try: 

578 validate_path_within(blob_file, self._root) 

579 except ValueError: 

580 log.warning("Refusing to remove blob outside models_dir: %s", blob_file) 

581 return 

582 if blob_file.exists(): 

583 blob_file.unlink() 

584 

585 def list_installed(self) -> list[ModelManifest]: 

586 """Return manifests for models whose blob is fully present on disk. 

587 

588 A manifest with a null blob field or a missing blob file is the 

589 residue of a canceled or partial download. Surfacing it would 

590 let the picker offer an unusable selection, so the read filter 

591 lives here at the source instead of in every UI caller. 

592 

593 An absent manifest tree is an empty list. A tree that is present but 

594 cannot be read raises ``OSError``, because "unknown" is not "none". 

595 """ 

596 manifests: list[ModelManifest] = [] 

597 try: 

598 repo_dirs = sorted(self._manifests_dir.iterdir()) 

599 except FileNotFoundError: 

600 return manifests 

601 for repo_dir in repo_dirs: 

602 if not repo_dir.is_dir(): 

603 continue 

604 for tag_file in _manifest_files(repo_dir): 

605 manifest = self._load_manifest_file(tag_file) 

606 if manifest is not None and self._blob_present(manifest): 

607 manifests.append(manifest) 

608 return manifests 

609 

610 def _blob_present(self, manifest: ModelManifest) -> bool: 

611 """True iff *manifest*'s bytes are on disk with a matching size.""" 

612 return self._manifest_backing_file(manifest) is not None 

613 

614 def _manifest_backing_file(self, manifest: ModelManifest) -> Path | None: 

615 """The on-disk file holding *manifest*'s bytes, or None if absent/truncated. 

616 

617 Normally the content-hashed blob; on a cache without symlinks (the 

618 Windows default) the bytes live at the snapshot path, so that is the 

619 fallback. ``resolve`` and ``list_installed`` must both gate on this 

620 one predicate, or installed-ness disagrees between ``pull`` and ``ask``. 

621 """ 

622 if manifest.blob is not None: 

623 blob_file = self._repo_cache_dir(manifest.hf_repo) / "blobs" / manifest.blob 

624 if _blob_size_matches(blob_file, manifest.size_bytes): 

625 return blob_file 

626 recovered = self._find_cached_gguf(manifest.hf_repo, manifest.gguf_filename) 

627 if recovered is not None and _blob_size_matches(recovered, manifest.size_bytes): 

628 return recovered 

629 return None 

630 

631 def get_manifest(self, ref: str) -> ModelManifest | None: 

632 """Return the manifest for *ref* or None if not installed.""" 

633 try: 

634 hf_repo, gguf_filename = parse_hf_ref(ref) 

635 except ValueError: 

636 return None 

637 return self._read_manifest(hf_repo, gguf_filename) 

638 

639 def installed_ref_for_repo(self, hf_repo: str) -> str | None: 

640 """Full ``<repo>/<file>.gguf`` ref of an installed quant of *hf_repo*, or None. 

641 

642 Alphabetical-first when several quants are installed, matching 

643 ``_resolve_repo_only``'s determinism. 

644 """ 

645 refs = sorted(m.ref for m in self.list_installed() if m.hf_repo == hf_repo) 

646 return refs[0] if refs else None 

647 

648 def _manifest_path(self, hf_repo: str, gguf_filename: str) -> Path: 

649 repo = _validate_hf_repo(hf_repo) 

650 filename = _validate_gguf_filename(gguf_filename) 

651 path = self._manifests_dir / repo_to_dir(repo) / f"{filename}.json" 

652 validate_path_within(path, self._manifests_dir) 

653 return path 

654 

655 def _read_manifest(self, hf_repo: str, gguf_filename: str) -> ModelManifest | None: 

656 return self._load_manifest_file(self._manifest_path(hf_repo, gguf_filename)) 

657 

658 def _write_manifest(self, manifest: ModelManifest) -> None: 

659 path = self._manifest_path(manifest.hf_repo, manifest.gguf_filename) 

660 path.parent.mkdir(parents=True, exist_ok=True) 

661 data = json.dumps(asdict(manifest), indent=2) 

662 tmp_path: str | None = None 

663 try: 

664 with tempfile.NamedTemporaryFile( 

665 dir=path.parent, suffix=".tmp", mode="w", encoding="utf-8", delete=False 

666 ) as tmp: 

667 tmp_path = tmp.name 

668 tmp.write(data) 

669 os.replace(tmp_path, path) 

670 except BaseException: 

671 if tmp_path is not None: 

672 Path(tmp_path).unlink(missing_ok=True) 

673 raise 

674 

675 def _load_manifest_file(self, path: Path) -> ModelManifest | None: 

676 if not path.exists(): 

677 return None 

678 try: 

679 data = json.loads(path.read_text(encoding="utf-8")) 

680 return ModelManifest(**data) 

681 # UnicodeDecodeError is a ValueError, not a JSONDecodeError. 

682 except (json.JSONDecodeError, UnicodeDecodeError, TypeError, KeyError): 

683 log.warning("Corrupt manifest: %s", path) 

684 return None 

685 

686 

687_HF_SNAPSHOTS_DIR = "snapshots" 

688 

689 

690def _repo_relative_gguf_name(file_path: Path) -> str: 

691 """Recover the repo-relative GGUF filename, keeping any subdir prefix. 

692 

693 HF caches a file at ``models--<repo>/snapshots/<rev>/[<subdir>/]<name>``. A 

694 subdir-quant giant (unsloth stores quants under e.g. ``Q4_K_M/``) must 

695 register under that subdir-relative name so its manifest key round-trips with 

696 the ref; ``file_path.name`` alone would drop the subdir. Falls back to the 

697 basename when the path is not under a snapshot revision dir. 

698 """ 

699 parts = file_path.parts 

700 if _HF_SNAPSHOTS_DIR not in parts: 

701 return file_path.name 

702 rev_index = parts.index(_HF_SNAPSHOTS_DIR) + 1 

703 relative_parts = parts[rev_index + 1 :] 

704 return "/".join(relative_parts) if relative_parts else file_path.name 

705 

706 

707def _shard_accounting(first_shard_path: Path) -> tuple[int | None, list[str]]: 

708 """Total on-disk size and non-primary shard blob digests for a split GGUF. 

709 

710 ``(None, [])`` for a single-file model. For a split GGUF, the sibling shards 

711 live next to the first shard; ``_blob_digest`` yields each shard's blob digest 

712 (the HF cache blob name, or a content hash in copy/non-symlink mode), so the 

713 shards are summed and the digests of shards 2..N collected for removal-time 

714 garbage collection. 

715 """ 

716 shard_names = split_shard_filenames(first_shard_path.name) 

717 if len(shard_names) <= 1: 

718 return None, [] 

719 total = 0 

720 shard_blobs: list[str] = [] 

721 for index, name in enumerate(shard_names): 

722 shard_path = first_shard_path.with_name(name) 

723 if not shard_path.exists(): 

724 continue 

725 total += shard_path.stat().st_size 

726 if index > 0: # the primary blob is tracked separately as manifest.blob 

727 shard_blobs.append(_blob_digest(shard_path)) 

728 return total, shard_blobs 

729 

730 

731def register_downloaded_model(entry: CatalogModel, file_path: Path) -> None: 

732 """Write a registry manifest for a freshly downloaded GGUF. 

733 

734 A failed manifest write is logged, not raised, when the GGUF is still in the 

735 HF cache (``resolve`` recovers from it); if it isn't, the download itself is 

736 broken and the failure propagates so the caller reports it. 

737 """ 

738 registry = ModelRegistry(cfg.models_dir) 

739 gguf_filename = _repo_relative_gguf_name(file_path) 

740 total_size, shard_blobs = _shard_accounting(file_path) 

741 manifest = ModelManifest( 

742 hf_repo=entry.hf_repo, 

743 gguf_filename=gguf_filename, 

744 size_bytes=file_path.stat().st_size, 

745 task=entry.task, 

746 downloaded_at=datetime.now(UTC).isoformat(), 

747 total_size_bytes=total_size, 

748 shard_blobs=shard_blobs, 

749 ) 

750 try: 

751 registry.install(entry.hf_repo, gguf_filename, file_path, manifest) 

752 log.info("Registered %s/%s in manifest", entry.hf_repo, gguf_filename) 

753 except Exception: 

754 ref = format_native_gguf_ref(entry.hf_repo, gguf_filename) 

755 if not registry.is_installed(ref): 

756 raise 

757 log.warning( 

758 "Manifest write failed for %s; recovered via the model cache", ref, exc_info=True 

759 )