Coverage for src/lilbee/modelhub/registry.py: 100%
379 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
1"""Manifest store keyed by ``(hf_repo, gguf_filename)`` over the HF cache.
3Canonical ref: ``<hf_repo>/<gguf_filename>``. Two quants of the same
4repo are two distinct installations. Manifests live at
5``manifests/<repo--repo>/<filename>.json``; blobs at
6``models--<repo--repo>/blobs/<sha>``.
7"""
9from __future__ import annotations
11import contextlib
12import hashlib
13import json
14import logging
15import os
16import re
17import shutil
18import tempfile
19from dataclasses import asdict, dataclass, field
20from datetime import UTC, datetime
21from pathlib import Path
22from typing import TYPE_CHECKING
24from lilbee.catalog.query import reclassify_by_name
25from lilbee.catalog.refs import (
26 GGUF_SUFFIX,
27 NATIVE_GGUF_REF_MIN_SLASHES,
28 format_native_gguf_ref,
29 is_bare_hf_repo,
30 split_shard_filenames,
31)
32from lilbee.catalog.types import ModelTask
33from lilbee.core.config.model import cfg
34from lilbee.core.security import validate_path_within
36if TYPE_CHECKING:
37 from lilbee.catalog.models import CatalogModel
39log = logging.getLogger(__name__)
41_HASH_CHUNK_SIZE = 8192 # bytes read per iteration when hashing
42_REPO_SEGMENT_RE = re.compile(r"^[a-zA-Z0-9._-]+/[a-zA-Z0-9._-]+$")
43# A GGUF filename, optionally under repo subdirectories (unsloth stores quants
44# in e.g. ``Q4_K_M/Model-...-00001-of-00003.gguf``). Path separators are allowed;
45# ``..`` and absolute paths are rejected in the validator to stay inside the repo.
46_FILENAME_RE = re.compile(r"^[a-zA-Z0-9._/-]+\.gguf$")
48REPO_DIR_SEPARATOR = "--"
49_MANIFEST_SUFFIX = f"{GGUF_SUFFIX}.json"
52def _validate_hf_repo(hf_repo: str) -> str:
53 """Validate that a HuggingFace repo id has the form ``org/name``."""
54 if not hf_repo or not _REPO_SEGMENT_RE.match(hf_repo) or ".." in hf_repo:
55 raise ValueError(f"Invalid hf_repo: {hf_repo!r}")
56 return hf_repo
59def _validate_gguf_filename(filename: str) -> str:
60 """Validate a ``.gguf`` filename, allowing repo subdirectories but no traversal."""
61 if (
62 not filename
63 or not _FILENAME_RE.match(filename)
64 or ".." in filename
65 or filename.startswith("/")
66 ):
67 raise ValueError(f"Invalid gguf_filename: {filename!r}")
68 return filename
71_REF_SHAPE_HINT = "Use '<org>/<repo>/<filename>.gguf'."
74def parse_hf_ref(ref: str) -> tuple[str, str]:
75 """Split ``<org>/<repo>/<file>.gguf`` into ``(hf_repo, gguf_filename)``.
77 The repo is always the first two segments (``<org>/<repo>``); everything
78 after is the filename, which may include repo subdirectories (unsloth stores
79 quants under e.g. ``Q4_K_M/Model-...-00001-of-00003.gguf``).
80 """
81 if not ref.endswith(GGUF_SUFFIX) or ref.count("/") < NATIVE_GGUF_REF_MIN_SLASHES:
82 raise ValueError(f"Model ref {ref!r} is not a HuggingFace ref. {_REF_SHAPE_HINT}")
83 parts = ref.split("/")
84 hf_repo = "/".join(parts[:NATIVE_GGUF_REF_MIN_SLASHES])
85 gguf_filename = "/".join(parts[NATIVE_GGUF_REF_MIN_SLASHES:])
86 return _validate_hf_repo(hf_repo), _validate_gguf_filename(gguf_filename)
89def repo_to_dir(hf_repo: str) -> str:
90 """Encode an HF repo for use as a directory name (HF cache convention)."""
91 return hf_repo.replace("/", REPO_DIR_SEPARATOR)
94@dataclass
95class ModelManifest:
96 """One installed model's metadata. Identity: ``(hf_repo, gguf_filename)``."""
98 hf_repo: str
99 gguf_filename: str
100 size_bytes: int # primary (first-shard) blob size; validated against the blob on disk
101 task: ModelTask
102 downloaded_at: str # ISO 8601
103 blob: str | None = None # SHA-256 hex of the blob in the HF cache; None pre-install
104 # A split GGUF has further shard blobs beyond ``blob``. ``total_size_bytes``
105 # is the sum across every shard (None = single file, use ``size_bytes``);
106 # ``shard_blobs`` are the non-primary shard digests, so removal frees them all.
107 total_size_bytes: int | None = None
108 shard_blobs: list[str] = field(default_factory=list)
110 @property
111 def ref(self) -> str:
112 return format_native_gguf_ref(self.hf_repo, self.gguf_filename)
114 @property
115 def disk_size_bytes(self) -> int:
116 """Total bytes this model occupies on disk, across all shards."""
117 return self.total_size_bytes if self.total_size_bytes is not None else self.size_bytes
120def _copy_atomic(source_path: Path, blob_path: Path) -> None:
121 """Copy *source_path* to *blob_path* via a temp file + atomic rename.
123 A crash mid-copy leaves only the temp file, never a partial blob at
124 the final digest path that callers would treat as complete.
125 """
126 fd, tmp_name = tempfile.mkstemp(dir=str(blob_path.parent), suffix=".part")
127 tmp_path = Path(tmp_name)
128 try:
129 with os.fdopen(fd, "wb") as dst, source_path.open("rb") as src:
130 shutil.copyfileobj(src, dst)
131 os.replace(tmp_path, blob_path)
132 except BaseException:
133 tmp_path.unlink(missing_ok=True)
134 raise
137def _blob_size_matches(blob_file: Path, expected_size: int) -> bool:
138 """True iff *blob_file* exists and its byte size equals *expected_size*.
140 A blob shorter than the manifest's recorded size is a truncated /
141 interrupted download and must not count as installed.
142 """
143 try:
144 return blob_file.stat().st_size == expected_size
145 except OSError:
146 return False
149def _sha256_file(path: Path) -> str:
150 """Compute SHA-256 hex digest of a file."""
151 h = hashlib.sha256()
152 with path.open("rb") as f:
153 while True:
154 chunk = f.read(_HASH_CHUNK_SIZE)
155 if not chunk:
156 break
157 h.update(chunk)
158 return h.hexdigest()
161_SHA256_HEX = re.compile(r"[0-9a-f]{64}")
164def _blob_digest(source_path: Path) -> str:
165 """Digest for *source_path*, reusing the HF-cache blob name when possible.
167 huggingface_hub names cache blobs by their sha256, so a snapshot path that
168 resolves into a ``blobs/`` dir already carries its digest. Re-hashing a
169 multi-GB GGUF only to recompute that name is slow and, on network volumes,
170 I/O-fragile enough to fail registration outright. Plain files still hash.
171 """
172 real = source_path.resolve()
173 if real.parent.name == "blobs" and _SHA256_HEX.fullmatch(real.name):
174 return real.name
175 return _sha256_file(source_path)
178def _reraise_walk_error(error: OSError) -> None:
179 """Re-raise a directory-read fault; a directory that vanished is not one."""
180 if not isinstance(error, FileNotFoundError):
181 raise error
184def _manifest_files(repo_dir: Path) -> list[Path]:
185 """Sorted manifest paths under *repo_dir*, quant subdirectories included.
187 Raises ``OSError`` when a directory under *repo_dir* cannot be read, so an
188 unreadable tree never reads as an empty one.
189 """
190 found: list[Path] = []
191 for dirpath, _dirnames, filenames in os.walk(repo_dir, onerror=_reraise_walk_error):
192 found.extend(Path(dirpath) / name for name in filenames if name.endswith(_MANIFEST_SUFFIX))
193 return sorted(found)
196class ModelRegistry:
197 """Read/write manifests and resolve refs to blobs in the HF cache."""
199 def __init__(self, models_dir: Path) -> None:
200 self._root = models_dir
201 self._manifests_dir = models_dir / "manifests"
203 def _repo_cache_dir(self, hf_repo: str) -> Path:
204 """The HuggingFace cache directory for *hf_repo* under this registry root."""
205 return self._root / f"models--{repo_to_dir(hf_repo)}"
207 def resolve(self, ref: str) -> Path:
208 """Return the loadable GGUF path for *ref*; ``KeyError`` if not installed.
210 A single-file GGUF resolves to its content-hashed blob (the snapshot
211 file on a cache without symlinks); a split GGUF to its first shard's
212 snapshot symlink (so llama.cpp finds the sibling shards).
214 The canonical *ref* is ``<org>/<repo>/<file>.gguf`` resolved via the
215 lilbee manifest. Two other shapes are accepted as a backwards-compat
216 concession for builds already published (whose on-disk layout differs),
217 not as the intended contract: a bare ``<org>/<repo>`` (older builds
218 persisted these into ``config.toml``) resolves to the one quant of that
219 repo that's installed, and a manifest that's missing / unparseable /
220 blob-less falls back to whatever GGUF ``huggingface_hub`` reports the
221 cache holds for that ref. The HF cache layout is stable, so this lets an
222 upgrade keep working without anyone purging their lilbee data dir; it is
223 deliberately the exception here, not a pattern to follow elsewhere.
224 """
225 if is_bare_hf_repo(ref):
226 return self._resolve_repo_only(_validate_hf_repo(ref))
227 hf_repo, gguf_filename = parse_hf_ref(ref)
228 shards = split_shard_filenames(gguf_filename)
229 if len(shards) > 1:
230 return self._resolve_split(ref, hf_repo, shards)
231 manifest = self._read_manifest(hf_repo, gguf_filename)
232 if manifest is not None:
233 backing = self._manifest_backing_file(manifest)
234 if backing is not None:
235 return backing
236 recovered = self._find_cached_gguf(hf_repo, gguf_filename)
237 if recovered is not None:
238 self._reregister_from_cache(hf_repo, gguf_filename, recovered)
239 return recovered
240 if manifest is None:
241 raise KeyError(f"Model {ref} not installed")
242 # Manifest present but neither it nor the cache yields a blob; keep the
243 # specific diagnostic so a corrupted cache stays debuggable.
244 cache_path = self._repo_cache_dir(manifest.hf_repo)
245 if not cache_path.exists():
246 raise KeyError(f"Cache folder missing for {ref}: {cache_path.name}")
247 if manifest.blob is None:
248 raise KeyError(f"Manifest for {ref} has no blob hash; install incomplete")
249 blob_file = cache_path / "blobs" / manifest.blob
250 if blob_file.exists():
251 raise KeyError(
252 f"Blob for {ref} is truncated: {blob_file.stat().st_size} of "
253 f"{manifest.size_bytes} bytes; re-download required"
254 )
255 raise KeyError(f"Blob file missing for {ref}: {manifest.blob}")
257 def _resolve_split(self, ref: str, hf_repo: str, shards: list[str]) -> Path:
258 """Resolve a split GGUF to its first shard's snapshot symlink.
260 llama.cpp loads the whole set from the first shard, locating the siblings
261 by filename next to it. Only the snapshot dir co-locates the shards under
262 their real names (the blobs dir names them by hash), so hand back the
263 symlink, not the blob. Every shard must be present first: the first shard
264 alone used to read as installed, registering an unloadable model that a
265 re-pull then skipped.
266 """
267 if not self._split_shards_present(hf_repo, shards[0]):
268 raise KeyError(f"Split GGUF {ref} is missing shards; re-pull to fetch the full set")
269 first_shard = self._snapshot_gguf_path(hf_repo, shards[0])
270 if first_shard is None:
271 raise KeyError(f"Model {ref} not installed")
272 if self._read_manifest(hf_repo, shards[0]) is None:
273 # Same cache recovery as the single-file path; resolve the symlink so
274 # the manifest records the content-hashed blob, not the link, and pass
275 # the snapshot path so the shard accounting is recovered too.
276 self._reregister_from_cache(
277 hf_repo, shards[0], first_shard.resolve(), snapshot_path=first_shard
278 )
279 return first_shard
281 def _resolve_repo_only(self, hf_repo: str) -> Path:
282 """Resolve a bare ``<org>/<repo>`` ref to the GGUF of that repo on disk.
284 Older builds persisted bare repo refs for the chat / embedding model.
285 Prefers a current-format manifest under the repo; otherwise asks
286 ``huggingface_hub`` what GGUFs the cache holds for the repo and returns
287 the first one (alphabetical for determinism if more than one quant is
288 installed).
289 """
290 manifest_dir = self._manifests_dir / repo_to_dir(hf_repo)
291 if manifest_dir.is_dir():
292 for mf in _manifest_files(manifest_dir):
293 manifest = self._load_manifest_file(mf)
294 if manifest is None:
295 continue
296 backing = self._manifest_backing_file(manifest)
297 if backing is not None:
298 return backing
299 for filename in sorted(self._cached_gguf_names(hf_repo)):
300 shards = split_shard_filenames(filename)
301 if len(shards) > 1:
302 # A split set in the cache: skip its non-first shards and resolve
303 # the whole set from shard 1 so we hand back the snapshot symlink
304 # (siblings co-located, loadable) with shard accounting, not
305 # shard 1's blob as an unloadable single file.
306 if filename != shards[0]:
307 continue
308 with contextlib.suppress(KeyError):
309 return self._resolve_split(
310 format_native_gguf_ref(hf_repo, filename), hf_repo, shards
311 )
312 continue
313 recovered = self._find_cached_gguf(hf_repo, filename)
314 if recovered is not None:
315 self._reregister_from_cache(hf_repo, filename, recovered)
316 return recovered
317 raise KeyError(f"Model {hf_repo} not installed")
319 def _cached_gguf_names(self, hf_repo: str) -> set[str]:
320 """``.gguf`` filenames the HuggingFace cache holds for *hf_repo*."""
321 if not self._root.is_dir():
322 return set()
323 from huggingface_hub import scan_cache_dir
325 info = scan_cache_dir(self._root)
326 return {
327 f.file_name
328 for repo in info.repos
329 if repo.repo_id == hf_repo
330 for rev in repo.revisions
331 for f in rev.files
332 if f.file_name.endswith(GGUF_SUFFIX)
333 }
335 def _snapshot_gguf_path(self, hf_repo: str, gguf_filename: str) -> Path | None:
336 """Return the snapshot *symlink* path for a cached GGUF, or None.
338 Returns the symlink, not the blob, so a split GGUF loads from a dir where
339 its sibling shards are co-located under their real names.
340 """
341 from huggingface_hub import try_to_load_from_cache
343 hit = try_to_load_from_cache(
344 repo_id=hf_repo, filename=gguf_filename, cache_dir=str(self._root)
345 )
346 candidate: Path | None = None
347 if isinstance(hit, str): # exact repo-relative match
348 candidate = Path(hit)
349 else: # None or the _CACHED_NO_EXIST sentinel: locate the basename instead
350 snapshots = self._repo_cache_dir(hf_repo) / "snapshots"
351 if snapshots.is_dir():
352 basename = Path(gguf_filename).name
353 # Several cached revisions can hold the basename; prefer the most
354 # recently materialized one over an arbitrary lexicographic pick.
355 candidate = max(
356 snapshots.rglob(basename), key=lambda p: p.lstat().st_mtime, default=None
357 )
358 if candidate is None:
359 return None
360 try:
361 validate_path_within(candidate.resolve(), self._root)
362 except ValueError:
363 return None
364 return candidate
366 def _find_cached_gguf(self, hf_repo: str, gguf_filename: str) -> Path | None:
367 """Return the cached blob path for ``hf_repo``/``gguf_filename``, or None.
369 Locates the snapshot symlink (subdir-aware) and resolves it to its blob,
370 bounded to the cache directory.
371 """
372 symlink = self._snapshot_gguf_path(hf_repo, gguf_filename)
373 return symlink.resolve() if symlink is not None else None
375 def _split_shards_present(self, hf_repo: str, gguf_filename: str) -> bool:
376 """True unless *gguf_filename* is a split GGUF missing one of its shards.
378 A single-file GGUF is always present here. For a split set
379 (``<base>-0000N-of-0000M.gguf``) every shard must be cached, since
380 llama.cpp loads the whole set from the first shard but needs them all.
381 """
382 shards = split_shard_filenames(gguf_filename)
383 if len(shards) == 1:
384 return True
385 return all(self._find_cached_gguf(hf_repo, shard) is not None for shard in shards)
387 def shard_paths(self, ref: str) -> list[Path]:
388 """On-disk paths of *ref*'s GGUF shards that exist next to its resolved path.
390 A split GGUF resolves to its first shard's snapshot symlink with the
391 siblings co-located, so every shard is returned; a single-file GGUF
392 resolves to its content-hashed blob, where no sibling exists under the
393 real filename. Raises ``KeyError`` / ``ValueError`` like :meth:`resolve`.
394 """
395 first = self.resolve(ref)
396 _repo, filename = parse_hf_ref(ref)
397 candidates = (
398 first.parent / Path(shard).name for shard in split_shard_filenames(Path(filename).name)
399 )
400 return [path for path in candidates if path.exists()]
402 def _reregister_from_cache(
403 self,
404 hf_repo: str,
405 gguf_filename: str,
406 blob_path: Path,
407 snapshot_path: Path | None = None,
408 ) -> None:
409 """Best-effort manifest write for a cache-recovered model so listings see it.
411 *snapshot_path* is the first shard's snapshot path (siblings co-located);
412 when given, the split-shard accounting is recovered too, so a cache-only
413 split GGUF still frees every shard and reports its full size.
414 """
415 ref = format_native_gguf_ref(hf_repo, gguf_filename)
416 try:
417 task = ModelTask(reclassify_by_name(ref, ModelTask.CHAT))
418 total_size, shard_blobs = (
419 _shard_accounting(snapshot_path) if snapshot_path is not None else (None, [])
420 )
421 self._write_manifest(
422 ModelManifest(
423 hf_repo=hf_repo,
424 gguf_filename=gguf_filename,
425 size_bytes=blob_path.stat().st_size,
426 task=task,
427 downloaded_at=datetime.now(UTC).isoformat(),
428 # Only a file under blobs/ is named by its sha; a symlink-less
429 # snapshot file is not, so no digest is recorded for it.
430 blob=blob_path.name if blob_path.parent.name == "blobs" else None,
431 total_size_bytes=total_size,
432 shard_blobs=shard_blobs,
433 )
434 )
435 log.info("Recovered manifest for %s from the model cache", ref)
436 except Exception: # cache-warming write; the resolve already returned a path
437 log.debug("Could not re-register %s from the model cache", ref, exc_info=True)
439 def is_installed(self, ref: str) -> bool:
440 """Return True if a model is installed and its blob is present."""
441 try:
442 self.resolve(ref)
443 return True
444 except (KeyError, ValueError):
445 return False
447 def install(
448 self,
449 hf_repo: str,
450 gguf_filename: str,
451 source_path: Path,
452 manifest: ModelManifest,
453 ) -> Path:
454 """Write a manifest, copying *source_path* into the HF cache if needed."""
455 digest = _blob_digest(source_path)
456 cache_path = self._repo_cache_dir(hf_repo)
457 blobs_dir = cache_path / "blobs"
458 blob_path = blobs_dir / digest
459 if not blob_path.exists():
460 blobs_dir.mkdir(parents=True, exist_ok=True)
461 _copy_atomic(source_path, blob_path)
463 updated = ModelManifest(
464 hf_repo=hf_repo,
465 gguf_filename=gguf_filename,
466 # Record the size install actually wrote, not the caller's claim,
467 # so the on-disk size check has a trustworthy reference.
468 size_bytes=source_path.stat().st_size,
469 task=manifest.task,
470 downloaded_at=manifest.downloaded_at,
471 blob=digest,
472 # Carry the split-shard accounting through unchanged (computed by the
473 # caller from the full shard set); install only rewrites the primary.
474 total_size_bytes=manifest.total_size_bytes,
475 shard_blobs=manifest.shard_blobs,
476 )
477 self._write_manifest(updated)
478 return blob_path
480 def remove(self, ref: str) -> bool:
481 """Remove a manifest and its backing blob.
483 The blob is shared via SHA-256 digest, so it only goes away
484 when no other installed manifest references the same digest.
485 Empty cache directories (``blobs/``, the per-repo ``models--``
486 folder, and the per-repo manifest folder) are pruned so a
487 deleted model leaves no orphan bytes behind.
488 """
489 try:
490 hf_repo, gguf_filename = parse_hf_ref(ref)
491 except ValueError:
492 return False
493 manifest = self._read_manifest(hf_repo, gguf_filename)
494 if manifest is None:
495 return False
496 # Manifests written before shard accounting existed have no shard_blobs, so
497 # recover them from the cache *before* unlinking (resolve needs the manifest).
498 shard_blobs = manifest.shard_blobs or self._recover_legacy_shard_blobs(ref)
499 manifest_path = self._manifest_path(hf_repo, gguf_filename)
500 manifest_path.unlink()
501 repo_dir = manifest_path.parent
502 if repo_dir.exists() and not any(repo_dir.iterdir()):
503 repo_dir.rmdir()
504 self._unlink_snapshot_entries(manifest)
505 # Free the primary blob and every extra shard blob; a split GGUF has more
506 # than one, and leaving the others orphans them when a sibling quant keeps
507 # the repo cache dir alive. The surviving manifests for this repo are read
508 # once here rather than per digest (list_installed walks the whole tree).
509 # A symlink-less recovery records no digest; GC still runs once so an
510 # unused repo cache dir is pruned.
511 siblings = [m for m in self.list_installed() if m.hf_repo == manifest.hf_repo]
512 digests: list[str | None] = [d for d in [manifest.blob, *shard_blobs] if d is not None]
513 for digest in digests or [None]:
514 self._gc_blob(manifest.hf_repo, digest, siblings=siblings)
515 log.info("Removed model %s", ref)
516 return True
518 def _unlink_snapshot_entries(self, manifest: ModelManifest) -> None:
519 """Drop the snapshot entries for *manifest*'s shards.
521 On a symlink-less cache they hold the bytes and would resurrect the
522 model via cache recovery; on a symlinked cache they would dangle once
523 the blob is gc'd.
524 """
525 for shard in split_shard_filenames(manifest.gguf_filename):
526 snapshot_entry = self._snapshot_gguf_path(manifest.hf_repo, shard)
527 if snapshot_entry is not None:
528 snapshot_entry.unlink(missing_ok=True)
530 def _recover_legacy_shard_blobs(self, ref: str) -> list[str]:
531 """Extra shard blob digests for a pre-accounting split GGUF, best-effort.
533 Older manifests recorded only the first shard, so removing them would
534 orphan the rest. Derive the sibling shards from the cache; empty on any
535 failure or for a single-file model, so removal never breaks.
536 """
537 with contextlib.suppress(Exception):
538 shards = self.shard_paths(ref)
539 return [_blob_digest(path) for path in shards[1:]]
540 return []
542 def _gc_blob(
543 self, hf_repo: str, digest: str | None, *, siblings: list[ModelManifest] | None = None
544 ) -> None:
545 """Drop blob bytes and HuggingFace cache cruft now that *digest*
546 and possibly the whole repo are unused. A None *digest* (a manifest
547 recovered on a symlink-less cache records no digest) only prunes the
548 repo dir when no installed manifest is left.
550 When the per-repo ``models--<repo>/`` directory has no installed
551 manifests left, the whole directory is wiped so HF's ``refs/``,
552 ``snapshots/``, and stale ``blobs/`` all go with it. Otherwise
553 only the specific blob file is removed when no remaining
554 manifest still references its digest.
556 ``siblings`` is the surviving-manifest list for *hf_repo*; callers
557 freeing several blobs at once pass it in so the manifest tree is walked
558 once instead of per digest. Defaults to reading it when omitted.
559 """
560 cache_path = self._repo_cache_dir(hf_repo)
561 try:
562 validate_path_within(cache_path, self._root)
563 except ValueError:
564 log.warning("Refusing to remove cache outside models_dir: %s", cache_path)
565 return
566 if siblings is None:
567 siblings = [m for m in self.list_installed() if m.hf_repo == hf_repo]
568 if not siblings:
569 if cache_path.exists():
570 shutil.rmtree(cache_path)
571 return
572 if digest is None: # no digest recorded (symlink-less recovery): dir-prune only
573 return
574 if any(digest == m.blob or digest in m.shard_blobs for m in siblings):
575 return
576 blob_file = cache_path / "blobs" / digest
577 try:
578 validate_path_within(blob_file, self._root)
579 except ValueError:
580 log.warning("Refusing to remove blob outside models_dir: %s", blob_file)
581 return
582 if blob_file.exists():
583 blob_file.unlink()
585 def list_installed(self) -> list[ModelManifest]:
586 """Return manifests for models whose blob is fully present on disk.
588 A manifest with a null blob field or a missing blob file is the
589 residue of a canceled or partial download. Surfacing it would
590 let the picker offer an unusable selection, so the read filter
591 lives here at the source instead of in every UI caller.
593 An absent manifest tree is an empty list. A tree that is present but
594 cannot be read raises ``OSError``, because "unknown" is not "none".
595 """
596 manifests: list[ModelManifest] = []
597 try:
598 repo_dirs = sorted(self._manifests_dir.iterdir())
599 except FileNotFoundError:
600 return manifests
601 for repo_dir in repo_dirs:
602 if not repo_dir.is_dir():
603 continue
604 for tag_file in _manifest_files(repo_dir):
605 manifest = self._load_manifest_file(tag_file)
606 if manifest is not None and self._blob_present(manifest):
607 manifests.append(manifest)
608 return manifests
610 def _blob_present(self, manifest: ModelManifest) -> bool:
611 """True iff *manifest*'s bytes are on disk with a matching size."""
612 return self._manifest_backing_file(manifest) is not None
614 def _manifest_backing_file(self, manifest: ModelManifest) -> Path | None:
615 """The on-disk file holding *manifest*'s bytes, or None if absent/truncated.
617 Normally the content-hashed blob; on a cache without symlinks (the
618 Windows default) the bytes live at the snapshot path, so that is the
619 fallback. ``resolve`` and ``list_installed`` must both gate on this
620 one predicate, or installed-ness disagrees between ``pull`` and ``ask``.
621 """
622 if manifest.blob is not None:
623 blob_file = self._repo_cache_dir(manifest.hf_repo) / "blobs" / manifest.blob
624 if _blob_size_matches(blob_file, manifest.size_bytes):
625 return blob_file
626 recovered = self._find_cached_gguf(manifest.hf_repo, manifest.gguf_filename)
627 if recovered is not None and _blob_size_matches(recovered, manifest.size_bytes):
628 return recovered
629 return None
631 def get_manifest(self, ref: str) -> ModelManifest | None:
632 """Return the manifest for *ref* or None if not installed."""
633 try:
634 hf_repo, gguf_filename = parse_hf_ref(ref)
635 except ValueError:
636 return None
637 return self._read_manifest(hf_repo, gguf_filename)
639 def installed_ref_for_repo(self, hf_repo: str) -> str | None:
640 """Full ``<repo>/<file>.gguf`` ref of an installed quant of *hf_repo*, or None.
642 Alphabetical-first when several quants are installed, matching
643 ``_resolve_repo_only``'s determinism.
644 """
645 refs = sorted(m.ref for m in self.list_installed() if m.hf_repo == hf_repo)
646 return refs[0] if refs else None
648 def _manifest_path(self, hf_repo: str, gguf_filename: str) -> Path:
649 repo = _validate_hf_repo(hf_repo)
650 filename = _validate_gguf_filename(gguf_filename)
651 path = self._manifests_dir / repo_to_dir(repo) / f"{filename}.json"
652 validate_path_within(path, self._manifests_dir)
653 return path
655 def _read_manifest(self, hf_repo: str, gguf_filename: str) -> ModelManifest | None:
656 return self._load_manifest_file(self._manifest_path(hf_repo, gguf_filename))
658 def _write_manifest(self, manifest: ModelManifest) -> None:
659 path = self._manifest_path(manifest.hf_repo, manifest.gguf_filename)
660 path.parent.mkdir(parents=True, exist_ok=True)
661 data = json.dumps(asdict(manifest), indent=2)
662 tmp_path: str | None = None
663 try:
664 with tempfile.NamedTemporaryFile(
665 dir=path.parent, suffix=".tmp", mode="w", encoding="utf-8", delete=False
666 ) as tmp:
667 tmp_path = tmp.name
668 tmp.write(data)
669 os.replace(tmp_path, path)
670 except BaseException:
671 if tmp_path is not None:
672 Path(tmp_path).unlink(missing_ok=True)
673 raise
675 def _load_manifest_file(self, path: Path) -> ModelManifest | None:
676 if not path.exists():
677 return None
678 try:
679 data = json.loads(path.read_text(encoding="utf-8"))
680 return ModelManifest(**data)
681 # UnicodeDecodeError is a ValueError, not a JSONDecodeError.
682 except (json.JSONDecodeError, UnicodeDecodeError, TypeError, KeyError):
683 log.warning("Corrupt manifest: %s", path)
684 return None
687_HF_SNAPSHOTS_DIR = "snapshots"
690def _repo_relative_gguf_name(file_path: Path) -> str:
691 """Recover the repo-relative GGUF filename, keeping any subdir prefix.
693 HF caches a file at ``models--<repo>/snapshots/<rev>/[<subdir>/]<name>``. A
694 subdir-quant giant (unsloth stores quants under e.g. ``Q4_K_M/``) must
695 register under that subdir-relative name so its manifest key round-trips with
696 the ref; ``file_path.name`` alone would drop the subdir. Falls back to the
697 basename when the path is not under a snapshot revision dir.
698 """
699 parts = file_path.parts
700 if _HF_SNAPSHOTS_DIR not in parts:
701 return file_path.name
702 rev_index = parts.index(_HF_SNAPSHOTS_DIR) + 1
703 relative_parts = parts[rev_index + 1 :]
704 return "/".join(relative_parts) if relative_parts else file_path.name
707def _shard_accounting(first_shard_path: Path) -> tuple[int | None, list[str]]:
708 """Total on-disk size and non-primary shard blob digests for a split GGUF.
710 ``(None, [])`` for a single-file model. For a split GGUF, the sibling shards
711 live next to the first shard; ``_blob_digest`` yields each shard's blob digest
712 (the HF cache blob name, or a content hash in copy/non-symlink mode), so the
713 shards are summed and the digests of shards 2..N collected for removal-time
714 garbage collection.
715 """
716 shard_names = split_shard_filenames(first_shard_path.name)
717 if len(shard_names) <= 1:
718 return None, []
719 total = 0
720 shard_blobs: list[str] = []
721 for index, name in enumerate(shard_names):
722 shard_path = first_shard_path.with_name(name)
723 if not shard_path.exists():
724 continue
725 total += shard_path.stat().st_size
726 if index > 0: # the primary blob is tracked separately as manifest.blob
727 shard_blobs.append(_blob_digest(shard_path))
728 return total, shard_blobs
731def register_downloaded_model(entry: CatalogModel, file_path: Path) -> None:
732 """Write a registry manifest for a freshly downloaded GGUF.
734 A failed manifest write is logged, not raised, when the GGUF is still in the
735 HF cache (``resolve`` recovers from it); if it isn't, the download itself is
736 broken and the failure propagates so the caller reports it.
737 """
738 registry = ModelRegistry(cfg.models_dir)
739 gguf_filename = _repo_relative_gguf_name(file_path)
740 total_size, shard_blobs = _shard_accounting(file_path)
741 manifest = ModelManifest(
742 hf_repo=entry.hf_repo,
743 gguf_filename=gguf_filename,
744 size_bytes=file_path.stat().st_size,
745 task=entry.task,
746 downloaded_at=datetime.now(UTC).isoformat(),
747 total_size_bytes=total_size,
748 shard_blobs=shard_blobs,
749 )
750 try:
751 registry.install(entry.hf_repo, gguf_filename, file_path, manifest)
752 log.info("Registered %s/%s in manifest", entry.hf_repo, gguf_filename)
753 except Exception:
754 ref = format_native_gguf_ref(entry.hf_repo, gguf_filename)
755 if not registry.is_installed(ref):
756 raise
757 log.warning(
758 "Manifest write failed for %s; recovered via the model cache", ref, exc_info=True
759 )