Coverage for src/lilbee/catalog/download.py: 100%

291 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""GGUF download, mmproj resolution, post-download hooks.""" 

2 

3import fnmatch 

4import logging 

5import os 

6import shutil 

7import sys 

8import time 

9from collections.abc import Callable, Iterable 

10from http import HTTPStatus 

11from pathlib import Path 

12from typing import Any, NamedTuple, TypeVar 

13 

14import httpx 

15from pydantic import BaseModel 

16 

17from lilbee.catalog.compat import UnsupportedQuantError, classify, file_header 

18from lilbee.catalog.download_progress import ProgressCallback, _ProgressTracker 

19from lilbee.catalog.hf_client import ( 

20 DEFAULT_TIMEOUT, 

21 HF_API_URL, 

22 hf_headers, 

23 hf_token, 

24 repo_has_mmproj, 

25) 

26from lilbee.catalog.models import CatalogModel 

27from lilbee.catalog.refs import ( 

28 DEFAULT_MMPROJ_PATTERN, 

29 FLOAT_QUANTS, 

30 WILDCARD, 

31 quant_label, 

32 rank_gguf_candidates, 

33 split_shard_filenames, 

34) 

35from lilbee.catalog.types import ModelCompat, ModelTask 

36from lilbee.runtime.cancellation import CancelSignal, TaskCancelledError 

37 

38_T = TypeVar("_T") 

39 

40CompleteCallback = Callable[[CatalogModel, Path], None] 

41# Raises UnsupportedQuantError when the engine cannot decode the named file. 

42LoadCheck = Callable[[str, str], None] 

43# Called once per file resolved with the Hub, before any bytes transfer, with 

44# the cache blob that file occupies (None when the Hub reports no blob). 

45ProbeCallback = Callable[[str | None], None] 

46 

47log = logging.getLogger(__name__) 

48 

49 

50class RemoteFile(NamedTuple): 

51 """The Hub's answer about one repo file. 

52 

53 *blob* is the name the file takes in the cache's blob directory, and it is 

54 None when the Hub reports no entity tag for the file. 

55 """ 

56 

57 size: int 

58 blob: str | None 

59 

60 

61def _models_dir() -> Path: 

62 """Deferred cfg read: a module-level cfg import is circular via Config()'s 

63 model-ref validator (config -> model_ref -> catalog -> here -> config).""" 

64 from lilbee.core.config.model import cfg 

65 

66 return cfg.models_dir 

67 

68 

69class DownloadConfig(BaseModel): 

70 model_config = {"arbitrary_types_allowed": True} 

71 

72 repo_id: str 

73 filename: str 

74 token: str | None 

75 force_download: bool = False 

76 cache_dir: str | None = None 

77 tqdm_class: Any = None 

78 

79 

80_BYTES_PER_GB = 1024**3 

81 

82 

83def _free_bytes(path: Path) -> int | None: 

84 """Free space on the volume that will hold *path*, which need not exist yet. 

85 

86 Measured at the nearest existing ancestor, since shutil.disk_usage raises on 

87 a missing path. 

88 """ 

89 probe = path.resolve() 

90 while True: 

91 try: 

92 return shutil.disk_usage(probe).free 

93 except OSError: 

94 if probe.parent == probe: 

95 return None 

96 probe = probe.parent 

97 

98 

99def disk_shortfall(models_dir: Path, hf_repo: str, needed: int) -> str | None: 

100 """Describe why *needed* bytes will not fit, or None when they will. 

101 

102 A partial blob from an interrupted attempt is not counted: huggingface_hub 

103 writes each transfer to a fresh temporary file, so those bytes are spent. 

104 """ 

105 if needed == _SIZE_UNKNOWN: 

106 return None # offline or unresolvable; nothing to compare against 

107 free = _free_bytes(models_dir) 

108 if free is None: 

109 return None # unmeasurable volume; let the download report the truth 

110 if needed <= free: 

111 return None 

112 return ( 

113 f"Not enough disk space for {hf_repo}: needs " 

114 f"{needed / _BYTES_PER_GB:.1f} GB, {free / _BYTES_PER_GB:.1f} GB free." 

115 ) 

116 

117 

118def discard_partial_blobs(models_dir: Path, hf_repo: str, blobs: Iterable[str]) -> None: 

119 """Delete the temporary files of *blobs*, which no later download reads. 

120 

121 huggingface_hub writes every transfer to a fresh ``<blob>.<unique>.incomplete`` 

122 name and unlinks it on the way out, so a leftover belongs to an attempt that 

123 never unwound: a terminated child, a power loss, a build that predates this 

124 sweep. Left alone the bytes are lost for the life of the cache. 

125 

126 Scoped to the blobs the caller resolved, because another quant of the same 

127 repo downloads into the same directory at the same time and its temporary 

128 file is live. 

129 """ 

130 from huggingface_hub.file_download import repo_folder_name 

131 

132 repo_dir = models_dir / repo_folder_name(repo_id=hf_repo, repo_type="model") 

133 for blob in blobs: 

134 for partial in repo_dir.glob(f"blobs/{blob}.*.incomplete"): 

135 try: 

136 partial.unlink() 

137 except OSError: 

138 log.warning("Left a partial download behind: %s", partial) 

139 

140 

141def _require_disk_space(entry: CatalogModel, models_dir: Path, needed: int) -> None: 

142 """Refuse a download the disk cannot hold, naming the shortfall. 

143 

144 huggingface_hub only warns, and the xet path reports a full disk as a 

145 reconstruction error naming neither the disk nor the file. 

146 """ 

147 message = disk_shortfall(models_dir, entry.hf_repo, needed) 

148 if message is not None: 

149 raise RuntimeError(message) 

150 

151 

152_LOW_DISK_FLOOR = 512 * 1024**2 

153"""Free bytes below which a failed download is reported as a full disk. 

154 

155Catches a volume that filled mid-transfer, which the pre-flight cannot see.""" 

156 

157 

158def _raise_if_disk_exhausted( 

159 entry: CatalogModel, config: DownloadConfig, cause: BaseException 

160) -> None: 

161 """Re-raise a failed download as a disk problem when the volume is full. 

162 

163 Low free space is a heuristic, not a diagnosis, so *cause* stays in the 

164 message. 

165 """ 

166 if config.cache_dir is None: 

167 return 

168 try: 

169 free = shutil.disk_usage(config.cache_dir).free 

170 except OSError: 

171 return # the path went away with the failure; leave the original error 

172 if free >= _LOW_DISK_FLOOR: 

173 return 

174 raise RuntimeError( 

175 f"Ran out of disk space downloading {entry.hf_repo}: " 

176 f"{free / _BYTES_PER_GB:.1f} GB free. {type(cause).__name__}: {cause}" 

177 ) from None 

178 

179 

180_XET_HIGH_PERFORMANCE_ENV = "HF_XET_HIGH_PERFORMANCE" 

181 

182_XET_DISABLE_ENV = "HF_HUB_DISABLE_XET" 

183 

184 

185def _disable_xet_where_it_stalls() -> None: 

186 """Fall back to the plain HTTP download path on Windows. 

187 

188 hf_xet transfers stall or deadlock on Windows (xet-core issues #446, 

189 #789, #850), while the plain path downloads at line speed. Everywhere 

190 else xet stays on deliberately: it is the fast path. A user who 

191 exported the variable keeps whatever they chose. huggingface_hub 

192 parses the variable once at import, so the hub constant must change 

193 too; the environment write covers worker subprocesses, which parse it 

194 fresh. 

195 """ 

196 if sys.platform != "win32": 

197 return 

198 if _XET_DISABLE_ENV in os.environ: 

199 return 

200 from huggingface_hub import constants 

201 

202 os.environ[_XET_DISABLE_ENV] = "1" 

203 constants.HF_HUB_DISABLE_XET = True 

204 

205 

206def _apply_fast_download_mode() -> None: 

207 """Publish the high-performance setting to xet before it builds a session. 

208 

209 hf_xet reads it from the environment in Rust and caches it when the session 

210 is built, so a change lands on restart. 

211 """ 

212 # circular: catalog.download -> core.config via cfg, the same cycle 

213 # _models_dir documents (config -> model_ref -> catalog -> here). 

214 from lilbee.core.config.model import cfg 

215 

216 if cfg.fast_model_downloads: 

217 os.environ[_XET_HIGH_PERFORMANCE_ENV] = "1" 

218 else: 

219 os.environ.pop(_XET_HIGH_PERFORMANCE_ENV, None) 

220 

221 

222_TRANSFER_RETRIES = 2 

223_RETRY_BACKOFF_SECONDS = 5 

224 

225 

226class _TransientDownloadError(RuntimeError): 

227 """A transfer fault another attempt can clear: a network or an I/O error.""" 

228 

229 

230def _retry_transient(work: Callable[[], _T]) -> _T: 

231 """Run *work*, retrying only the faults another attempt can clear. 

232 

233 Every other error propagates on the first attempt, cancellation included, 

234 so a defect or a configuration error is never hidden behind a retry. A 

235 transfer that goes quiet raises nothing here, so it is the parent process 

236 that ends it, not this loop. 

237 """ 

238 last_error: Exception | None = None 

239 for attempt in range(_TRANSFER_RETRIES + 1): 

240 try: 

241 return work() 

242 except _TransientDownloadError as exc: 

243 last_error = exc 

244 log.warning("%s (attempt %d/%d)", exc, attempt + 1, _TRANSFER_RETRIES + 1) 

245 if attempt < _TRANSFER_RETRIES: 

246 time.sleep(_RETRY_BACKOFF_SECONDS * (attempt + 1)) 

247 raise RuntimeError( 

248 f"{last_error}. The transfer failed {_TRANSFER_RETRIES + 1} times. Check the " 

249 "network connection and retry; the files that finished are kept." 

250 ) from last_error 

251 

252 

253def _download_with_retry(entry: CatalogModel, config: DownloadConfig) -> Path: 

254 """Run one file's transfer, retrying the faults another attempt can clear.""" 

255 return _retry_transient(lambda: _hf_download_or_translate(entry, config)) 

256 

257 

258def _hf_download_or_translate(entry: CatalogModel, config: DownloadConfig) -> Path: 

259 """Run the HF download and translate every error class into a clean exception.""" 

260 from huggingface_hub import hf_hub_download 

261 from huggingface_hub.utils import EntryNotFoundError, GatedRepoError, RepositoryNotFoundError 

262 

263 _disable_xet_where_it_stalls() 

264 try: 

265 return Path(hf_hub_download(**config.model_dump(exclude_none=True))) 

266 except TaskCancelledError: 

267 raise 

268 except GatedRepoError: 

269 raise PermissionError( 

270 f"{entry.hf_repo} requires HuggingFace authentication. " 

271 "Set HF_TOKEN env var or visit the repo page to request access." 

272 ) from None 

273 except RepositoryNotFoundError: 

274 raise RuntimeError(f"Repository {entry.hf_repo!r} not found on HuggingFace.") from None 

275 except EntryNotFoundError: 

276 raise RuntimeError(_missing_file_message(entry.hf_repo, config.filename)) from None 

277 except (httpx.TimeoutException, httpx.ConnectError) as exc: 

278 raise _TransientDownloadError(f"Network error downloading {entry.hf_repo}: {exc}") from None 

279 except OSError as exc: 

280 raise _TransientDownloadError(f"I/O error downloading {entry.hf_repo}: {exc}") from None 

281 except Exception as exc: 

282 _raise_if_disk_exhausted(entry, config, exc) 

283 raise RuntimeError( 

284 f"Failed to download {entry.hf_repo}: {type(exc).__name__}: {exc}" 

285 ) from None 

286 

287 

288class _NeverCancelled: 

289 """Cancel signal for a caller that has no way to stop the download.""" 

290 

291 def is_set(self) -> bool: 

292 """Always False; nothing holds a handle that could set this.""" 

293 return False 

294 

295 

296_NEVER_CANCELLED = _NeverCancelled() 

297 

298 

299def download_model( 

300 entry: CatalogModel, 

301 *, 

302 on_progress: ProgressCallback | None = None, 

303 on_complete: CompleteCallback | None = None, 

304 cancel: CancelSignal | None = None, 

305) -> Path: 

306 """Download a GGUF model from HuggingFace to the models dir. 

307 Uses huggingface_hub for caching and auth; a stalled file starts again 

308 from the top, and files already finished are kept. 

309 The optional *on_progress(downloaded, total)* callback receives byte counts. 

310 The optional *on_complete(entry, file_path)* callback runs after every file 

311 is on disk; modelhub uses it to write a registry manifest. For vision 

312 models, also downloads the mmproj (CLIP projection) file. 

313 

314 The transfer always runs in its own child process, because terminating that 

315 process is the only stop a wedged transfer cannot refuse. A caller that 

316 passes no *cancel* signal gets one that is never set, so the watchdog that 

317 ends a quiet child covers every download. 

318 

319 A split GGUF has every shard fetched before the model is finalized, so the 

320 registry manifest (and thus "installed") only lands once the full set is on 

321 disk; an interrupted multi-part pull leaves the model not-installed and 

322 re-pullable rather than registered-but-unloadable. 

323 

324 Raises: 

325 PermissionError: gated repo requiring authentication 

326 RuntimeError: repo not found or download failure with details 

327 TaskCancelledError: the cancel signal was set 

328 """ 

329 _apply_fast_download_mode() 

330 models_dir = _models_dir() 

331 models_dir.mkdir(parents=True, exist_ok=True) 

332 token = hf_token() 

333 # circular: download -> download_process via fetch_model_files 

334 from lilbee.catalog.download_process import download_in_subprocess 

335 

336 dest = download_in_subprocess( 

337 entry, 

338 models_dir, 

339 token, 

340 on_progress=on_progress, 

341 cancel=_NEVER_CANCELLED if cancel is None else cancel, 

342 ) 

343 if on_complete is not None: 

344 on_complete(entry, dest) 

345 return dest 

346 

347 

348def fetch_model_files( 

349 entry: CatalogModel, 

350 models_dir: Path, 

351 token: str | None, 

352 *, 

353 on_progress: ProgressCallback | None = None, 

354 on_probe: ProbeCallback | None = None, 

355) -> Path: 

356 """Fetch *entry*'s GGUF shards, plus its projector when the repo ships one. 

357 

358 Takes the models dir and token as arguments so a download child process 

359 can run it without reading cfg. Writes no registry state. *on_probe* fires 

360 once per file resolved with the Hub, which is the only sign of life a 

361 caller gets before the first bytes, and it names the cache blob so the 

362 caller can clear that file's leftovers. 

363 """ 

364 filename = resolve_filename(entry) 

365 shards = split_shard_filenames(filename) 

366 dest = models_dir / shards[0] 

367 if all(_shard_is_cached(entry, models_dir, shard, on_probe) for shard in shards): 

368 log.info("Model already downloaded: %s", dest) 

369 if on_progress is not None: 

370 size = sum((models_dir / shard).stat().st_size for shard in shards) 

371 on_progress(size, size) # Report 100% immediately (every shard) 

372 _ensure_projector(entry, models_dir, token, on_progress=on_progress, on_probe=on_probe) 

373 return dest 

374 

375 remote = [_probed_shard(entry, shard, on_probe) for shard in shards] 

376 shard_sizes = [file.size for file in remote] 

377 sizes_known = all(size != _SIZE_UNKNOWN for size in shard_sizes) 

378 # Reclaim first: a leftover partial is unreadable bytes that still occupy 

379 # the volume the space check is about to measure. 

380 discard_partial_blobs(models_dir, entry.hf_repo, _blobs(remote)) 

381 _require_disk_space(entry, models_dir, sum(shard_sizes) if sizes_known else 0) 

382 

383 # Sum the shard sizes up front so a multi-shard pull reports one monotonic 

384 # 0->100% against the real total, not N separate per-shard cycles. Only use 

385 # the sum when every shard size is known (0 = unresolved/offline); a partial 

386 # sum would undercount the total and let progress run past 100%. 

387 grand_total = sum(shard_sizes) if len(shards) > 1 and sizes_known else 0 

388 tracker = _ProgressTracker(on_progress, grand_total=grand_total) if on_progress else None 

389 shard_paths: list[Path] = [] 

390 for shard in shards: 

391 log.info("Downloading %s/%s → %s", entry.hf_repo, shard, models_dir) 

392 config = DownloadConfig( 

393 repo_id=entry.hf_repo, 

394 filename=shard, 

395 token=token, 

396 cache_dir=str(models_dir), 

397 tqdm_class=tracker.make_tqdm_class() if tracker else None, 

398 ) 

399 shard_path = _download_with_retry(entry, config) 

400 shard_paths.append(shard_path) 

401 if tracker is not None: 

402 tracker.shard_done(shard_path.stat().st_size) 

403 first_shard_path = shard_paths[0] # the 00001-of-N shard llama.cpp loads from 

404 

405 if on_progress: 

406 total_size = sum(path.stat().st_size for path in shard_paths) 

407 if not tracker or not tracker.was_used: 

408 log.info("Model found in HuggingFace cache: %s", first_shard_path) 

409 on_progress(total_size, total_size) 

410 _ensure_projector(entry, models_dir, token, on_progress=on_progress, on_probe=on_probe) 

411 return first_shard_path 

412 

413 

414def _blobs(files: Iterable[RemoteFile]) -> list[str]: 

415 """The cache blob names among *files*, dropping the ones the Hub did not report.""" 

416 return [file.blob for file in files if file.blob is not None] 

417 

418 

419def _probe(on_probe: ProbeCallback | None, file: RemoteFile) -> RemoteFile: 

420 """Report *file* as resolved, then hand it back to the caller.""" 

421 if on_probe is not None: 

422 on_probe(file.blob) 

423 return file 

424 

425 

426def _shard_is_cached( 

427 entry: CatalogModel, models_dir: Path, shard: str, on_probe: ProbeCallback | None 

428) -> bool: 

429 """Whether *shard* is on disk at the size the Hub reports for it.""" 

430 file = _probe(on_probe, fetch_remote_file(entry.hf_repo, shard)) 

431 path = models_dir / shard 

432 return path.exists() and _size_matches(path, file.size) 

433 

434 

435def _probed_shard(entry: CatalogModel, shard: str, on_probe: ProbeCallback | None) -> RemoteFile: 

436 """The Hub's answer about *shard*, reported to *on_probe* as it arrives.""" 

437 return _probe(on_probe, fetch_remote_file(entry.hf_repo, shard)) 

438 

439 

440def _ensure_projector( 

441 entry: CatalogModel, 

442 models_dir: Path, 

443 token: str | None, 

444 *, 

445 on_progress: ProgressCallback | None = None, 

446 on_probe: ProbeCallback | None = None, 

447) -> None: 

448 """Fetch the projector whenever the repo ships one, not only for VISION entries. 

449 

450 Dual-use VL repos (Qwen-VL, InternVL, SmolVLM, gemma-3) classify as chat by 

451 name and arch, and without their projector the vision role dies at plan 

452 time with a missing-mmproj warning a re-pull cannot cure. 

453 """ 

454 if entry.task == ModelTask.VISION or repo_has_mmproj(entry.hf_repo): 

455 _fetch_mmproj(entry, models_dir, token, on_progress=on_progress, on_probe=on_probe) 

456 

457 

458def download_mmproj( 

459 entry: CatalogModel, 

460 *, 

461 on_progress: ProgressCallback | None = None, 

462) -> Path | None: 

463 """Download the mmproj (CLIP projection) file for a vision model. 

464 Returns the path to the downloaded file, or None if no mmproj is configured. 

465 The optional ``on_progress`` callback receives ``(downloaded, total)`` byte 

466 counts and is wired through the same tqdm hook used by the main download. 

467 """ 

468 _apply_fast_download_mode() 

469 return _fetch_mmproj(entry, _models_dir(), hf_token(), on_progress=on_progress) 

470 

471 

472def _fetch_mmproj( 

473 entry: CatalogModel, 

474 models_dir: Path, 

475 token: str | None, 

476 *, 

477 on_progress: ProgressCallback | None = None, 

478 on_probe: ProbeCallback | None = None, 

479) -> Path | None: 

480 """Fetch *entry*'s mmproj into *models_dir*, or None when the repo names none.""" 

481 mmproj_filename = _resolve_mmproj_filename(entry.hf_repo, DEFAULT_MMPROJ_PATTERN) 

482 if not mmproj_filename: 

483 log.warning("Could not resolve mmproj file for %s", entry.hf_repo) 

484 return None 

485 

486 tracker = _ProgressTracker(on_progress) if on_progress else None 

487 log.info("Downloading mmproj %s/%s → %s", entry.hf_repo, mmproj_filename, models_dir) 

488 projector = _probe(on_probe, fetch_remote_file(entry.hf_repo, mmproj_filename)) 

489 discard_partial_blobs(models_dir, entry.hf_repo, _blobs([projector])) 

490 _require_disk_space(entry, models_dir, projector.size) 

491 # The projector gets the same error translation and retry as the GGUF. 

492 path = _download_with_retry( 

493 entry, 

494 DownloadConfig( 

495 repo_id=entry.hf_repo, 

496 filename=mmproj_filename, 

497 token=token, 

498 cache_dir=str(models_dir), 

499 tqdm_class=tracker.make_tqdm_class() if tracker else None, 

500 ), 

501 ) 

502 if on_progress is not None and (not tracker or not tracker.was_used): 

503 # Cache hit: HF returned the cached path without invoking tqdm. 

504 size = path.stat().st_size 

505 on_progress(size, size) 

506 return path 

507 

508 

509def _repo_sibling_files(hf_repo: str) -> list[str]: 

510 """Every filename the HuggingFace API lists for *hf_repo*. 

511 

512 Raises: 

513 PermissionError: the repo is gated and needs authentication. 

514 RuntimeError: the listing could not be fetched. 

515 """ 

516 try: 

517 resp = httpx.get( 

518 f"{HF_API_URL}/{hf_repo}", 

519 timeout=DEFAULT_TIMEOUT, 

520 headers=hf_headers(), 

521 ) 

522 if resp.status_code == HTTPStatus.UNAUTHORIZED: 

523 raise PermissionError( 

524 f"{hf_repo} requires HuggingFace authentication. " 

525 "Set HF_TOKEN env var or visit the repo page to request access." 

526 ) 

527 resp.raise_for_status() 

528 siblings = resp.json().get("siblings", []) 

529 except PermissionError: 

530 raise 

531 except Exception as exc: 

532 raise RuntimeError(f"Cannot query files for {hf_repo}: {exc}") from exc 

533 return [s.get("rfilename", "") for s in siblings] 

534 

535 

536def _mmproj_rank(filename: str) -> tuple[bool, str]: 

537 """Sort key preferring an unquantized projector, ties broken by name.""" 

538 return (quant_label(filename) not in FLOAT_QUANTS, filename) 

539 

540 

541def _resolve_mmproj_filename(hf_repo: str, pattern: str) -> str | None: 

542 """Resolve an mmproj filename pattern to a concrete filename via the HF API.""" 

543 if WILDCARD not in pattern: 

544 return pattern 

545 try: 

546 names = _repo_sibling_files(hf_repo) 

547 except (PermissionError, RuntimeError) as exc: 

548 log.warning("Cannot query mmproj files for %s: %s", hf_repo, exc) 

549 return None 

550 matches = [name for name in names if fnmatch.fnmatch(name, pattern)] 

551 return min(matches, key=_mmproj_rank) if matches else None 

552 

553 

554def resolve_filename(entry: CatalogModel, *, can_load: LoadCheck | None = None) -> str: 

555 """The repo file a pull of *entry* fetches, gated on each candidate's GGUF header. 

556 

557 Quant labels only order the candidates. A file whose header calls it a 

558 projector or an adapter is never the model, so a repo that labels its 

559 projector ``Q8_0`` cannot install it as one. 

560 

561 *can_load* is the engine's verdict on one file, supplied by the layer that 

562 owns the engine. It runs last because it is the expensive question, and it 

563 decides between candidates rather than judging the one already chosen: a 

564 repo publishing the same weights in several packings holds files the engine 

565 reads and files it cannot, and only one of them is worth downloading. 

566 

567 Where no architecture is supported the best-ranked weights still come back. 

568 Refusing here would report a generic error and disable ``--allow-unsupported``, 

569 so that verdict belongs to the architecture guard. 

570 

571 Raises: 

572 PermissionError: the repo is gated and needs authentication. 

573 UnsupportedQuantError: every candidate carries weights the engine cannot 

574 decode; the first such refusal is re-raised, naming its file. 

575 RuntimeError: the repo listing failed, or it holds no model weights. 

576 """ 

577 named = entry.gguf_filename 

578 if WILDCARD not in named and file_header(entry.hf_repo, named).is_model: 

579 return named 

580 unsupported: str | None = None 

581 refused: UnsupportedQuantError | None = None 

582 for candidate in rank_gguf_candidates(_repo_sibling_files(entry.hf_repo)): 

583 header = file_header(entry.hf_repo, candidate) 

584 if not header.is_model: 

585 continue 

586 if classify(header.architecture) is ModelCompat.UNSUPPORTED: 

587 unsupported = unsupported or candidate 

588 continue 

589 if can_load is not None: 

590 try: 

591 can_load(entry.hf_repo, candidate) 

592 except UnsupportedQuantError as exc: 

593 refused = refused or exc 

594 continue 

595 return candidate 

596 if unsupported is not None: 

597 return unsupported 

598 if refused is not None: 

599 raise refused 

600 raise RuntimeError(f"No GGUF model weights found in {entry.hf_repo}") 

601 

602 

603_SIZE_UNKNOWN = 0 

604 

605 

606def _size_matches(dest: Path, expected: int) -> bool: 

607 """Whether *dest* holds the *expected* byte count, accepting an unknown one. 

608 

609 An unknown size is accepted because there is nothing to verify against and 

610 refusing would block every offline reuse. A size that disagrees is a 

611 truncated or corrupt file, so the caller fetches it again. 

612 """ 

613 if expected == _SIZE_UNKNOWN: 

614 return True 

615 actual = dest.stat().st_size 

616 if actual == expected: 

617 return True 

618 log.warning( 

619 "Cached %s is %d bytes but HuggingFace reports %d; re-downloading", 

620 dest, 

621 actual, 

622 expected, 

623 ) 

624 return False 

625 

626 

627def _hf_file_metadata(hf_repo: str, filename: str) -> tuple[int | None, str | None]: 

628 """Byte size and cache blob huggingface_hub resolves for *filename*.""" 

629 from huggingface_hub import get_hf_file_metadata, hf_hub_url 

630 

631 metadata = get_hf_file_metadata(hf_hub_url(hf_repo, filename), token=hf_token()) 

632 return metadata.size, metadata.etag 

633 

634 

635def _missing_file_message(hf_repo: str, filename: str) -> str: 

636 """User-facing error for a file the Hub reports as nonexistent.""" 

637 return ( 

638 f"File {filename!r} does not exist in {hf_repo} on HuggingFace. " 

639 "Check the filename on the repo page." 

640 ) 

641 

642 

643def fetch_remote_file(hf_repo: str, filename: str) -> RemoteFile: 

644 """Return the size and cache blob huggingface_hub reports for *filename*. 

645 

646 Resolves via hf_hub's own file metadata (correct revision, redirects, and 

647 LFS/Xet handled uniformly) instead of scraping the repo tree. Reports an 

648 unknown size when offline or unresolvable, in which case the caller keeps 

649 the cached file. A file the Hub reports as nonexistent raises instead: that 

650 answer is definitive, and treating it as unknown let a pull of a mistyped 

651 filename accept a stale local file and report success without downloading 

652 anything. 

653 """ 

654 from huggingface_hub.errors import RemoteEntryNotFoundError 

655 

656 try: 

657 size, blob = _hf_file_metadata(hf_repo, filename) 

658 except RemoteEntryNotFoundError: 

659 raise RuntimeError(_missing_file_message(hf_repo, filename)) from None 

660 except Exception: 

661 return RemoteFile(size=_SIZE_UNKNOWN, blob=None) 

662 return RemoteFile(size=size or _SIZE_UNKNOWN, blob=blob) 

663 

664 

665def download_bytes(hf_repo: str, filename: str) -> int: 

666 """Bytes a pull of *filename* fetches, every shard summed, or 0 when unknown. 

667 

668 The exact figure HuggingFace reports, not the catalog row's approximation: 

669 a disk check refuses a real download, so it asks about the real file. A 

670 single unresolvable shard makes the sum unknown rather than short. 

671 """ 

672 sizes = [fetch_remote_file(hf_repo, shard).size for shard in split_shard_filenames(filename)] 

673 return sum(sizes) if all(sizes) else _SIZE_UNKNOWN