Coverage for src/lilbee/data/extract/backends/tokenizer.py: 100%

31 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""lilbee's embedder tokenizer exposed as a xberg plugin tokenizer backend. 

2 

3xberg's chunk sizer can budget in tokens, but only with tokenizers it loads itself; 

4lilbee's embedder is a GGUF model whose vocab isn't published that way. This lets 

5``chunk_size`` be a real token ceiling instead of a chars-per-token guess (off by 

62-4x on token-dense text). 

7""" 

8 

9from __future__ import annotations 

10 

11import logging 

12from typing import TYPE_CHECKING 

13 

14from lilbee.data.types import TokenizerBackendName 

15 

16from .registry import BackendKind, XbergBinding, register_binding 

17 

18if TYPE_CHECKING: 

19 from collections.abc import Callable 

20 

21 from lilbee.core.config.model import Config 

22 from lilbee.providers.base import LLMProvider 

23 

24log = logging.getLogger(__name__) 

25 

26# Over-counting fallback for when the exact count is unavailable: splits a touch 

27# early rather than emitting an over-length chunk. Mirrors providers.fleet.client. 

28_FALLBACK_CHARS_PER_TOKEN = 3 

29 

30# Non-empty input for the pre-registration count probe; the result is discarded. 

31_PROBE_TEXT = "probe" 

32 

33 

34def _estimate_tokens(text: str) -> int: 

35 """Conservative token estimate from character length (ceiling division).""" 

36 return max(1, -(-len(text) // _FALLBACK_CHARS_PER_TOKEN)) 

37 

38 

39class LilbeeTokenizerBackend: 

40 """Counts chunk-sizing tokens with lilbee's embedder tokenizer. 

41 

42 ``count_fn`` is read live (an embedding-model swap needs no re-registration). 

43 Only a backend without a local tokenizer degrades to the character estimate; 

44 any other failure raises to the caller. xberg requires a non-zero count for 

45 non-empty text, so a zero count degrades to the estimate as well. 

46 """ 

47 

48 def __init__(self, *, count_fn: Callable[[str], int]) -> None: 

49 self._count_fn = count_fn 

50 

51 def name(self) -> str: 

52 return TokenizerBackendName.LILBEE 

53 

54 def initialize(self) -> None: ... 

55 

56 def shutdown(self) -> None: ... 

57 

58 def count_tokens(self, text: str) -> int: 

59 if not text: 

60 return 0 

61 try: 

62 count = self._count_fn(text) 

63 except NotImplementedError: # SDK embedders expose no local tokenizer 

64 log.debug("exact token count failed; using character estimate", exc_info=True) 

65 return _estimate_tokens(text) 

66 return count if count > 0 else _estimate_tokens(text) 

67 

68 

69def _make_tokenizer_backend(provider: LLMProvider, cfg: Config) -> LilbeeTokenizerBackend: 

70 """Build the tokenizer backend, probing its count before xberg does.""" 

71 backend = LilbeeTokenizerBackend(count_fn=provider.count_tokens) 

72 # xberg masks a probe failure as a validation error; fail first with the real one. 

73 backend.count_tokens(_PROBE_TEXT) 

74 return backend 

75 

76 

77register_binding( 

78 XbergBinding( 

79 kind=BackendKind.TOKENIZER, 

80 name=TokenizerBackendName.LILBEE, 

81 enabled=lambda cfg: cfg.token_sizing, 

82 make=_make_tokenizer_backend, 

83 ) 

84)