Coverage for src/lilbee/data/extract/backends/tokenizer.py: 100%
31 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
1"""lilbee's embedder tokenizer exposed as a xberg plugin tokenizer backend.
3xberg's chunk sizer can budget in tokens, but only with tokenizers it loads itself;
4lilbee's embedder is a GGUF model whose vocab isn't published that way. This lets
5``chunk_size`` be a real token ceiling instead of a chars-per-token guess (off by
62-4x on token-dense text).
7"""
9from __future__ import annotations
11import logging
12from typing import TYPE_CHECKING
14from lilbee.data.types import TokenizerBackendName
16from .registry import BackendKind, XbergBinding, register_binding
18if TYPE_CHECKING:
19 from collections.abc import Callable
21 from lilbee.core.config.model import Config
22 from lilbee.providers.base import LLMProvider
24log = logging.getLogger(__name__)
26# Over-counting fallback for when the exact count is unavailable: splits a touch
27# early rather than emitting an over-length chunk. Mirrors providers.fleet.client.
28_FALLBACK_CHARS_PER_TOKEN = 3
30# Non-empty input for the pre-registration count probe; the result is discarded.
31_PROBE_TEXT = "probe"
34def _estimate_tokens(text: str) -> int:
35 """Conservative token estimate from character length (ceiling division)."""
36 return max(1, -(-len(text) // _FALLBACK_CHARS_PER_TOKEN))
39class LilbeeTokenizerBackend:
40 """Counts chunk-sizing tokens with lilbee's embedder tokenizer.
42 ``count_fn`` is read live (an embedding-model swap needs no re-registration).
43 Only a backend without a local tokenizer degrades to the character estimate;
44 any other failure raises to the caller. xberg requires a non-zero count for
45 non-empty text, so a zero count degrades to the estimate as well.
46 """
48 def __init__(self, *, count_fn: Callable[[str], int]) -> None:
49 self._count_fn = count_fn
51 def name(self) -> str:
52 return TokenizerBackendName.LILBEE
54 def initialize(self) -> None: ...
56 def shutdown(self) -> None: ...
58 def count_tokens(self, text: str) -> int:
59 if not text:
60 return 0
61 try:
62 count = self._count_fn(text)
63 except NotImplementedError: # SDK embedders expose no local tokenizer
64 log.debug("exact token count failed; using character estimate", exc_info=True)
65 return _estimate_tokens(text)
66 return count if count > 0 else _estimate_tokens(text)
69def _make_tokenizer_backend(provider: LLMProvider, cfg: Config) -> LilbeeTokenizerBackend:
70 """Build the tokenizer backend, probing its count before xberg does."""
71 backend = LilbeeTokenizerBackend(count_fn=provider.count_tokens)
72 # xberg masks a probe failure as a validation error; fail first with the real one.
73 backend.count_tokens(_PROBE_TEXT)
74 return backend
77register_binding(
78 XbergBinding(
79 kind=BackendKind.TOKENIZER,
80 name=TokenizerBackendName.LILBEE,
81 enabled=lambda cfg: cfg.token_sizing,
82 make=_make_tokenizer_backend,
83 )
84)