Coverage for src/lilbee/data/extract/chunk.py: 100%
92 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
1"""Text chunking with optional heading-aware and topic-aware splitting."""
3from __future__ import annotations
5from dataclasses import replace
6from typing import TYPE_CHECKING
8from lilbee.core.config import active_config
9from lilbee.data.types import EmbeddingBackendName, TokenizerBackendName
11if TYPE_CHECKING:
12 from xberg import Chunk, ChunkingConfig, ChunkSizing, EmbeddingConfig, TableChunkingMode
14# Char->token ratio for English.
15CHARS_PER_TOKEN = 4
17_SEMANTIC_CHUNKER = "semantic"
18_MARKDOWN_CHUNKER = "markdown"
19# xberg's default sizer: chunk_size counts characters.
20_CHARACTER_SIZING = "characters"
21# Markdown heading path rendered into a chunk: "# Setup > ## Install".
22_HEADING_MARK = "#"
23_BREADCRUMB_SEPARATOR = " > "
26def _embed_token_cap() -> int | None:
27 """Tokens the embedder truncates one input to; None when it has no fixed cap."""
28 # circular: chunk -> app.services -> retrieval.embedder -> chunk via CHARS_PER_TOKEN
29 from lilbee.app.services import get_services
31 return get_services().provider.embed_token_cap()
34def token_sizing_in_effect(cap: int | None) -> bool:
35 """Whether chunks are sized in the embedder's tokens: opted in, or its window binds.
37 The window binds when *cap* is below the character budget ``chunk_size``
38 allows, so a character-sized chunk could exceed it at embedding time.
39 """
40 config = active_config()
41 if config.token_sizing:
42 return True
43 return cap is not None and cap < config.chunk_size * CHARS_PER_TOKEN
46def _char_budget() -> tuple[int, int]:
47 """Return (max_chars, max_overlap) in characters from the token-based cfg."""
48 config = active_config()
49 max_chars = config.chunk_size * CHARS_PER_TOKEN
50 max_overlap = min(config.chunk_overlap * CHARS_PER_TOKEN, max_chars // 2)
51 return max_chars, max_overlap
54def _tokenizer_sizing() -> ChunkSizing:
55 """xberg sizing through lilbee's tokenizer backend, bound to the provider on demand.
57 xberg fails extraction when a sizing names an unregistered tokenizer, so the
58 bind precedes every sizing that routes to it. The bind is a no-op once the
59 current provider holds the binding.
60 """
61 from xberg import ChunkSizing
63 from lilbee.app.services import get_services
65 from .backends import BackendKind, bind_backend
67 bind_backend(BackendKind.TOKENIZER, get_services().provider)
68 return ChunkSizing(type="tokenizer", model=TokenizerBackendName.LILBEE)
71def _size_params() -> tuple[int, int, ChunkSizing | str]:
72 """Return (max, overlap, sizing) for the plain and heading chunkers.
74 Under token sizing (``cfg.token_sizing``, or an embed window below the
75 character budget) the budget is a token count no larger than the embedder's
76 cap and ``sizing`` routes to lilbee's tokenizer backend, so no chunk loses
77 text at embedding time. Otherwise the character heuristic with xberg's
78 default character sizer. The semantic chunker does not use this -- it sizes
79 by characters and ignores ChunkSizing."""
80 config = active_config()
81 cap = _embed_token_cap()
82 if not token_sizing_in_effect(cap):
83 max_chars, max_overlap = _char_budget()
84 return max_chars, max_overlap, _CHARACTER_SIZING
85 max_tokens = config.chunk_size if cap is None else min(config.chunk_size, cap)
86 overlap = min(config.chunk_overlap, max_tokens // 2)
87 return max_tokens, overlap, _tokenizer_sizing()
90def _semantic_embedding_config() -> EmbeddingConfig:
91 """EmbeddingConfig for semantic chunking. Boundary-detection embeddings route to
92 lilbee's embedder, registered as xberg's plugin backend in
93 ``lilbee.data.extract.backends.registry`` (embedding binding), so the model that
94 vectorizes chunks for retrieval is the one that decides where they split."""
95 from xberg import EmbeddingConfig, EmbeddingModelType
97 model = EmbeddingModelType.plugin(EmbeddingBackendName.LILBEE)
98 return EmbeddingConfig(model=model)
101def _table_chunking() -> TableChunkingMode | None:
102 """Header-repeating table splits when table extraction is on, else None for
103 xberg's default.
105 REPEAT_HEADER carries the header row into every piece of a long table, so
106 no chunk holds headerless rows.
107 """
108 config = active_config()
109 if not config.table_extraction:
110 return None
111 from xberg import TableChunkingMode
113 return TableChunkingMode.REPEAT_HEADER
116def build_chunking_config(*, use_semantic: bool = True) -> ChunkingConfig:
117 """Build an xberg ChunkingConfig from the current cfg."""
118 from xberg import ChunkingConfig
120 config = active_config()
121 if use_semantic and config.semantic_chunking:
122 # The semantic chunker sizes by characters and ignores ChunkSizing, so it
123 # stays on the character budget regardless of cfg.token_sizing.
124 max_chars, max_overlap = _char_budget()
125 chunking = ChunkingConfig(
126 chunker_type=_SEMANTIC_CHUNKER,
127 embedding=_semantic_embedding_config(),
128 topic_threshold=config.topic_threshold,
129 max_characters=max_chars,
130 overlap=max_overlap,
131 )
132 else:
133 max_size, max_overlap, sizing = _size_params()
134 chunking = ChunkingConfig(
135 max_characters=max_size,
136 overlap=max_overlap,
137 sizing=sizing,
138 )
139 # table_chunking has no "unset" value on xberg's frozen ChunkingConfig, so the
140 # field is left at its default rather than overwritten with None.
141 mode = _table_chunking()
142 return chunking if mode is None else replace(chunking, table_chunking=mode)
145def chunk_text(
146 text: str,
147 *,
148 mime_type: str = "text/plain",
149 heading_context: bool = False,
150 use_semantic: bool = True,
151) -> list[str]:
152 """Split text into chunks; heading_context wins over use_semantic wins over char-budget."""
153 if not text or not text.strip():
154 return []
156 from xberg import ChunkingConfig, ExtractionConfig
158 from .xberg import extract_document
160 if heading_context:
161 max_size, max_overlap, sizing = _size_params()
162 chunking = ChunkingConfig(
163 max_characters=max_size,
164 overlap=max_overlap,
165 sizing=sizing,
166 chunker_type=_MARKDOWN_CHUNKER,
167 )
168 else:
169 chunking = build_chunking_config(use_semantic=use_semantic)
171 config = ExtractionConfig(chunking=chunking)
172 doc = extract_document(text.encode("utf-8"), mime_type, config=config)
173 if not doc.chunks:
174 return []
175 if heading_context:
176 return [_with_heading_breadcrumb(c) for c in doc.chunks]
177 return [c.content for c in doc.chunks]
180def _with_heading_breadcrumb(chunk: Chunk) -> str:
181 """Prefix a markdown chunk with its heading path, e.g. ``# Setup > ## Install``.
183 xberg's ``render_heading_breadcrumb`` is Rust-only; the Python binding exposes
184 the headings as ``metadata.heading_context``.
185 """
186 context = chunk.metadata.heading_context
187 if context is None or not context.headings:
188 return chunk.content
189 breadcrumb = _BREADCRUMB_SEPARATOR.join(
190 f"{_HEADING_MARK * h.level} {h.text}" for h in context.headings
191 )
192 return f"{breadcrumb}\n\n{chunk.content}"
195class ChunkLimitError(Exception):
196 """One file produced more chunks than ``cfg.max_chunks_per_file`` allows."""
198 def __init__(self, count: int, limit: int, member: str | None = None) -> None:
199 prefix = f"{member}: " if member else ""
200 super().__init__(
201 f"{prefix}{count} chunks exceed the per-file limit of {limit}; "
202 f"raise max_chunks_per_file (0 = no limit), then retry skipped files"
203 )
204 self.count = count
205 self.member = member
206 self.limit = limit
209def enforce_chunk_limit(count: int) -> None:
210 """Refuse a file whose *count* chunks exceed the per-file limit; a limit of 0 accepts any."""
211 limit = active_config().max_chunks_per_file
212 if limit and count > limit:
213 raise ChunkLimitError(count, limit)