Coverage for src/lilbee/data/extract/chunk.py: 100%

92 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""Text chunking with optional heading-aware and topic-aware splitting.""" 

2 

3from __future__ import annotations 

4 

5from dataclasses import replace 

6from typing import TYPE_CHECKING 

7 

8from lilbee.core.config import active_config 

9from lilbee.data.types import EmbeddingBackendName, TokenizerBackendName 

10 

11if TYPE_CHECKING: 

12 from xberg import Chunk, ChunkingConfig, ChunkSizing, EmbeddingConfig, TableChunkingMode 

13 

14# Char->token ratio for English. 

15CHARS_PER_TOKEN = 4 

16 

17_SEMANTIC_CHUNKER = "semantic" 

18_MARKDOWN_CHUNKER = "markdown" 

19# xberg's default sizer: chunk_size counts characters. 

20_CHARACTER_SIZING = "characters" 

21# Markdown heading path rendered into a chunk: "# Setup > ## Install". 

22_HEADING_MARK = "#" 

23_BREADCRUMB_SEPARATOR = " > " 

24 

25 

26def _embed_token_cap() -> int | None: 

27 """Tokens the embedder truncates one input to; None when it has no fixed cap.""" 

28 # circular: chunk -> app.services -> retrieval.embedder -> chunk via CHARS_PER_TOKEN 

29 from lilbee.app.services import get_services 

30 

31 return get_services().provider.embed_token_cap() 

32 

33 

34def token_sizing_in_effect(cap: int | None) -> bool: 

35 """Whether chunks are sized in the embedder's tokens: opted in, or its window binds. 

36 

37 The window binds when *cap* is below the character budget ``chunk_size`` 

38 allows, so a character-sized chunk could exceed it at embedding time. 

39 """ 

40 config = active_config() 

41 if config.token_sizing: 

42 return True 

43 return cap is not None and cap < config.chunk_size * CHARS_PER_TOKEN 

44 

45 

46def _char_budget() -> tuple[int, int]: 

47 """Return (max_chars, max_overlap) in characters from the token-based cfg.""" 

48 config = active_config() 

49 max_chars = config.chunk_size * CHARS_PER_TOKEN 

50 max_overlap = min(config.chunk_overlap * CHARS_PER_TOKEN, max_chars // 2) 

51 return max_chars, max_overlap 

52 

53 

54def _tokenizer_sizing() -> ChunkSizing: 

55 """xberg sizing through lilbee's tokenizer backend, bound to the provider on demand. 

56 

57 xberg fails extraction when a sizing names an unregistered tokenizer, so the 

58 bind precedes every sizing that routes to it. The bind is a no-op once the 

59 current provider holds the binding. 

60 """ 

61 from xberg import ChunkSizing 

62 

63 from lilbee.app.services import get_services 

64 

65 from .backends import BackendKind, bind_backend 

66 

67 bind_backend(BackendKind.TOKENIZER, get_services().provider) 

68 return ChunkSizing(type="tokenizer", model=TokenizerBackendName.LILBEE) 

69 

70 

71def _size_params() -> tuple[int, int, ChunkSizing | str]: 

72 """Return (max, overlap, sizing) for the plain and heading chunkers. 

73 

74 Under token sizing (``cfg.token_sizing``, or an embed window below the 

75 character budget) the budget is a token count no larger than the embedder's 

76 cap and ``sizing`` routes to lilbee's tokenizer backend, so no chunk loses 

77 text at embedding time. Otherwise the character heuristic with xberg's 

78 default character sizer. The semantic chunker does not use this -- it sizes 

79 by characters and ignores ChunkSizing.""" 

80 config = active_config() 

81 cap = _embed_token_cap() 

82 if not token_sizing_in_effect(cap): 

83 max_chars, max_overlap = _char_budget() 

84 return max_chars, max_overlap, _CHARACTER_SIZING 

85 max_tokens = config.chunk_size if cap is None else min(config.chunk_size, cap) 

86 overlap = min(config.chunk_overlap, max_tokens // 2) 

87 return max_tokens, overlap, _tokenizer_sizing() 

88 

89 

90def _semantic_embedding_config() -> EmbeddingConfig: 

91 """EmbeddingConfig for semantic chunking. Boundary-detection embeddings route to 

92 lilbee's embedder, registered as xberg's plugin backend in 

93 ``lilbee.data.extract.backends.registry`` (embedding binding), so the model that 

94 vectorizes chunks for retrieval is the one that decides where they split.""" 

95 from xberg import EmbeddingConfig, EmbeddingModelType 

96 

97 model = EmbeddingModelType.plugin(EmbeddingBackendName.LILBEE) 

98 return EmbeddingConfig(model=model) 

99 

100 

101def _table_chunking() -> TableChunkingMode | None: 

102 """Header-repeating table splits when table extraction is on, else None for 

103 xberg's default. 

104 

105 REPEAT_HEADER carries the header row into every piece of a long table, so 

106 no chunk holds headerless rows. 

107 """ 

108 config = active_config() 

109 if not config.table_extraction: 

110 return None 

111 from xberg import TableChunkingMode 

112 

113 return TableChunkingMode.REPEAT_HEADER 

114 

115 

116def build_chunking_config(*, use_semantic: bool = True) -> ChunkingConfig: 

117 """Build an xberg ChunkingConfig from the current cfg.""" 

118 from xberg import ChunkingConfig 

119 

120 config = active_config() 

121 if use_semantic and config.semantic_chunking: 

122 # The semantic chunker sizes by characters and ignores ChunkSizing, so it 

123 # stays on the character budget regardless of cfg.token_sizing. 

124 max_chars, max_overlap = _char_budget() 

125 chunking = ChunkingConfig( 

126 chunker_type=_SEMANTIC_CHUNKER, 

127 embedding=_semantic_embedding_config(), 

128 topic_threshold=config.topic_threshold, 

129 max_characters=max_chars, 

130 overlap=max_overlap, 

131 ) 

132 else: 

133 max_size, max_overlap, sizing = _size_params() 

134 chunking = ChunkingConfig( 

135 max_characters=max_size, 

136 overlap=max_overlap, 

137 sizing=sizing, 

138 ) 

139 # table_chunking has no "unset" value on xberg's frozen ChunkingConfig, so the 

140 # field is left at its default rather than overwritten with None. 

141 mode = _table_chunking() 

142 return chunking if mode is None else replace(chunking, table_chunking=mode) 

143 

144 

145def chunk_text( 

146 text: str, 

147 *, 

148 mime_type: str = "text/plain", 

149 heading_context: bool = False, 

150 use_semantic: bool = True, 

151) -> list[str]: 

152 """Split text into chunks; heading_context wins over use_semantic wins over char-budget.""" 

153 if not text or not text.strip(): 

154 return [] 

155 

156 from xberg import ChunkingConfig, ExtractionConfig 

157 

158 from .xberg import extract_document 

159 

160 if heading_context: 

161 max_size, max_overlap, sizing = _size_params() 

162 chunking = ChunkingConfig( 

163 max_characters=max_size, 

164 overlap=max_overlap, 

165 sizing=sizing, 

166 chunker_type=_MARKDOWN_CHUNKER, 

167 ) 

168 else: 

169 chunking = build_chunking_config(use_semantic=use_semantic) 

170 

171 config = ExtractionConfig(chunking=chunking) 

172 doc = extract_document(text.encode("utf-8"), mime_type, config=config) 

173 if not doc.chunks: 

174 return [] 

175 if heading_context: 

176 return [_with_heading_breadcrumb(c) for c in doc.chunks] 

177 return [c.content for c in doc.chunks] 

178 

179 

180def _with_heading_breadcrumb(chunk: Chunk) -> str: 

181 """Prefix a markdown chunk with its heading path, e.g. ``# Setup > ## Install``. 

182 

183 xberg's ``render_heading_breadcrumb`` is Rust-only; the Python binding exposes 

184 the headings as ``metadata.heading_context``. 

185 """ 

186 context = chunk.metadata.heading_context 

187 if context is None or not context.headings: 

188 return chunk.content 

189 breadcrumb = _BREADCRUMB_SEPARATOR.join( 

190 f"{_HEADING_MARK * h.level} {h.text}" for h in context.headings 

191 ) 

192 return f"{breadcrumb}\n\n{chunk.content}" 

193 

194 

195class ChunkLimitError(Exception): 

196 """One file produced more chunks than ``cfg.max_chunks_per_file`` allows.""" 

197 

198 def __init__(self, count: int, limit: int, member: str | None = None) -> None: 

199 prefix = f"{member}: " if member else "" 

200 super().__init__( 

201 f"{prefix}{count} chunks exceed the per-file limit of {limit}; " 

202 f"raise max_chunks_per_file (0 = no limit), then retry skipped files" 

203 ) 

204 self.count = count 

205 self.member = member 

206 self.limit = limit 

207 

208 

209def enforce_chunk_limit(count: int) -> None: 

210 """Refuse a file whose *count* chunks exceed the per-file limit; a limit of 0 accepts any.""" 

211 limit = active_config().max_chunks_per_file 

212 if limit and count > limit: 

213 raise ChunkLimitError(count, limit)