Coverage for src/lilbee/runtime/progress/types.py: 100%

130 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""Event type enums and Pydantic models for the progress protocol.""" 

2 

3from enum import StrEnum 

4 

5from pydantic import BaseModel 

6 

7 

8class EventType(StrEnum): 

9 """Progress event types emitted during sync/ingest. 

10 

11 No value may repeat one in :class:`SseEvent`. The HTTP layer writes a 

12 progress event's value straight into the SSE ``event:`` field, so a shared 

13 value reaches a client as one event name carrying two payload shapes. 

14 """ 

15 

16 FILE_START = "file_start" 

17 FILE_DONE = "file_done" 

18 BATCH_PROGRESS = "batch_progress" 

19 SYNC_DONE = "sync_done" 

20 EMBED = "embed" 

21 EXTRACT = "extract" 

22 OCR_START = "ocr_start" 

23 CRAWL_START = "crawl_start" 

24 CRAWL_PAGE = "crawl_page" 

25 CRAWL_PAGE_FAILED = "crawl_page_failed" 

26 CRAWL_DONE = "crawl_done" 

27 SETUP_START = "setup_start" 

28 SETUP_PROGRESS = "setup_progress" 

29 SETUP_DONE = "setup_done" 

30 WIKI_PHASE = "wiki_phase" 

31 WIKI_PAGE = "wiki_page" 

32 

33 

34class SseEvent(StrEnum): 

35 """SSE event names used in the HTTP streaming protocol. 

36 

37 ``DONE`` is terminal and a stream emits it at most once. 

38 """ 

39 

40 TOKEN = "token" # noqa: S105 -- SSE event name, not a credential 

41 REASONING = "reasoning" 

42 SOURCES = "sources" 

43 RETRIEVAL_QUERY = "retrieval_query" 

44 ERROR = "error" 

45 DONE = "done" 

46 PROGRESS = "progress" 

47 HEARTBEAT = "heartbeat" 

48 ALREADY_INGESTING = "already_ingesting" 

49 WARMING = "warming" 

50 WARM = "warm" 

51 COMPACTING = "compacting" 

52 COMPACTION = "compaction" 

53 MEMORY_EXTRACTED = "memory_extracted" 

54 GPU_STATS = "gpu_stats" 

55 

56 

57class SseErrorCode(StrEnum): 

58 """Stable ``code`` values on SSE error events for clients to branch on.""" 

59 

60 MODEL_TOO_LARGE = "model_too_large" 

61 MODEL_NOT_INSTALLED = "model_not_installed" 

62 INDEX_EMBEDDER_MISMATCH = "index_embedder_mismatch" 

63 

64 

65class FileStartEvent(BaseModel): 

66 """Emitted when a file begins ingestion.""" 

67 

68 file: str 

69 total_files: int 

70 current_file: int 

71 

72 

73class FileDoneEvent(BaseModel): 

74 """Emitted when a file finishes ingestion (success or error).""" 

75 

76 file: str 

77 status: str 

78 chunks: int 

79 

80 

81class BatchStatus(StrEnum): 

82 """Status values for BatchProgressEvent.status.""" 

83 

84 INGESTED = "ingested" 

85 SKIPPED = "skipped" 

86 FAILED = "failed" 

87 RASTERIZING = "rasterizing" 

88 

89 

90class BatchProgressEvent(BaseModel): 

91 """Emitted after each file completes during batch ingestion.""" 

92 

93 file: str 

94 status: BatchStatus 

95 current: int 

96 total: int 

97 

98 

99class OcrBackendUsed(StrEnum): 

100 """The OCR backend an extraction ran with: none (OCR off), Tesseract, or the vision model.""" 

101 

102 NONE = "none" 

103 TESSERACT = "tesseract" 

104 VISION = "vision" 

105 

106 @classmethod 

107 def chosen(cls, enable_ocr: bool | None, vision_model: str) -> "OcrBackendUsed": 

108 """The backend a configuration picks: OCR off wins, then a set vision model.""" 

109 if enable_ocr is False: 

110 return cls.NONE 

111 return cls.VISION if vision_model else cls.TESSERACT 

112 

113 

114_EXTRACT_STEP_NAMES: dict[OcrBackendUsed, str] = { 

115 OcrBackendUsed.NONE: "Extracted", 

116 OcrBackendUsed.TESSERACT: "Tesseract OCR", 

117 OcrBackendUsed.VISION: "Vision OCR", 

118} 

119 

120 

121class ExtractEvent(BaseModel): 

122 """Emitted with page-level extraction progress. 

123 

124 Vision OCR fires one event per page as xberg processes it, as a running 

125 count against the page total known before extraction for a PDF or image 

126 source, or ``0`` when that count could not be read. Extraction then fires 

127 once per file with ``page == total_pages`` so subscribers see "extracted N 

128 pages" before the embed phase ticks. ``ocr_backend`` is the OCR backend 

129 that produced the pages, or ``none`` when no page was OCR'd. 

130 """ 

131 

132 file: str 

133 page: int 

134 total_pages: int 

135 ocr_backend: OcrBackendUsed 

136 

137 @property 

138 def step(self) -> str: 

139 """The step that produced these pages, as a progress line names it.""" 

140 return _EXTRACT_STEP_NAMES[self.ocr_backend] 

141 

142 

143class OcrStartEvent(BaseModel): 

144 """Emitted once when Tesseract starts OCR on a file, which reports no per-page progress. 

145 

146 ``total_pages`` is the file's page count, read before extraction. Tesseract 

147 OCRs the pages that lack a text layer, which may be fewer. 

148 """ 

149 

150 file: str 

151 total_pages: int 

152 

153 @property 

154 def status_text(self) -> str: 

155 """The progress line for this event, naming the file's page count.""" 

156 return ( 

157 f"Tesseract OCR on the scanned pages of {self.file} " 

158 f"({self.total_pages} pages in the file)" 

159 ) 

160 

161 

162class EmbedEvent(BaseModel): 

163 """Emitted per batch during embedding.""" 

164 

165 file: str 

166 chunk: int 

167 total_chunks: int 

168 

169 

170class CrawlStartEvent(BaseModel): 

171 """Emitted when a crawl operation begins.""" 

172 

173 url: str 

174 depth: int 

175 

176 

177# Sentinel used in CrawlPageEvent.total when the crawl's final page count is 

178# not yet known (BFS streaming, page N emitted before N+1 is discovered). 

179# Consumers (plugin, TUI, CLI) treat total <= 0 as indeterminate progress. 

180CRAWL_TOTAL_UNKNOWN = -1 

181 

182 

183class CrawlPageEvent(BaseModel): 

184 """Emitted per page during crawling.""" 

185 

186 url: str 

187 current: int 

188 total: int 

189 

190 

191class CrawlPageFailedEvent(BaseModel): 

192 """Emitted when a crawled page yields nothing to save, with the reason why.""" 

193 

194 url: str 

195 reason: str 

196 

197 

198class CrawlDoneEvent(BaseModel): 

199 """Emitted when a crawl operation completes.""" 

200 

201 pages_crawled: int 

202 files_written: int 

203 

204 

205class SyncDoneEvent(BaseModel): 

206 """Emitted when the sync operation completes.""" 

207 

208 added: int 

209 updated: int 

210 removed: int 

211 failed: int 

212 skipped: int = 0 

213 relocated: int = 0 

214 

215 

216class SetupStartEvent(BaseModel): 

217 """Emitted when a setup/bootstrap operation begins.""" 

218 

219 component: str 

220 size_estimate_bytes: int | None = None 

221 

222 

223class SetupProgressEvent(BaseModel): 

224 """Emitted periodically during a setup/bootstrap operation.""" 

225 

226 component: str 

227 downloaded_bytes: int 

228 total_bytes: int | None = None 

229 detail: str = "" 

230 

231 

232class SetupDoneEvent(BaseModel): 

233 """Emitted when a setup/bootstrap operation completes.""" 

234 

235 component: str 

236 success: bool 

237 error: str | None = None 

238 

239 

240class WikiPhase(StrEnum): 

241 """Stage of a wiki build or synthesis run.""" 

242 

243 EXTRACT = "extract" 

244 GENERATE = "generate" 

245 INDEX = "index" 

246 

247 

248class WikiPhaseEvent(BaseModel): 

249 """Emitted when a wiki run enters a new phase. 

250 

251 ``total`` is the number of units the phase will process (sources for a 

252 build, clusters for synthesis), 0 where the phase has no unit count. 

253 """ 

254 

255 phase: WikiPhase 

256 total: int = 0 

257 

258 

259class WikiPageEvent(BaseModel): 

260 """Emitted after each source (build) or cluster (synthesis) is written.""" 

261 

262 label: str 

263 pages: int 

264 current: int 

265 total: int