Coverage for src/lilbee/server/models.py: 100%

486 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""Request and response models for the lilbee HTTP API. 

2 

3Typed pydantic models so Litestar's OpenAPI schema has field-level detail. 

4""" 

5 

6from __future__ import annotations 

7 

8from typing import TYPE_CHECKING, Any, Literal 

9 

10from pydantic import BaseModel, Field, field_validator 

11 

12from lilbee.app.agent_configs.document import AgentClient, AgentSurface, ConfigFormat 

13from lilbee.app.settings_map import SettingGroup 

14from lilbee.catalog.types import KeyStatus, ModelCompat, ModelSource, ModelTask 

15from lilbee.core.config.enums import CrawlRenderMode, KvCacheType 

16from lilbee.core.health_warnings import HealthWarning 

17from lilbee.data.store import ChunkType, IndexMismatch, MemoryKind, scope_to_chunk_type 

18from lilbee.data.types import SkippedSource 

19from lilbee.providers.roles import EngineBackend, WorkerRole 

20from lilbee.runtime.hardware import FitLevel, SizeVariantInfo 

21from lilbee.sessions import MessageRole 

22from lilbee.wiki.entity_extractor import EntityKind 

23 

24if TYPE_CHECKING: 

25 from lilbee.app.agent_configs.detect import ClientDetection 

26 from lilbee.app.agent_configs.document import AgentConfigDocument 

27 from lilbee.app.placement import PlacementView 

28 

29 

30def decode_chunk_type(value: str | None) -> ChunkType | None: 

31 """Decode a ``chunk_type`` string into a ``ChunkType`` at the HTTP boundary. 

32 

33 Delegates to the canonical :func:`scope_to_chunk_type` so query-param and 

34 request-body routes share one decoder: only ``"raw"`` or ``"wiki"`` filter 

35 the pool; everything else (including ``None`` and the UI-side ``"both"``) 

36 means no filter. Any other string raises ``ValueError`` with boundary- 

37 friendly guidance. 

38 """ 

39 try: 

40 return scope_to_chunk_type(value) 

41 except ValueError as exc: 

42 raise ValueError( 

43 f"chunk_type must be one of 'raw', 'wiki', 'both', or omitted; got {value!r}" 

44 ) from exc 

45 

46 

47class AskRequest(BaseModel): 

48 """Request body for /api/ask.""" 

49 

50 question: str 

51 top_k: int = Field(default=0, ge=0, le=100) 

52 options: dict[str, Any] | None = None 

53 chunk_type: ChunkType | None = None 

54 

55 @field_validator("chunk_type", mode="before") 

56 @classmethod 

57 def _check_chunk_type(cls, v: str | None) -> ChunkType | None: 

58 return decode_chunk_type(v) 

59 

60 

61class ChatRequest(BaseModel): 

62 """Request body for /api/chat.""" 

63 

64 question: str 

65 history: list[ChatMessage] = [] 

66 # None (unspecified) grounds with the configured top_k; an explicit 0 is a 

67 # pure-LLM call that skips retrieval entirely. 

68 top_k: int | None = Field(default=None, ge=0, le=100) 

69 options: dict[str, Any] | None = None 

70 chunk_type: ChunkType | None = None 

71 summary: str = "" 

72 """Carry-forward notes from earlier compactions, folded into the prompt.""" 

73 session_id: str | None = None 

74 """Session that receives the new summary when this turn compacts.""" 

75 

76 @field_validator("chunk_type", mode="before") 

77 @classmethod 

78 def _check_chunk_type(cls, v: str | None) -> ChunkType | None: 

79 return decode_chunk_type(v) 

80 

81 

82class SyncRequest(BaseModel): 

83 """Request body for /api/sync. 

84 

85 ``force_rebuild`` triggers a full drop-and-reingest equivalent to ``lilbee rebuild``. 

86 Use it to recover from an embedding-model switch (when the store refuses search 

87 or ingest because ``cfg.embedding_model`` no longer matches the persisted vectors). 

88 ``retry_skipped`` is the lighter recovery: it clears the markers for files that 

89 failed a previous sync (Tesseract timeout, decode failure, no usable text) so this 

90 sync attempts them again, without dropping the existing store. The default is an 

91 incremental sync. 

92 ``prune_ignored`` drops sources a ``.lilbeeignore`` now excludes. Off by default: 

93 the patterns govern what sync takes in, not what a past sync already indexed. 

94 """ 

95 

96 enable_ocr: bool | None = None 

97 ocr_timeout: float | None = None 

98 force_rebuild: bool = False 

99 retry_skipped: bool = False 

100 prune_ignored: bool = False 

101 

102 

103class AddRequest(BaseModel): 

104 """Request body for /api/add.""" 

105 

106 paths: list[str] 

107 force: bool = False 

108 enable_ocr: bool | None = None 

109 ocr_timeout: float | None = None 

110 

111 

112class SetModelRequest(BaseModel): 

113 """Request body for /api/models/chat.""" 

114 

115 model: str 

116 

117 

118class SourceContentResponse(BaseModel): 

119 """JSON body for ``GET /api/source`` (``raw=0``); empty ``markdown`` for binary types.""" 

120 

121 markdown: str 

122 content_type: str 

123 title: str | None = None 

124 

125 

126class ChatMessage(BaseModel): 

127 """A single message in a chat conversation.""" 

128 

129 role: Literal["user", "assistant"] 

130 content: str 

131 

132 

133class CleanedChunk(BaseModel): 

134 """A search result chunk with vector stripped and distance renamed.""" 

135 

136 source: str 

137 content_type: str 

138 chunk: str 

139 distance: float | None = None 

140 relevance_score: float | None = None 

141 rerank_score: float | None = None 

142 # Canonical [0, 1] relevance from retrieval fusion; the ranking signal 

143 # HTTP clients should sort and threshold on (relevance_score is legacy). 

144 score: float | None = None 

145 page_start: int = 0 

146 page_end: int = 0 

147 line_start: int = 0 

148 line_end: int = 0 

149 chunk_index: int = 0 

150 # Vault-relative path when ``cfg.vault_base`` is set and the source file 

151 # lives inside the vault. Absent when the server is running headless or 

152 # the source isn't resolvable as a vault file. Clients use this to open 

153 # the source in a native editor instead of fetching ``/api/source``. 

154 vault_path: str | None = None 

155 # Set only on recalled-memory sources (``source`` is ``memory:<id>``), so 

156 # clients can mark them as memory and link them to ``GET /api/memories``. 

157 memory_id: str | None = None 

158 

159 

160class StatusSourceInfo(BaseModel): 

161 """A single indexed source in a status response.""" 

162 

163 filename: str 

164 file_hash: str 

165 chunk_count: int 

166 ingested_at: str 

167 

168 

169class StatusConfigInfo(BaseModel): 

170 """Configuration section of a status response. 

171 

172 Exposes all four role-bound model fields so plugins/TUI can show 

173 what's active per role without a second round trip. 

174 """ 

175 

176 documents_dir: str 

177 data_dir: str 

178 chat_model: str 

179 embedding_model: str 

180 vision_model: str = "" 

181 reranker_model: str = "" 

182 enable_ocr: bool | None = None 

183 num_ctx: int | None = None 

184 num_ctx_max: int | None = None 

185 chat_n_ctx_target: int | None = None 

186 flash_attention: bool | None = None 

187 kv_cache_type: KvCacheType | None = None 

188 n_gpu_layers: int | None = None 

189 cpu_moe: bool | None = None 

190 n_cpu_moe: int | None = None 

191 main_gpu: int | None = None 

192 gpu_devices: str | None = None 

193 

194 

195class StatusEntityInfo(BaseModel): 

196 """Entity-extraction section of a status response (present when enabled).""" 

197 

198 types: list[str] 

199 rows: int 

200 

201 

202class StatusIndexInfo(BaseModel): 

203 """The embedder that built the persisted index.""" 

204 

205 embedding_model: str 

206 embedding_dim: int 

207 

208 

209class StatusResponse(BaseModel): 

210 """Response for GET /api/status.""" 

211 

212 command: str = "status" 

213 config: StatusConfigInfo 

214 sources: list[StatusSourceInfo] 

215 document_count: int 

216 total_chunks: int 

217 index: StatusIndexInfo | None = None 

218 """The embedder that built the index; absent before the first sync. Compare it 

219 with ``config.embedding_model`` to tell a stale index before a search refuses it.""" 

220 entities: StatusEntityInfo | None = None 

221 skipped: list[SkippedSource] = [] 

222 """Files a skip marker holds out of the index, capped; ``skipped_total`` is the real count.""" 

223 skipped_total: int = 0 

224 ocr_warning: str | None = None 

225 """Set when a vision model is configured but ``enable_ocr`` is false.""" 

226 ocr_note: str | None = None 

227 """Which OCR engine runs for scanned pages; None when OCR is off.""" 

228 

229 

230class ShutdownResponse(BaseModel): 

231 """Response for /api/shutdown.""" 

232 

233 status: Literal["shutting_down"] 

234 

235 

236class HealthResponse(BaseModel): 

237 """Response for /api/health.""" 

238 

239 status: str 

240 version: str 

241 chat_ready: bool = False 

242 """True once the chat engine is loaded and ready to serve a first token. 

243 

244 A launcher polls this to wait out the cold model load before handing off to 

245 a client, so the client never lands on an apparently-dead stream. 

246 """ 

247 chat_status: Literal["ready", "loading", "not_started", "error"] = "not_started" 

248 """Finer-grained chat readiness than the ``chat_ready`` bool. 

249 

250 Lets a polling client tell a fleet that is still loading (wait) apart from one 

251 that never started warming (``not_started`` -- no chat model resolved / planned, 

252 so it will not come up on its own) or failed (``error``). Without this a bare 

253 ``chat_ready:false`` reads the same for "loading" and "hung", which looked like a 

254 silent hang on a fresh box with no chat model installed.""" 

255 chat_error: str | None = None 

256 """The reason the chat engine failed to come up when ``chat_status`` is 

257 ``error`` (e.g. a wedged GPU device probe), so a polling client can report 

258 the cause instead of retrying forever.""" 

259 chat_ctx: int | None = None 

260 """Per-slot context the chat engine serves, so a launcher can tell the client 

261 its window and the client trims history to fit. None until the engine is up.""" 

262 chat_slots: int | None = None 

263 """Batching slots the chat engine serves (its real request concurrency), so a 

264 script driving parallel agents can read the granted shape instead of assuming 

265 the configured one. None until the engine is up.""" 

266 chat_prefill_processed: int | None = None 

267 """Prompt tokens the chat engine has processed for a prefill in flight. A 

268 large model's first agent turn can spend minutes here with nothing streamed; 

269 polling this tells a working engine apart from a hung one. None when idle.""" 

270 chat_prefill_total: int | None = None 

271 """Prompt tokens the in-flight chat prefill will process in total. None when 

272 no prefill is running.""" 

273 embed_token_cap: int | None = None 

274 """Tokens the embedding engine truncates one input to. The chunker bounds its 

275 budget to this, so it is the largest chunk that reaches the index whole. None 

276 when no managed embedder is configured.""" 

277 warnings: list[HealthWarning] = [] 

278 """Degradations that answer correctly but worse, so a client can say so. 

279 

280 Retrieval falling back to vector-only, or an index whose documents predate 

281 the embedder's prefixes, both return results and look healthy.""" 

282 

283 

284class CompactionInfo(BaseModel): 

285 """What one pre-turn compaction folded out of a conversation.""" 

286 

287 summary: str 

288 condensed: int 

289 """Turns folded into the notes.""" 

290 stranded: int 

291 """Turns dropped with no notes; a client must say so rather than hide it.""" 

292 

293 

294class AskResponse(BaseModel): 

295 """Response for /api/ask and /api/chat. 

296 

297 ``sources`` is the full retrieved set; ``cited_sources`` is the subset the answer 

298 actually cited, so a client can tell a grounded answer from an off-corpus one. 

299 """ 

300 

301 answer: str 

302 sources: list[CleanedChunk] 

303 cited_sources: list[CleanedChunk] = Field(default_factory=list) 

304 compaction: CompactionInfo | None = None 

305 """Set when a /api/chat turn compacted its history before answering.""" 

306 retrieval_query: str | None = None 

307 """The standalone rewrite retrieval ran on, when a follow-up was rewritten.""" 

308 dropped_sources: list[CleanedChunk] = Field(default_factory=list) 

309 """Chunks the budget fit shed, so a client can say what was trimmed.""" 

310 

311 

312class SetModelResponse(BaseModel): 

313 """Response for PUT /api/models/{chat|embedding|vision|reranker}. 

314 

315 ``reindex_required`` is ``True`` only when the new embedding model differs from 

316 the model that built the persisted vector store. The chat, vision, and reranker 

317 handlers always return ``False`` because their changes do not invalidate stored 

318 vectors. Mirrors the ``reindex_required`` flag on ``ConfigUpdateResponse``. 

319 """ 

320 

321 model: str 

322 reindex_required: bool = False 

323 warnings: list[str] = [] 

324 

325 

326class ConfigUpdateResponse(BaseModel): 

327 """Response for PATCH /api/config.""" 

328 

329 updated: list[str] 

330 reindex_required: bool 

331 warnings: list[str] = [] 

332 

333 

334class CrawlRequest(BaseModel): 

335 """Request body for /api/crawl. 

336 

337 depth: null / omitted = whole-site unbounded recursion. 0 = single URL 

338 only. Positive int = max depth. max_pages: null / omitted = the protective 

339 safety cap. 0 = explicitly unlimited (the CRAWL_PAGES_UNLIMITED sentinel the 

340 TUI and crawler honor). Positive int = explicit page cap. render_mode: null / 

341 omitted = configured default; "http" is browserless, "browser" runs Chromium 

342 with JavaScript. 

343 """ 

344 

345 url: str 

346 depth: int | None = Field(default=None, ge=0) 

347 max_pages: int | None = Field(default=None, ge=0) 

348 render_mode: CrawlRenderMode | None = Field(default=None) 

349 include_subdomains: bool = Field(default=False) 

350 

351 

352class DocumentInfo(BaseModel): 

353 """A single indexed document in a list response.""" 

354 

355 filename: str 

356 chunk_count: int = 0 

357 ingested_at: str = "" 

358 

359 

360class DocumentListResponse(BaseModel): 

361 """Response for GET /api/documents.""" 

362 

363 documents: list[DocumentInfo] 

364 total: int 

365 limit: int 

366 offset: int 

367 has_more: bool = False 

368 

369 

370class DocumentRemoveResponse(BaseModel): 

371 """Response for POST /api/documents/remove.""" 

372 

373 removed: list[str] = Field( 

374 description="Names removed: indexed sources, files an ingestion failure held out, " 

375 "and registered root labels." 

376 ) 

377 not_found: list[str] = Field(description="Names that matched nothing removable.") 

378 

379 

380class ConfigResponse(BaseModel): 

381 """Response for GET /api/config.""" 

382 

383 model_config = {"extra": "allow"} 

384 

385 

386class ConfigFieldSchema(BaseModel): 

387 """Metadata for one configuration field, so a client can render its control. 

388 

389 Field names match the MCP ``settings_list`` wire shape, which carries the 

390 same metadata for agents. 

391 """ 

392 

393 key: str 

394 type: str 

395 nullable: bool 

396 writable: bool 

397 reindex_required: bool 

398 group: SettingGroup 

399 help: str 

400 choices: list[str] | None 

401 

402 

403class ConfigSchemaResponse(BaseModel): 

404 """Response for GET /api/config/schema.""" 

405 

406 fields: list[ConfigFieldSchema] 

407 

408 

409class ModelsShowResponse(BaseModel): 

410 """Response for POST /api/models/show.""" 

411 

412 model_config = {"extra": "allow"} 

413 

414 

415class CatalogEntryResponse(BaseModel): 

416 """A single model in the catalog browser. 

417 

418 ``fit`` and ``size_variants`` carry server-computed hardware-fit 

419 data so clients (TUI, plugin) can render fit chips and size strips 

420 without probing local memory themselves. ``fit`` is ``None`` when 

421 the row's footprint cannot be assessed against host memory (e.g. 

422 a future cloud-only entry whose weights live off-host). 

423 """ 

424 

425 hf_repo: str 

426 gguf_filename: str 

427 task: ModelTask 

428 display_name: str 

429 param_count: str 

430 size_gb: float 

431 min_ram_gb: float 

432 description: str 

433 quality_tier: str 

434 featured: bool 

435 downloads: int 

436 installed: bool 

437 source: ModelSource 

438 fit: FitLevel | None = None 

439 size_variants: list[SizeVariantInfo] = [] 

440 architecture: str = "" 

441 compat: ModelCompat = ModelCompat.UNKNOWN 

442 safety_stripped: bool = False 

443 provider: str = "" 

444 key_status: KeyStatus | None = None 

445 

446 

447class ModelsCatalogResponse(BaseModel): 

448 """Response for GET /api/models/catalog. 

449 

450 Filters apply before paging. ``next_offset`` names the offset to request 

451 next, None on the last page. ``truncated`` is True when the HuggingFace scan 

452 stopped at its bound with rows left unread, so the listing is cut short. 

453 """ 

454 

455 total: int | None 

456 limit: int 

457 offset: int 

458 models: list[CatalogEntryResponse] 

459 has_more: bool = False 

460 next_offset: int | None 

461 truncated: bool = False 

462 

463 

464class InstalledModelEntry(BaseModel): 

465 """A single installed model.""" 

466 

467 name: str 

468 source: ModelSource 

469 

470 

471class ModelsInstalledResponse(BaseModel): 

472 """Response for GET /api/models/installed.""" 

473 

474 models: list[InstalledModelEntry] 

475 

476 

477class ModelsDeleteResponse(BaseModel): 

478 """Response for DELETE /api/models/{model}.""" 

479 

480 deleted: bool 

481 model: str 

482 freed_gb: float 

483 

484 

485class ExternalModelsResponse(BaseModel): 

486 """Response for GET /api/models/external.""" 

487 

488 models: list[str] 

489 error: str | None = None 

490 

491 

492class SyncSummary(BaseModel): 

493 """Embedded sync result within an add-files response.""" 

494 

495 added: list[str] = [] 

496 updated: list[str] = [] 

497 removed: list[str] = [] 

498 unchanged: int = 0 

499 relocated: list[str] = [] 

500 failed: list[str] = [] 

501 skipped: list[str] = [] 

502 held_out: list[SkippedSource] = [] 

503 truncated: int = 0 

504 index_mismatch: IndexMismatch | None = None 

505 

506 

507class AddSummary(BaseModel): 

508 """Summary returned by the add-files handler.""" 

509 

510 copied: list[str] 

511 errors: list[str] 

512 name_taken: list[str] = [] 

513 """Labels held by a different source; nothing was registered and no sync ran.""" 

514 overlapping: list[str] = [] 

515 """Paths inside or around a registered source; that source covers them in the sync.""" 

516 tracked: list[str] = [] 

517 """Named sources the knowledge base already tracks, so nothing was registered. 

518 

519 These need no action from the caller: the sync in the same request covers them. 

520 """ 

521 sync: SyncSummary | None = None 

522 already_ingesting: list[str] = [] 

523 """Sources another ingest held a lock on, so this run never attempted them. 

524 

525 Distinct from ``skipped``, which means the file was examined and needed no 

526 work. These were not looked at and are worth retrying. Carried on the 

527 terminal event so a client that missed the earlier ``already_ingesting`` 

528 frames can still tell the batch was partial. 

529 """ 

530 

531 

532class WikiCitationRecord(BaseModel): 

533 """A citation record from the store, used in reverse lookup responses.""" 

534 

535 wiki_source: str = "" 

536 wiki_chunk_index: int = 0 

537 citation_key: str = "" 

538 claim_type: str = "fact" 

539 source_filename: str = "" 

540 source_hash: str = "" 

541 page_start: int = 0 

542 page_end: int = 0 

543 line_start: int = 0 

544 line_end: int = 0 

545 excerpt: str = "" 

546 created_at: str = "" 

547 

548 

549class WikiEntityCandidateResponse(BaseModel): 

550 """One NER entity candidate, with the evidence a page would be built from.""" 

551 

552 slug: str 

553 label: str = "" 

554 kind: EntityKind = EntityKind.ENTITY 

555 type_hint: str = "" 

556 mentions: int = 0 

557 sources: list[str] = [] 

558 

559 

560class WikiBuildDryRunResult(BaseModel): 

561 """Entity candidates a build would cover, with no LLM call made.""" 

562 

563 dry_run: bool = True 

564 entities: list[WikiEntityCandidateResponse] = [] 

565 count: int = 0 

566 note: str = "" 

567 

568 

569class WikiPageDetail(BaseModel): 

570 """Full content of a single wiki page, with its parsed frontmatter.""" 

571 

572 slug: str 

573 title: str = "" 

574 content: str = "" 

575 frontmatter: dict[str, Any] = {} 

576 

577 

578class WikiCitationsResult(BaseModel): 

579 """Citations attached to a single wiki page.""" 

580 

581 slug: str 

582 citations: list[WikiCitationRecord] = [] 

583 

584 

585class WikiLintIssueItem(BaseModel): 

586 """A single lint finding on a wiki page.""" 

587 

588 wiki_source: str = "" 

589 issue_type: str = "" 

590 severity: str = "" 

591 message: str = "" 

592 

593 

594class WikiLintResult(BaseModel): 

595 """Result of a wiki lint run, whole-wiki or single-page.""" 

596 

597 issues: list[WikiLintIssueItem] = [] 

598 total: int = 0 

599 errors: int = 0 

600 warnings: int = 0 

601 

602 

603class WikiPruneRecordResponse(BaseModel): 

604 """A single pruning action.""" 

605 

606 wiki_source: str 

607 action: str 

608 reason: str 

609 

610 

611class WikiPruneResult(BaseModel): 

612 """Result of wiki pruning.""" 

613 

614 records: list[WikiPruneRecordResponse] = [] 

615 archived: int = 0 

616 flagged: int = 0 

617 reconciled: int = 0 

618 

619 

620class WikiIndexResult(BaseModel): 

621 """Result of rebuilding the browse index. Costs no LLM call.""" 

622 

623 entries: int = 0 

624 

625 

626class WikiGenerateResult(BaseModel): 

627 """Result of generating one indexed page.""" 

628 

629 slug: str 

630 path: str 

631 

632 

633class WikiWipeResult(BaseModel): 

634 """Result of wiping the wiki. 

635 

636 ``rows_deleted`` is false when the pages went but the store delete failed, 

637 so a client is never told the wiki is gone while its rows still answer. 

638 """ 

639 

640 pages_removed: int = 0 

641 sources_cleared: int = 0 

642 rows_deleted: bool = True 

643 

644 

645class WikiStatusResult(BaseModel): 

646 """Wiki layer status counters.""" 

647 

648 wiki_enabled: bool 

649 summaries: int = 0 

650 drafts: int = 0 

651 pages: int = 0 

652 lint_errors: int = 0 

653 lint_warnings: int = 0 

654 

655 

656class DraftInfoResponse(BaseModel): 

657 """Metadata about a single wiki draft, mirroring ``DraftInfo.to_dict()``. 

658 

659 ``pending_kind`` distinguishes drift drafts (``None``) from 

660 batched-generation markers (``"parse"``, ``"collision"``). 

661 """ 

662 

663 slug: str 

664 path: str 

665 drift_ratio: float | None = None 

666 faithfulness_score: float | None = None 

667 bad_title: bool = False 

668 published_path: str | None = None 

669 published_exists: bool = False 

670 mtime: float = 0.0 

671 pending_kind: str | None = None 

672 

673 

674class WikiDraftDiffResponse(BaseModel): 

675 """Unified diff of a draft against its published counterpart.""" 

676 

677 slug: str 

678 diff: str 

679 

680 

681class WikiDraftAcceptResponse(BaseModel): 

682 """Outcome of accepting a draft: where it landed and how many chunks reindexed. 

683 

684 ``slug`` is the slug where the content was published. 

685 ``requested_slug`` is the slug the client asked to accept. The two 

686 differ for PENDING-COLLISION drafts, where the request slug carries 

687 a ``-collision-<hash>`` suffix that is stripped on publish. 

688 """ 

689 

690 slug: str 

691 requested_slug: str 

692 moved_to: str 

693 reindexed_chunks: int 

694 

695 

696class WikiDraftRejectResponse(BaseModel): 

697 """Outcome of rejecting a draft.""" 

698 

699 slug: str 

700 

701 

702class RememberRequest(BaseModel): 

703 """Request body for ``POST /api/memories``.""" 

704 

705 text: str 

706 kind: MemoryKind = MemoryKind.FACT 

707 shared: bool = False 

708 

709 

710class RememberResponse(BaseModel): 

711 """Outcome of storing a memory.""" 

712 

713 id: str 

714 kind: MemoryKind 

715 

716 

717class MemoryItem(BaseModel): 

718 """A single stored memory in a list response.""" 

719 

720 id: str 

721 kind: MemoryKind 

722 shared: bool 

723 text: str 

724 

725 

726class MemoryListResponse(BaseModel): 

727 """Body for ``GET /api/memories``.""" 

728 

729 memories: list[MemoryItem] 

730 

731 

732class MemorySharedRequest(BaseModel): 

733 """Request body for ``PATCH /api/memories/{memory_id}``.""" 

734 

735 shared: bool 

736 

737 

738class MemoryFlagsResponse(BaseModel): 

739 """Outcome of a flag update; ``updated`` is False when the id was unknown.""" 

740 

741 id: str 

742 updated: bool 

743 

744 

745class MemoryRemoveResponse(BaseModel): 

746 """Outcome of deleting a memory; ``deleted`` is False when the id was unknown.""" 

747 

748 id: str 

749 deleted: bool 

750 

751 

752class MemoryExtractedItem(BaseModel): 

753 """A single memory created by auto-extraction during a chat turn.""" 

754 

755 id: str 

756 kind: MemoryKind 

757 text: str 

758 

759 

760class MemoryExtractedEvent(BaseModel): 

761 """``memory_extracted`` SSE payload: how many memories a turn auto-saved. 

762 

763 Emitted on the chat stream after ``done`` when auto-extraction is on and the 

764 turn produced at least one memory, so a REST client (the Obsidian plugin) can 

765 toast the count and refresh its memories view without a separate fetch. 

766 """ 

767 

768 count: int 

769 items: list[MemoryExtractedItem] 

770 

771 

772class GpuInfoResponse(BaseModel): 

773 """One GPU as returned by GET /api/gpus and embedded in PlacementResponse.""" 

774 

775 index: int 

776 backend: str 

777 label: str 

778 name: str 

779 total_bytes: int 

780 free_bytes: int 

781 

782 

783class GpusResponse(BaseModel): 

784 """GET /api/gpus envelope: detected GPUs plus the host-level util notice.""" 

785 

786 gpus: list[GpuInfoResponse] 

787 notice: str | None = None 

788 

789 

790class RolePlacementResponse(BaseModel): 

791 """Where one role's model is placed in the resolved plan.""" 

792 

793 role: WorkerRole 

794 model: str 

795 devices: list[int] 

796 tensor_split: list[int] | None 

797 replicas: int 

798 

799 

800class SkippedRoleResponse(BaseModel): 

801 """A configured role left unplaced because its model isn't downloaded.""" 

802 

803 role: WorkerRole 

804 model: str 

805 

806 

807class PlacementResponse(BaseModel): 

808 """Response for placement read, preview, set, and clear routes.""" 

809 

810 gpus: list[GpuInfoResponse] 

811 roles: list[RolePlacementResponse] 

812 unplaceable: list[str] 

813 manual: bool 

814 spec_json: str | None 

815 skipped_not_installed: list[SkippedRoleResponse] = [] 

816 co_tenants: list[str] = [] 

817 notice: str | None = None 

818 rejected_spec_json: str | None = None 

819 # The backend the engine selected, reported rather than inferred. ``gpus`` 

820 # being empty does not mean ``cpu``: a host whose device probe never answered 

821 # reports ``unknown``, so a client never mislabels a GPU box as a CPU one. 

822 engine_backend: EngineBackend = EngineBackend.UNKNOWN 

823 

824 @classmethod 

825 def from_view(cls, view: PlacementView) -> PlacementResponse: 

826 """The canonical serialized placement view, shared by the HTTP, MCP, and CLI surfaces.""" 

827 return cls( 

828 gpus=[GpuInfoResponse(**vars(g)) for g in view.gpus], 

829 roles=[ 

830 RolePlacementResponse( 

831 role=r.role, 

832 model=r.model, 

833 devices=list(r.devices), 

834 tensor_split=list(r.tensor_split) if r.tensor_split else None, 

835 replicas=r.replicas, 

836 ) 

837 for r in view.roles 

838 ], 

839 unplaceable=[r.value for r in view.unplaceable], 

840 manual=view.manual, 

841 spec_json=view.spec_json, 

842 skipped_not_installed=[ 

843 SkippedRoleResponse(role=s.role, model=s.model) for s in view.skipped_not_installed 

844 ], 

845 co_tenants=[r.value for r in view.co_tenants], 

846 rejected_spec_json=view.rejected_spec_json, 

847 engine_backend=view.engine_backend, 

848 ) 

849 

850 

851class PlacementSpecBody(BaseModel): 

852 """Request body for placement routes that accept a manual spec.""" 

853 

854 spec: dict[str, dict[str, object]] | None = None 

855 

856 

857class SessionMetaItem(BaseModel): 

858 """A session's metadata in a list or detail response.""" 

859 

860 id: str 

861 title: str 

862 created_at: str 

863 updated_at: str 

864 model_ref: str 

865 scope: str 

866 message_count: int 

867 origin: str = "tui" 

868 """Owning surface. tui/http/cli are one domain and append freely to each 

869 other's sessions; appends across the human/agent (mcp) boundary are 409.""" 

870 forked_from: str = "" 

871 """Id of the session this one was forked from; empty when it is not a fork.""" 

872 

873 

874class SessionListResponse(BaseModel): 

875 """Body for ``GET /api/sessions``.""" 

876 

877 sessions: list[SessionMetaItem] 

878 

879 

880class SessionMessageItem(BaseModel): 

881 """One message in a session transcript.""" 

882 

883 role: MessageRole 

884 content: str 

885 sources: list[str] 

886 ts: str 

887 

888 

889class SessionDetailResponse(BaseModel): 

890 """Body for ``GET /api/sessions/{session_id}``: metadata plus transcript. 

891 

892 ``summary`` carries what compaction folded the oldest turns into (empty when 

893 a conversation has not been compacted). A client that resumes and continues 

894 the conversation needs it: without it, it rebuilds history from the raw 

895 transcript, re-sending turns the summary had already condensed and risking 

896 the context overflow compaction exists to prevent. 

897 """ 

898 

899 meta: SessionMetaItem 

900 messages: list[SessionMessageItem] 

901 summary: str = "" 

902 

903 

904class SessionCreateRequest(BaseModel): 

905 """Request body for ``POST /api/sessions``.""" 

906 

907 model_ref: str 

908 scope: str 

909 

910 

911class SessionMessageCreateRequest(BaseModel): 

912 """Request body for ``POST /api/sessions/{session_id}/messages``.""" 

913 

914 role: MessageRole 

915 content: str 

916 sources: list[str] = [] 

917 

918 

919class SessionForkRequest(BaseModel): 

920 """Request body for ``POST /api/sessions/{session_id}/fork``. 

921 

922 ``message_count`` is the number of leading messages to copy; null copies all. 

923 """ 

924 

925 message_count: int | None = None 

926 

927 @field_validator("message_count", mode="before") 

928 @classmethod 

929 def _check_message_count(cls, value: object) -> object: 

930 """Refuse ``true``, ``"2"`` and ``1.0`` rather than coerce them into a count.""" 

931 # isinstance: the raw JSON value, before pydantic's lax coercion runs. 

932 if value is None or (isinstance(value, int) and not isinstance(value, bool)): 

933 return value 

934 raise ValueError("message_count must be a whole number or null") 

935 

936 

937class SessionSummaryRequest(BaseModel): 

938 """Request body for ``PUT /api/sessions/{session_id}/summary``.""" 

939 

940 summary: str 

941 

942 

943class SessionRenameRequest(BaseModel): 

944 """Request body for ``PATCH /api/sessions/{session_id}``.""" 

945 

946 title: str 

947 

948 

949class SessionRenameResponse(BaseModel): 

950 """Outcome of a rename.""" 

951 

952 id: str 

953 title: str 

954 

955 

956class SessionDeleteResponse(BaseModel): 

957 """Outcome of a delete.""" 

958 

959 id: str 

960 deleted: bool 

961 

962 

963class AgentClientDetection(BaseModel): 

964 """Whether one agent client's CLI is installed on the machine lilbee runs on.""" 

965 

966 client: AgentClient 

967 cli_detected: bool 

968 cli_path: str | None 

969 

970 

971class AgentConfigIndexResponse(BaseModel): 

972 """Response for ``GET /api/agent-config``: every client lilbee can configure.""" 

973 

974 clients: list[AgentClientDetection] 

975 

976 @classmethod 

977 def from_detections(cls, detections: list[ClientDetection]) -> AgentConfigIndexResponse: 

978 """Serialize the probe results one entry per supported client.""" 

979 return cls( 

980 clients=[ 

981 AgentClientDetection( 

982 client=found.client, 

983 cli_detected=found.cli_detected, 

984 cli_path=found.cli_path, 

985 ) 

986 for found in detections 

987 ] 

988 ) 

989 

990 

991class AgentConfigResponse(BaseModel): 

992 """Response for ``GET /api/agent-config/{client}``: one client's live config. 

993 

994 ``config`` carries the block for a JSON client, ``content`` the rendered text 

995 for a YAML one. ``stdio_config`` is the alternative block for a client that 

996 can also run lilbee as a subprocess instead of calling this server. 

997 """ 

998 

999 client: AgentClient 

1000 format: ConfigFormat 

1001 surfaces: list[AgentSurface] 

1002 config: dict[str, Any] | None = None 

1003 content: str | None = None 

1004 stdio_config: dict[str, Any] | None = None 

1005 

1006 @classmethod 

1007 def from_document(cls, document: AgentConfigDocument) -> AgentConfigResponse: 

1008 """The canonical serialized config document, shared by the HTTP and CLI surfaces.""" 

1009 return cls( 

1010 client=document.client, 

1011 format=document.format, 

1012 surfaces=list(document.surfaces), 

1013 config=document.config, 

1014 content=document.content, 

1015 stdio_config=document.stdio_config, 

1016 )