Coverage for src/lilbee/server/models.py: 100%
486 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
1"""Request and response models for the lilbee HTTP API.
3Typed pydantic models so Litestar's OpenAPI schema has field-level detail.
4"""
6from __future__ import annotations
8from typing import TYPE_CHECKING, Any, Literal
10from pydantic import BaseModel, Field, field_validator
12from lilbee.app.agent_configs.document import AgentClient, AgentSurface, ConfigFormat
13from lilbee.app.settings_map import SettingGroup
14from lilbee.catalog.types import KeyStatus, ModelCompat, ModelSource, ModelTask
15from lilbee.core.config.enums import CrawlRenderMode, KvCacheType
16from lilbee.core.health_warnings import HealthWarning
17from lilbee.data.store import ChunkType, IndexMismatch, MemoryKind, scope_to_chunk_type
18from lilbee.data.types import SkippedSource
19from lilbee.providers.roles import EngineBackend, WorkerRole
20from lilbee.runtime.hardware import FitLevel, SizeVariantInfo
21from lilbee.sessions import MessageRole
22from lilbee.wiki.entity_extractor import EntityKind
24if TYPE_CHECKING:
25 from lilbee.app.agent_configs.detect import ClientDetection
26 from lilbee.app.agent_configs.document import AgentConfigDocument
27 from lilbee.app.placement import PlacementView
30def decode_chunk_type(value: str | None) -> ChunkType | None:
31 """Decode a ``chunk_type`` string into a ``ChunkType`` at the HTTP boundary.
33 Delegates to the canonical :func:`scope_to_chunk_type` so query-param and
34 request-body routes share one decoder: only ``"raw"`` or ``"wiki"`` filter
35 the pool; everything else (including ``None`` and the UI-side ``"both"``)
36 means no filter. Any other string raises ``ValueError`` with boundary-
37 friendly guidance.
38 """
39 try:
40 return scope_to_chunk_type(value)
41 except ValueError as exc:
42 raise ValueError(
43 f"chunk_type must be one of 'raw', 'wiki', 'both', or omitted; got {value!r}"
44 ) from exc
47class AskRequest(BaseModel):
48 """Request body for /api/ask."""
50 question: str
51 top_k: int = Field(default=0, ge=0, le=100)
52 options: dict[str, Any] | None = None
53 chunk_type: ChunkType | None = None
55 @field_validator("chunk_type", mode="before")
56 @classmethod
57 def _check_chunk_type(cls, v: str | None) -> ChunkType | None:
58 return decode_chunk_type(v)
61class ChatRequest(BaseModel):
62 """Request body for /api/chat."""
64 question: str
65 history: list[ChatMessage] = []
66 # None (unspecified) grounds with the configured top_k; an explicit 0 is a
67 # pure-LLM call that skips retrieval entirely.
68 top_k: int | None = Field(default=None, ge=0, le=100)
69 options: dict[str, Any] | None = None
70 chunk_type: ChunkType | None = None
71 summary: str = ""
72 """Carry-forward notes from earlier compactions, folded into the prompt."""
73 session_id: str | None = None
74 """Session that receives the new summary when this turn compacts."""
76 @field_validator("chunk_type", mode="before")
77 @classmethod
78 def _check_chunk_type(cls, v: str | None) -> ChunkType | None:
79 return decode_chunk_type(v)
82class SyncRequest(BaseModel):
83 """Request body for /api/sync.
85 ``force_rebuild`` triggers a full drop-and-reingest equivalent to ``lilbee rebuild``.
86 Use it to recover from an embedding-model switch (when the store refuses search
87 or ingest because ``cfg.embedding_model`` no longer matches the persisted vectors).
88 ``retry_skipped`` is the lighter recovery: it clears the markers for files that
89 failed a previous sync (Tesseract timeout, decode failure, no usable text) so this
90 sync attempts them again, without dropping the existing store. The default is an
91 incremental sync.
92 ``prune_ignored`` drops sources a ``.lilbeeignore`` now excludes. Off by default:
93 the patterns govern what sync takes in, not what a past sync already indexed.
94 """
96 enable_ocr: bool | None = None
97 ocr_timeout: float | None = None
98 force_rebuild: bool = False
99 retry_skipped: bool = False
100 prune_ignored: bool = False
103class AddRequest(BaseModel):
104 """Request body for /api/add."""
106 paths: list[str]
107 force: bool = False
108 enable_ocr: bool | None = None
109 ocr_timeout: float | None = None
112class SetModelRequest(BaseModel):
113 """Request body for /api/models/chat."""
115 model: str
118class SourceContentResponse(BaseModel):
119 """JSON body for ``GET /api/source`` (``raw=0``); empty ``markdown`` for binary types."""
121 markdown: str
122 content_type: str
123 title: str | None = None
126class ChatMessage(BaseModel):
127 """A single message in a chat conversation."""
129 role: Literal["user", "assistant"]
130 content: str
133class CleanedChunk(BaseModel):
134 """A search result chunk with vector stripped and distance renamed."""
136 source: str
137 content_type: str
138 chunk: str
139 distance: float | None = None
140 relevance_score: float | None = None
141 rerank_score: float | None = None
142 # Canonical [0, 1] relevance from retrieval fusion; the ranking signal
143 # HTTP clients should sort and threshold on (relevance_score is legacy).
144 score: float | None = None
145 page_start: int = 0
146 page_end: int = 0
147 line_start: int = 0
148 line_end: int = 0
149 chunk_index: int = 0
150 # Vault-relative path when ``cfg.vault_base`` is set and the source file
151 # lives inside the vault. Absent when the server is running headless or
152 # the source isn't resolvable as a vault file. Clients use this to open
153 # the source in a native editor instead of fetching ``/api/source``.
154 vault_path: str | None = None
155 # Set only on recalled-memory sources (``source`` is ``memory:<id>``), so
156 # clients can mark them as memory and link them to ``GET /api/memories``.
157 memory_id: str | None = None
160class StatusSourceInfo(BaseModel):
161 """A single indexed source in a status response."""
163 filename: str
164 file_hash: str
165 chunk_count: int
166 ingested_at: str
169class StatusConfigInfo(BaseModel):
170 """Configuration section of a status response.
172 Exposes all four role-bound model fields so plugins/TUI can show
173 what's active per role without a second round trip.
174 """
176 documents_dir: str
177 data_dir: str
178 chat_model: str
179 embedding_model: str
180 vision_model: str = ""
181 reranker_model: str = ""
182 enable_ocr: bool | None = None
183 num_ctx: int | None = None
184 num_ctx_max: int | None = None
185 chat_n_ctx_target: int | None = None
186 flash_attention: bool | None = None
187 kv_cache_type: KvCacheType | None = None
188 n_gpu_layers: int | None = None
189 cpu_moe: bool | None = None
190 n_cpu_moe: int | None = None
191 main_gpu: int | None = None
192 gpu_devices: str | None = None
195class StatusEntityInfo(BaseModel):
196 """Entity-extraction section of a status response (present when enabled)."""
198 types: list[str]
199 rows: int
202class StatusIndexInfo(BaseModel):
203 """The embedder that built the persisted index."""
205 embedding_model: str
206 embedding_dim: int
209class StatusResponse(BaseModel):
210 """Response for GET /api/status."""
212 command: str = "status"
213 config: StatusConfigInfo
214 sources: list[StatusSourceInfo]
215 document_count: int
216 total_chunks: int
217 index: StatusIndexInfo | None = None
218 """The embedder that built the index; absent before the first sync. Compare it
219 with ``config.embedding_model`` to tell a stale index before a search refuses it."""
220 entities: StatusEntityInfo | None = None
221 skipped: list[SkippedSource] = []
222 """Files a skip marker holds out of the index, capped; ``skipped_total`` is the real count."""
223 skipped_total: int = 0
224 ocr_warning: str | None = None
225 """Set when a vision model is configured but ``enable_ocr`` is false."""
226 ocr_note: str | None = None
227 """Which OCR engine runs for scanned pages; None when OCR is off."""
230class ShutdownResponse(BaseModel):
231 """Response for /api/shutdown."""
233 status: Literal["shutting_down"]
236class HealthResponse(BaseModel):
237 """Response for /api/health."""
239 status: str
240 version: str
241 chat_ready: bool = False
242 """True once the chat engine is loaded and ready to serve a first token.
244 A launcher polls this to wait out the cold model load before handing off to
245 a client, so the client never lands on an apparently-dead stream.
246 """
247 chat_status: Literal["ready", "loading", "not_started", "error"] = "not_started"
248 """Finer-grained chat readiness than the ``chat_ready`` bool.
250 Lets a polling client tell a fleet that is still loading (wait) apart from one
251 that never started warming (``not_started`` -- no chat model resolved / planned,
252 so it will not come up on its own) or failed (``error``). Without this a bare
253 ``chat_ready:false`` reads the same for "loading" and "hung", which looked like a
254 silent hang on a fresh box with no chat model installed."""
255 chat_error: str | None = None
256 """The reason the chat engine failed to come up when ``chat_status`` is
257 ``error`` (e.g. a wedged GPU device probe), so a polling client can report
258 the cause instead of retrying forever."""
259 chat_ctx: int | None = None
260 """Per-slot context the chat engine serves, so a launcher can tell the client
261 its window and the client trims history to fit. None until the engine is up."""
262 chat_slots: int | None = None
263 """Batching slots the chat engine serves (its real request concurrency), so a
264 script driving parallel agents can read the granted shape instead of assuming
265 the configured one. None until the engine is up."""
266 chat_prefill_processed: int | None = None
267 """Prompt tokens the chat engine has processed for a prefill in flight. A
268 large model's first agent turn can spend minutes here with nothing streamed;
269 polling this tells a working engine apart from a hung one. None when idle."""
270 chat_prefill_total: int | None = None
271 """Prompt tokens the in-flight chat prefill will process in total. None when
272 no prefill is running."""
273 embed_token_cap: int | None = None
274 """Tokens the embedding engine truncates one input to. The chunker bounds its
275 budget to this, so it is the largest chunk that reaches the index whole. None
276 when no managed embedder is configured."""
277 warnings: list[HealthWarning] = []
278 """Degradations that answer correctly but worse, so a client can say so.
280 Retrieval falling back to vector-only, or an index whose documents predate
281 the embedder's prefixes, both return results and look healthy."""
284class CompactionInfo(BaseModel):
285 """What one pre-turn compaction folded out of a conversation."""
287 summary: str
288 condensed: int
289 """Turns folded into the notes."""
290 stranded: int
291 """Turns dropped with no notes; a client must say so rather than hide it."""
294class AskResponse(BaseModel):
295 """Response for /api/ask and /api/chat.
297 ``sources`` is the full retrieved set; ``cited_sources`` is the subset the answer
298 actually cited, so a client can tell a grounded answer from an off-corpus one.
299 """
301 answer: str
302 sources: list[CleanedChunk]
303 cited_sources: list[CleanedChunk] = Field(default_factory=list)
304 compaction: CompactionInfo | None = None
305 """Set when a /api/chat turn compacted its history before answering."""
306 retrieval_query: str | None = None
307 """The standalone rewrite retrieval ran on, when a follow-up was rewritten."""
308 dropped_sources: list[CleanedChunk] = Field(default_factory=list)
309 """Chunks the budget fit shed, so a client can say what was trimmed."""
312class SetModelResponse(BaseModel):
313 """Response for PUT /api/models/{chat|embedding|vision|reranker}.
315 ``reindex_required`` is ``True`` only when the new embedding model differs from
316 the model that built the persisted vector store. The chat, vision, and reranker
317 handlers always return ``False`` because their changes do not invalidate stored
318 vectors. Mirrors the ``reindex_required`` flag on ``ConfigUpdateResponse``.
319 """
321 model: str
322 reindex_required: bool = False
323 warnings: list[str] = []
326class ConfigUpdateResponse(BaseModel):
327 """Response for PATCH /api/config."""
329 updated: list[str]
330 reindex_required: bool
331 warnings: list[str] = []
334class CrawlRequest(BaseModel):
335 """Request body for /api/crawl.
337 depth: null / omitted = whole-site unbounded recursion. 0 = single URL
338 only. Positive int = max depth. max_pages: null / omitted = the protective
339 safety cap. 0 = explicitly unlimited (the CRAWL_PAGES_UNLIMITED sentinel the
340 TUI and crawler honor). Positive int = explicit page cap. render_mode: null /
341 omitted = configured default; "http" is browserless, "browser" runs Chromium
342 with JavaScript.
343 """
345 url: str
346 depth: int | None = Field(default=None, ge=0)
347 max_pages: int | None = Field(default=None, ge=0)
348 render_mode: CrawlRenderMode | None = Field(default=None)
349 include_subdomains: bool = Field(default=False)
352class DocumentInfo(BaseModel):
353 """A single indexed document in a list response."""
355 filename: str
356 chunk_count: int = 0
357 ingested_at: str = ""
360class DocumentListResponse(BaseModel):
361 """Response for GET /api/documents."""
363 documents: list[DocumentInfo]
364 total: int
365 limit: int
366 offset: int
367 has_more: bool = False
370class DocumentRemoveResponse(BaseModel):
371 """Response for POST /api/documents/remove."""
373 removed: list[str] = Field(
374 description="Names removed: indexed sources, files an ingestion failure held out, "
375 "and registered root labels."
376 )
377 not_found: list[str] = Field(description="Names that matched nothing removable.")
380class ConfigResponse(BaseModel):
381 """Response for GET /api/config."""
383 model_config = {"extra": "allow"}
386class ConfigFieldSchema(BaseModel):
387 """Metadata for one configuration field, so a client can render its control.
389 Field names match the MCP ``settings_list`` wire shape, which carries the
390 same metadata for agents.
391 """
393 key: str
394 type: str
395 nullable: bool
396 writable: bool
397 reindex_required: bool
398 group: SettingGroup
399 help: str
400 choices: list[str] | None
403class ConfigSchemaResponse(BaseModel):
404 """Response for GET /api/config/schema."""
406 fields: list[ConfigFieldSchema]
409class ModelsShowResponse(BaseModel):
410 """Response for POST /api/models/show."""
412 model_config = {"extra": "allow"}
415class CatalogEntryResponse(BaseModel):
416 """A single model in the catalog browser.
418 ``fit`` and ``size_variants`` carry server-computed hardware-fit
419 data so clients (TUI, plugin) can render fit chips and size strips
420 without probing local memory themselves. ``fit`` is ``None`` when
421 the row's footprint cannot be assessed against host memory (e.g.
422 a future cloud-only entry whose weights live off-host).
423 """
425 hf_repo: str
426 gguf_filename: str
427 task: ModelTask
428 display_name: str
429 param_count: str
430 size_gb: float
431 min_ram_gb: float
432 description: str
433 quality_tier: str
434 featured: bool
435 downloads: int
436 installed: bool
437 source: ModelSource
438 fit: FitLevel | None = None
439 size_variants: list[SizeVariantInfo] = []
440 architecture: str = ""
441 compat: ModelCompat = ModelCompat.UNKNOWN
442 safety_stripped: bool = False
443 provider: str = ""
444 key_status: KeyStatus | None = None
447class ModelsCatalogResponse(BaseModel):
448 """Response for GET /api/models/catalog.
450 Filters apply before paging. ``next_offset`` names the offset to request
451 next, None on the last page. ``truncated`` is True when the HuggingFace scan
452 stopped at its bound with rows left unread, so the listing is cut short.
453 """
455 total: int | None
456 limit: int
457 offset: int
458 models: list[CatalogEntryResponse]
459 has_more: bool = False
460 next_offset: int | None
461 truncated: bool = False
464class InstalledModelEntry(BaseModel):
465 """A single installed model."""
467 name: str
468 source: ModelSource
471class ModelsInstalledResponse(BaseModel):
472 """Response for GET /api/models/installed."""
474 models: list[InstalledModelEntry]
477class ModelsDeleteResponse(BaseModel):
478 """Response for DELETE /api/models/{model}."""
480 deleted: bool
481 model: str
482 freed_gb: float
485class ExternalModelsResponse(BaseModel):
486 """Response for GET /api/models/external."""
488 models: list[str]
489 error: str | None = None
492class SyncSummary(BaseModel):
493 """Embedded sync result within an add-files response."""
495 added: list[str] = []
496 updated: list[str] = []
497 removed: list[str] = []
498 unchanged: int = 0
499 relocated: list[str] = []
500 failed: list[str] = []
501 skipped: list[str] = []
502 held_out: list[SkippedSource] = []
503 truncated: int = 0
504 index_mismatch: IndexMismatch | None = None
507class AddSummary(BaseModel):
508 """Summary returned by the add-files handler."""
510 copied: list[str]
511 errors: list[str]
512 name_taken: list[str] = []
513 """Labels held by a different source; nothing was registered and no sync ran."""
514 overlapping: list[str] = []
515 """Paths inside or around a registered source; that source covers them in the sync."""
516 tracked: list[str] = []
517 """Named sources the knowledge base already tracks, so nothing was registered.
519 These need no action from the caller: the sync in the same request covers them.
520 """
521 sync: SyncSummary | None = None
522 already_ingesting: list[str] = []
523 """Sources another ingest held a lock on, so this run never attempted them.
525 Distinct from ``skipped``, which means the file was examined and needed no
526 work. These were not looked at and are worth retrying. Carried on the
527 terminal event so a client that missed the earlier ``already_ingesting``
528 frames can still tell the batch was partial.
529 """
532class WikiCitationRecord(BaseModel):
533 """A citation record from the store, used in reverse lookup responses."""
535 wiki_source: str = ""
536 wiki_chunk_index: int = 0
537 citation_key: str = ""
538 claim_type: str = "fact"
539 source_filename: str = ""
540 source_hash: str = ""
541 page_start: int = 0
542 page_end: int = 0
543 line_start: int = 0
544 line_end: int = 0
545 excerpt: str = ""
546 created_at: str = ""
549class WikiEntityCandidateResponse(BaseModel):
550 """One NER entity candidate, with the evidence a page would be built from."""
552 slug: str
553 label: str = ""
554 kind: EntityKind = EntityKind.ENTITY
555 type_hint: str = ""
556 mentions: int = 0
557 sources: list[str] = []
560class WikiBuildDryRunResult(BaseModel):
561 """Entity candidates a build would cover, with no LLM call made."""
563 dry_run: bool = True
564 entities: list[WikiEntityCandidateResponse] = []
565 count: int = 0
566 note: str = ""
569class WikiPageDetail(BaseModel):
570 """Full content of a single wiki page, with its parsed frontmatter."""
572 slug: str
573 title: str = ""
574 content: str = ""
575 frontmatter: dict[str, Any] = {}
578class WikiCitationsResult(BaseModel):
579 """Citations attached to a single wiki page."""
581 slug: str
582 citations: list[WikiCitationRecord] = []
585class WikiLintIssueItem(BaseModel):
586 """A single lint finding on a wiki page."""
588 wiki_source: str = ""
589 issue_type: str = ""
590 severity: str = ""
591 message: str = ""
594class WikiLintResult(BaseModel):
595 """Result of a wiki lint run, whole-wiki or single-page."""
597 issues: list[WikiLintIssueItem] = []
598 total: int = 0
599 errors: int = 0
600 warnings: int = 0
603class WikiPruneRecordResponse(BaseModel):
604 """A single pruning action."""
606 wiki_source: str
607 action: str
608 reason: str
611class WikiPruneResult(BaseModel):
612 """Result of wiki pruning."""
614 records: list[WikiPruneRecordResponse] = []
615 archived: int = 0
616 flagged: int = 0
617 reconciled: int = 0
620class WikiIndexResult(BaseModel):
621 """Result of rebuilding the browse index. Costs no LLM call."""
623 entries: int = 0
626class WikiGenerateResult(BaseModel):
627 """Result of generating one indexed page."""
629 slug: str
630 path: str
633class WikiWipeResult(BaseModel):
634 """Result of wiping the wiki.
636 ``rows_deleted`` is false when the pages went but the store delete failed,
637 so a client is never told the wiki is gone while its rows still answer.
638 """
640 pages_removed: int = 0
641 sources_cleared: int = 0
642 rows_deleted: bool = True
645class WikiStatusResult(BaseModel):
646 """Wiki layer status counters."""
648 wiki_enabled: bool
649 summaries: int = 0
650 drafts: int = 0
651 pages: int = 0
652 lint_errors: int = 0
653 lint_warnings: int = 0
656class DraftInfoResponse(BaseModel):
657 """Metadata about a single wiki draft, mirroring ``DraftInfo.to_dict()``.
659 ``pending_kind`` distinguishes drift drafts (``None``) from
660 batched-generation markers (``"parse"``, ``"collision"``).
661 """
663 slug: str
664 path: str
665 drift_ratio: float | None = None
666 faithfulness_score: float | None = None
667 bad_title: bool = False
668 published_path: str | None = None
669 published_exists: bool = False
670 mtime: float = 0.0
671 pending_kind: str | None = None
674class WikiDraftDiffResponse(BaseModel):
675 """Unified diff of a draft against its published counterpart."""
677 slug: str
678 diff: str
681class WikiDraftAcceptResponse(BaseModel):
682 """Outcome of accepting a draft: where it landed and how many chunks reindexed.
684 ``slug`` is the slug where the content was published.
685 ``requested_slug`` is the slug the client asked to accept. The two
686 differ for PENDING-COLLISION drafts, where the request slug carries
687 a ``-collision-<hash>`` suffix that is stripped on publish.
688 """
690 slug: str
691 requested_slug: str
692 moved_to: str
693 reindexed_chunks: int
696class WikiDraftRejectResponse(BaseModel):
697 """Outcome of rejecting a draft."""
699 slug: str
702class RememberRequest(BaseModel):
703 """Request body for ``POST /api/memories``."""
705 text: str
706 kind: MemoryKind = MemoryKind.FACT
707 shared: bool = False
710class RememberResponse(BaseModel):
711 """Outcome of storing a memory."""
713 id: str
714 kind: MemoryKind
717class MemoryItem(BaseModel):
718 """A single stored memory in a list response."""
720 id: str
721 kind: MemoryKind
722 shared: bool
723 text: str
726class MemoryListResponse(BaseModel):
727 """Body for ``GET /api/memories``."""
729 memories: list[MemoryItem]
732class MemorySharedRequest(BaseModel):
733 """Request body for ``PATCH /api/memories/{memory_id}``."""
735 shared: bool
738class MemoryFlagsResponse(BaseModel):
739 """Outcome of a flag update; ``updated`` is False when the id was unknown."""
741 id: str
742 updated: bool
745class MemoryRemoveResponse(BaseModel):
746 """Outcome of deleting a memory; ``deleted`` is False when the id was unknown."""
748 id: str
749 deleted: bool
752class MemoryExtractedItem(BaseModel):
753 """A single memory created by auto-extraction during a chat turn."""
755 id: str
756 kind: MemoryKind
757 text: str
760class MemoryExtractedEvent(BaseModel):
761 """``memory_extracted`` SSE payload: how many memories a turn auto-saved.
763 Emitted on the chat stream after ``done`` when auto-extraction is on and the
764 turn produced at least one memory, so a REST client (the Obsidian plugin) can
765 toast the count and refresh its memories view without a separate fetch.
766 """
768 count: int
769 items: list[MemoryExtractedItem]
772class GpuInfoResponse(BaseModel):
773 """One GPU as returned by GET /api/gpus and embedded in PlacementResponse."""
775 index: int
776 backend: str
777 label: str
778 name: str
779 total_bytes: int
780 free_bytes: int
783class GpusResponse(BaseModel):
784 """GET /api/gpus envelope: detected GPUs plus the host-level util notice."""
786 gpus: list[GpuInfoResponse]
787 notice: str | None = None
790class RolePlacementResponse(BaseModel):
791 """Where one role's model is placed in the resolved plan."""
793 role: WorkerRole
794 model: str
795 devices: list[int]
796 tensor_split: list[int] | None
797 replicas: int
800class SkippedRoleResponse(BaseModel):
801 """A configured role left unplaced because its model isn't downloaded."""
803 role: WorkerRole
804 model: str
807class PlacementResponse(BaseModel):
808 """Response for placement read, preview, set, and clear routes."""
810 gpus: list[GpuInfoResponse]
811 roles: list[RolePlacementResponse]
812 unplaceable: list[str]
813 manual: bool
814 spec_json: str | None
815 skipped_not_installed: list[SkippedRoleResponse] = []
816 co_tenants: list[str] = []
817 notice: str | None = None
818 rejected_spec_json: str | None = None
819 # The backend the engine selected, reported rather than inferred. ``gpus``
820 # being empty does not mean ``cpu``: a host whose device probe never answered
821 # reports ``unknown``, so a client never mislabels a GPU box as a CPU one.
822 engine_backend: EngineBackend = EngineBackend.UNKNOWN
824 @classmethod
825 def from_view(cls, view: PlacementView) -> PlacementResponse:
826 """The canonical serialized placement view, shared by the HTTP, MCP, and CLI surfaces."""
827 return cls(
828 gpus=[GpuInfoResponse(**vars(g)) for g in view.gpus],
829 roles=[
830 RolePlacementResponse(
831 role=r.role,
832 model=r.model,
833 devices=list(r.devices),
834 tensor_split=list(r.tensor_split) if r.tensor_split else None,
835 replicas=r.replicas,
836 )
837 for r in view.roles
838 ],
839 unplaceable=[r.value for r in view.unplaceable],
840 manual=view.manual,
841 spec_json=view.spec_json,
842 skipped_not_installed=[
843 SkippedRoleResponse(role=s.role, model=s.model) for s in view.skipped_not_installed
844 ],
845 co_tenants=[r.value for r in view.co_tenants],
846 rejected_spec_json=view.rejected_spec_json,
847 engine_backend=view.engine_backend,
848 )
851class PlacementSpecBody(BaseModel):
852 """Request body for placement routes that accept a manual spec."""
854 spec: dict[str, dict[str, object]] | None = None
857class SessionMetaItem(BaseModel):
858 """A session's metadata in a list or detail response."""
860 id: str
861 title: str
862 created_at: str
863 updated_at: str
864 model_ref: str
865 scope: str
866 message_count: int
867 origin: str = "tui"
868 """Owning surface. tui/http/cli are one domain and append freely to each
869 other's sessions; appends across the human/agent (mcp) boundary are 409."""
870 forked_from: str = ""
871 """Id of the session this one was forked from; empty when it is not a fork."""
874class SessionListResponse(BaseModel):
875 """Body for ``GET /api/sessions``."""
877 sessions: list[SessionMetaItem]
880class SessionMessageItem(BaseModel):
881 """One message in a session transcript."""
883 role: MessageRole
884 content: str
885 sources: list[str]
886 ts: str
889class SessionDetailResponse(BaseModel):
890 """Body for ``GET /api/sessions/{session_id}``: metadata plus transcript.
892 ``summary`` carries what compaction folded the oldest turns into (empty when
893 a conversation has not been compacted). A client that resumes and continues
894 the conversation needs it: without it, it rebuilds history from the raw
895 transcript, re-sending turns the summary had already condensed and risking
896 the context overflow compaction exists to prevent.
897 """
899 meta: SessionMetaItem
900 messages: list[SessionMessageItem]
901 summary: str = ""
904class SessionCreateRequest(BaseModel):
905 """Request body for ``POST /api/sessions``."""
907 model_ref: str
908 scope: str
911class SessionMessageCreateRequest(BaseModel):
912 """Request body for ``POST /api/sessions/{session_id}/messages``."""
914 role: MessageRole
915 content: str
916 sources: list[str] = []
919class SessionForkRequest(BaseModel):
920 """Request body for ``POST /api/sessions/{session_id}/fork``.
922 ``message_count`` is the number of leading messages to copy; null copies all.
923 """
925 message_count: int | None = None
927 @field_validator("message_count", mode="before")
928 @classmethod
929 def _check_message_count(cls, value: object) -> object:
930 """Refuse ``true``, ``"2"`` and ``1.0`` rather than coerce them into a count."""
931 # isinstance: the raw JSON value, before pydantic's lax coercion runs.
932 if value is None or (isinstance(value, int) and not isinstance(value, bool)):
933 return value
934 raise ValueError("message_count must be a whole number or null")
937class SessionSummaryRequest(BaseModel):
938 """Request body for ``PUT /api/sessions/{session_id}/summary``."""
940 summary: str
943class SessionRenameRequest(BaseModel):
944 """Request body for ``PATCH /api/sessions/{session_id}``."""
946 title: str
949class SessionRenameResponse(BaseModel):
950 """Outcome of a rename."""
952 id: str
953 title: str
956class SessionDeleteResponse(BaseModel):
957 """Outcome of a delete."""
959 id: str
960 deleted: bool
963class AgentClientDetection(BaseModel):
964 """Whether one agent client's CLI is installed on the machine lilbee runs on."""
966 client: AgentClient
967 cli_detected: bool
968 cli_path: str | None
971class AgentConfigIndexResponse(BaseModel):
972 """Response for ``GET /api/agent-config``: every client lilbee can configure."""
974 clients: list[AgentClientDetection]
976 @classmethod
977 def from_detections(cls, detections: list[ClientDetection]) -> AgentConfigIndexResponse:
978 """Serialize the probe results one entry per supported client."""
979 return cls(
980 clients=[
981 AgentClientDetection(
982 client=found.client,
983 cli_detected=found.cli_detected,
984 cli_path=found.cli_path,
985 )
986 for found in detections
987 ]
988 )
991class AgentConfigResponse(BaseModel):
992 """Response for ``GET /api/agent-config/{client}``: one client's live config.
994 ``config`` carries the block for a JSON client, ``content`` the rendered text
995 for a YAML one. ``stdio_config`` is the alternative block for a client that
996 can also run lilbee as a subprocess instead of calling this server.
997 """
999 client: AgentClient
1000 format: ConfigFormat
1001 surfaces: list[AgentSurface]
1002 config: dict[str, Any] | None = None
1003 content: str | None = None
1004 stdio_config: dict[str, Any] | None = None
1006 @classmethod
1007 def from_document(cls, document: AgentConfigDocument) -> AgentConfigResponse:
1008 """The canonical serialized config document, shared by the HTTP and CLI surfaces."""
1009 return cls(
1010 client=document.client,
1011 format=document.format,
1012 surfaces=list(document.surfaces),
1013 config=document.config,
1014 content=document.content,
1015 stdio_config=document.stdio_config,
1016 )