Coverage for src/lilbee/server/chat_dispatch/canonical.py: 100%

111 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""Protocol-neutral chat request, response, stream-event, and token-count types.""" 

2 

3from __future__ import annotations 

4 

5from dataclasses import dataclass 

6from enum import StrEnum 

7from typing import Any, Literal 

8 

9 

10class StopReason(StrEnum): 

11 """Why a canonical chat response ended.""" 

12 

13 END_TURN = "end_turn" 

14 MAX_TOKENS = "max_tokens" 

15 TOOL_USE = "tool_use" 

16 

17 

18@dataclass(frozen=True) 

19class TextBlock: 

20 """Plain-text content block.""" 

21 

22 text: str 

23 type: Literal["text"] = "text" 

24 

25 

26@dataclass(frozen=True) 

27class ToolUseBlock: 

28 """Assistant-emitted tool invocation with parsed JSON arguments.""" 

29 

30 id: str 

31 name: str 

32 input: dict[str, Any] 

33 type: Literal["tool_use"] = "tool_use" 

34 

35 

36@dataclass(frozen=True) 

37class ToolResultBlock: 

38 """Caller-supplied tool result paired to a prior ToolUseBlock by id.""" 

39 

40 tool_use_id: str 

41 content: list[ContentBlock] 

42 is_error: bool = False 

43 type: Literal["tool_result"] = "tool_result" 

44 

45 

46ContentBlock = TextBlock | ToolUseBlock | ToolResultBlock 

47 

48 

49@dataclass(frozen=True) 

50class CanonicalMessage: 

51 """One chat turn; content is always a typed-block list.""" 

52 

53 role: Literal["user", "assistant", "tool"] 

54 content: list[ContentBlock] 

55 

56 @classmethod 

57 def from_string( 

58 cls, 

59 *, 

60 role: Literal["user", "assistant", "tool"], 

61 text: str, 

62 ) -> CanonicalMessage: 

63 """Build a single-text-block message from a raw string.""" 

64 return cls(role=role, content=[TextBlock(text=text)]) 

65 

66 

67@dataclass(frozen=True) 

68class CanonicalTool: 

69 """Tool definition (JSON-Schema input shape).""" 

70 

71 name: str 

72 description: str 

73 input_schema: dict[str, Any] 

74 

75 

76@dataclass(frozen=True) 

77class CanonicalToolChoice: 

78 """Tool-choice mode; ``tool_name`` is required only when ``mode == "tool"``.""" 

79 

80 mode: Literal["auto", "any", "none", "tool"] 

81 tool_name: str | None = None 

82 

83 def __post_init__(self) -> None: 

84 # Without this the None reaches the provider as 

85 # {"function": {"name": None}}, a malformed tool choice rather than a 

86 # rejected request. 

87 if self.mode == "tool" and not self.tool_name: 

88 raise ValueError('CanonicalToolChoice(mode="tool") requires a tool_name') 

89 

90 

91@dataclass(frozen=True) 

92class CanonicalChatRequest: 

93 """Canonical chat request consumed by the dispatch layer.""" 

94 

95 model: str 

96 messages: list[CanonicalMessage] 

97 system: str | None = None 

98 tools: list[CanonicalTool] | None = None 

99 tool_choice: CanonicalToolChoice | None = None 

100 temperature: float | None = None 

101 top_p: float | None = None 

102 top_k: int | None = None 

103 max_tokens: int | None = None 

104 seed: int | None = None 

105 frequency_penalty: float | None = None 

106 presence_penalty: float | None = None 

107 stop: list[str] | None = None 

108 stream: bool = False 

109 # Thinking-template control (chat_template_kwargs.enable_thinking); None 

110 # leaves the model's template default in place. 

111 think: bool | None = None 

112 

113 

114@dataclass(frozen=True) 

115class CanonicalUsage: 

116 """Token-count summary for one chat response.""" 

117 

118 input_tokens: int 

119 output_tokens: int 

120 # Prompt tokens served from the engine's cache, a subset of input_tokens. 

121 cached_input_tokens: int = 0 

122 

123 @property 

124 def uncached_input_tokens(self) -> int: 

125 """Prompt tokens the engine did not serve from its cache.""" 

126 return self.input_tokens - self.cached_input_tokens 

127 

128 

129class TokenCountAccuracy(StrEnum): 

130 """Whether a prompt token count was measured on the backend or estimated.""" 

131 

132 EXACT = "exact" 

133 ESTIMATED = "estimated" 

134 

135 

136@dataclass(frozen=True) 

137class PromptTokenCount: 

138 """Tokens a prompt costs, and how the number was arrived at.""" 

139 

140 tokens: int 

141 accuracy: TokenCountAccuracy 

142 

143 

144@dataclass(frozen=True) 

145class CanonicalResponse: 

146 """Canonical non-streaming chat response.""" 

147 

148 id: str 

149 model: str 

150 content: list[ContentBlock] 

151 stop_reason: StopReason 

152 usage: CanonicalUsage 

153 

154 

155@dataclass(frozen=True) 

156class MessageStart: 

157 """Stream prelude carrying the message id and model ref.""" 

158 

159 id: str 

160 model: str 

161 

162 

163@dataclass(frozen=True) 

164class ContentBlockStart: 

165 """Opens a fresh content block at ``index`` with an initial shell.""" 

166 

167 index: int 

168 block: ContentBlock 

169 

170 

171@dataclass(frozen=True) 

172class TextDelta: 

173 """One text-token delta within an open text block.""" 

174 

175 text: str 

176 

177 

178@dataclass(frozen=True) 

179class ToolUseDelta: 

180 """Accumulating JSON fragment within an open tool-use block.""" 

181 

182 partial_json: str 

183 

184 

185@dataclass(frozen=True) 

186class ContentBlockDelta: 

187 """Delta payload routed to the content block at ``index``.""" 

188 

189 index: int 

190 delta: TextDelta | ToolUseDelta 

191 

192 

193@dataclass(frozen=True) 

194class ContentBlockStop: 

195 """Closes the content block at ``index``.""" 

196 

197 index: int 

198 

199 

200@dataclass(frozen=True) 

201class MessageDelta: 

202 """Trailing metadata; either field may carry a value, never both required.""" 

203 

204 stop_reason: StopReason | None = None 

205 usage: CanonicalUsage | None = None 

206 

207 

208@dataclass(frozen=True) 

209class MessageStop: 

210 """Stream terminator.""" 

211 

212 

213CanonicalStreamEvent = ( 

214 MessageStart 

215 | ContentBlockStart 

216 | ContentBlockDelta 

217 | ContentBlockStop 

218 | MessageDelta 

219 | MessageStop 

220)