pi-python-core 0.8.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pi_python/__init__.py +160 -0
- pi_python/_version.py +1 -0
- pi_python/agent.py +396 -0
- pi_python/cancellation.py +24 -0
- pi_python/data/models.json +3315 -0
- pi_python/errors.py +49 -0
- pi_python/estimate.py +144 -0
- pi_python/events.py +138 -0
- pi_python/function_tools.py +438 -0
- pi_python/hooks.py +44 -0
- pi_python/limits.py +28 -0
- pi_python/loop.py +431 -0
- pi_python/lowlevel.py +179 -0
- pi_python/mcp.py +187 -0
- pi_python/messages.py +405 -0
- pi_python/models.py +155 -0
- pi_python/provider.py +123 -0
- pi_python/providers/__init__.py +21 -0
- pi_python/providers/anthropic.py +673 -0
- pi_python/providers/common.py +201 -0
- pi_python/providers/completions.py +1149 -0
- pi_python/providers/oauth.py +542 -0
- pi_python/providers/openai.py +681 -0
- pi_python/providers/transport.py +574 -0
- pi_python/proxy.py +304 -0
- pi_python/py.typed +0 -0
- pi_python/queues.py +76 -0
- pi_python/recovery.py +209 -0
- pi_python/run.py +419 -0
- pi_python/stream.py +251 -0
- pi_python/sync.py +78 -0
- pi_python/testing.py +25 -0
- pi_python/tools.py +546 -0
- pi_python/transcript.py +167 -0
- pi_python_core-0.8.1.dist-info/METADATA +119 -0
- pi_python_core-0.8.1.dist-info/RECORD +39 -0
- pi_python_core-0.8.1.dist-info/WHEEL +4 -0
- pi_python_core-0.8.1.dist-info/licenses/LICENSE +21 -0
- pi_python_core-0.8.1.dist-info/licenses/NOTICE +8 -0
|
@@ -0,0 +1,673 @@
|
|
|
1
|
+
"""Anthropic Messages API; protocol adapted from Pi v1.0.0 (MIT)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
from collections.abc import AsyncGenerator
|
|
5
|
+
from ..cancellation import CancelToken
|
|
6
|
+
from ..provider import ModelRequest
|
|
7
|
+
from ..messages import ToolDeclaration
|
|
8
|
+
from typing import Any
|
|
9
|
+
import json
|
|
10
|
+
import re
|
|
11
|
+
from copy import deepcopy
|
|
12
|
+
from ..errors import ConfigurationError, ProviderProtocolError, UnsupportedCapabilityError
|
|
13
|
+
from ..messages import (
|
|
14
|
+
AssistantMessage,
|
|
15
|
+
UserMessage,
|
|
16
|
+
ToolResultMessage,
|
|
17
|
+
TextContent,
|
|
18
|
+
ImageContent,
|
|
19
|
+
ThinkingContent,
|
|
20
|
+
ToolCall,
|
|
21
|
+
CustomMessage,
|
|
22
|
+
)
|
|
23
|
+
from ..provider import ModelEvent
|
|
24
|
+
from ..messages import SystemMessage
|
|
25
|
+
from ..estimate import clamp_max_tokens_to_context
|
|
26
|
+
from ..transcript import (
|
|
27
|
+
current_tools,
|
|
28
|
+
declared_tools,
|
|
29
|
+
has_tool_redefinitions,
|
|
30
|
+
initial_system_message,
|
|
31
|
+
render_system_update,
|
|
32
|
+
resolve_transcript,
|
|
33
|
+
system_message_text,
|
|
34
|
+
with_request_tools,
|
|
35
|
+
)
|
|
36
|
+
from ..tools import invoke
|
|
37
|
+
from ..stream import event_contract
|
|
38
|
+
from .common import RemoteProvider, normalize_usage, transform_messages
|
|
39
|
+
|
|
40
|
+
_CC_NAMES = "Read Write Edit Bash Grep Glob AskUserQuestion EnterPlanMode ExitPlanMode KillShell NotebookEdit Skill Task TaskOutput TodoWrite WebFetch WebSearch".split()
|
|
41
|
+
_CC = {name.lower(): name for name in _CC_NAMES}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
_FINE_GRAINED_TOOL_STREAMING = "fine-grained-tool-streaming-2025-05-14"
|
|
45
|
+
_INTERLEAVED_THINKING = "interleaved-thinking-2025-05-14"
|
|
46
|
+
_SERVER_SIDE_FALLBACK = "server-side-fallback-2026-07-01"
|
|
47
|
+
_MID_CONVERSATION_OUTPUT_CONFIG = "mid-conversation-output-config-2026-07-01"
|
|
48
|
+
_THINKING_BINDING_CONTROLS = "thinking-binding-controls-2026-08-01"
|
|
49
|
+
_MID_CONVERSATION_TOOL_CHANGES = "mid-conversation-tool-changes-2026-07-01"
|
|
50
|
+
_EFFORTS = {"low", "medium", "high", "xhigh", "max"}
|
|
51
|
+
_BUDGETS = {"minimal": 1024, "low": 2048, "medium": 8192, "high": 16384}
|
|
52
|
+
# Declared from the first request whenever native tool changes are used. Anthropic adds
|
|
53
|
+
# hidden scaffolding once any tool is deferred; declaring it early keeps that scaffolding
|
|
54
|
+
# in the cached prefix. It is never activated and the model cannot see it.
|
|
55
|
+
_DEFERRED_PLACEHOLDER = {
|
|
56
|
+
"name": "__pi_deferred_placeholder__",
|
|
57
|
+
"description": "Reserved placeholder. Never available. Never call this.",
|
|
58
|
+
"input_schema": {"type": "object", "properties": {}, "required": []},
|
|
59
|
+
"defer_loading": True,
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def content_blocks(content: list[TextContent | ImageContent]) -> str | list[dict[str, Any]]:
|
|
64
|
+
"""Pi convertContentBlocks for tool results: text-only content becomes one string."""
|
|
65
|
+
if not any(isinstance(b, ImageContent) for b in content):
|
|
66
|
+
return "\n".join(b.text for b in content if isinstance(b, TextContent))
|
|
67
|
+
blocks = user_blocks(content)
|
|
68
|
+
if not any(b["type"] == "text" for b in blocks):
|
|
69
|
+
blocks.insert(0, {"type": "text", "text": "(see attached image)"})
|
|
70
|
+
return blocks
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def user_blocks(content: list[TextContent | ImageContent]) -> list[dict[str, Any]]:
|
|
74
|
+
result: list[dict[str, Any]] = []
|
|
75
|
+
for block in content:
|
|
76
|
+
if isinstance(block, TextContent):
|
|
77
|
+
result.append({"type": "text", "text": block.text})
|
|
78
|
+
elif isinstance(block, ImageContent):
|
|
79
|
+
result.append(
|
|
80
|
+
{
|
|
81
|
+
"type": "image",
|
|
82
|
+
"source": {"type": "base64", "media_type": block.mime_type, "data": block.data},
|
|
83
|
+
}
|
|
84
|
+
)
|
|
85
|
+
else:
|
|
86
|
+
raise UnsupportedCapabilityError("Unsupported Anthropic user/tool content")
|
|
87
|
+
return result
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def effort_for(level: str, level_map: dict[str, str | None]) -> str:
|
|
91
|
+
"""Pi mapThinkingLevelToEffort; a level mapped to None falls back by name."""
|
|
92
|
+
mapped = level_map.get(level)
|
|
93
|
+
if isinstance(mapped, str):
|
|
94
|
+
return mapped
|
|
95
|
+
return {"minimal": "low", "low": "low", "medium": "medium"}.get(level, "high")
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class AnthropicProvider(RemoteProvider):
|
|
99
|
+
name = "anthropic"
|
|
100
|
+
api = "anthropic-messages"
|
|
101
|
+
|
|
102
|
+
def __init__(
|
|
103
|
+
self, *, base_url: str = "https://api.anthropic.com", auth_mode: str = "auto", **kwargs: Any
|
|
104
|
+
) -> None:
|
|
105
|
+
super().__init__(**kwargs)
|
|
106
|
+
if auth_mode not in {"auto", "api_key", "oauth"}:
|
|
107
|
+
raise ConfigurationError("auth_mode must be auto, api_key or oauth")
|
|
108
|
+
self.base_url = base_url.rstrip("/")
|
|
109
|
+
self.auth_mode = auth_mode
|
|
110
|
+
|
|
111
|
+
def build_request(self, request: ModelRequest, oauth: bool = False) -> dict[str, Any]:
|
|
112
|
+
return self._build(request, oauth)[0]
|
|
113
|
+
|
|
114
|
+
def _build(
|
|
115
|
+
self, request: ModelRequest, oauth: bool = False
|
|
116
|
+
) -> tuple[dict[str, Any], list[str], str | None]:
|
|
117
|
+
"""Return the body, the anthropic-beta list and the managed effort, if any."""
|
|
118
|
+
options = request.options
|
|
119
|
+
model = self.model_info(request)
|
|
120
|
+
compat = model.compat
|
|
121
|
+
managed = compat.get("supportsMidConvoEffort") is True
|
|
122
|
+
model_max_tokens = model.max_tokens
|
|
123
|
+
adaptive = compat.get("forceAdaptiveThinking") is True
|
|
124
|
+
level_map = model.thinking_level_map
|
|
125
|
+
retention = options.get("cache_retention", "short")
|
|
126
|
+
if retention not in {"none", "short", "long"}:
|
|
127
|
+
raise ConfigurationError("Invalid cache_retention")
|
|
128
|
+
cache = (
|
|
129
|
+
None
|
|
130
|
+
if retention == "none"
|
|
131
|
+
else {
|
|
132
|
+
"type": "ephemeral",
|
|
133
|
+
**(
|
|
134
|
+
{"ttl": "1h"}
|
|
135
|
+
if retention == "long" and compat.get("supportsLongCacheRetention", True)
|
|
136
|
+
else {}
|
|
137
|
+
),
|
|
138
|
+
}
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
def tool_name(name: str) -> str:
|
|
142
|
+
return _CC.get(name.lower(), name) if oauth else name
|
|
143
|
+
|
|
144
|
+
source = with_request_tools(request.messages, request.tools)
|
|
145
|
+
|
|
146
|
+
def fit(limit: int) -> int:
|
|
147
|
+
# Pi clampMaxTokensToContext: leave room for the estimated prompt.
|
|
148
|
+
return clamp_max_tokens_to_context(model.context_window, source, limit)
|
|
149
|
+
|
|
150
|
+
transcript = resolve_transcript(
|
|
151
|
+
source, compat.get("supportsMidConvoSystemMessages") is True
|
|
152
|
+
)
|
|
153
|
+
initial = initial_system_message(transcript)
|
|
154
|
+
initial_tools = initial.tools_added if initial else []
|
|
155
|
+
# Native changes name tools, so a redefined name cannot be expressed, and an
|
|
156
|
+
# all-deferred tool list is rejected, so an initial active tool must anchor them.
|
|
157
|
+
native = (
|
|
158
|
+
compat.get("supportsMidConvoSystemMessages") is True
|
|
159
|
+
and compat.get("supportsMidConvoToolChanges") is True
|
|
160
|
+
and bool(initial_tools)
|
|
161
|
+
and not has_tool_redefinitions(transcript)
|
|
162
|
+
)
|
|
163
|
+
transformed = transform_messages(
|
|
164
|
+
transcript,
|
|
165
|
+
self.name,
|
|
166
|
+
self.api,
|
|
167
|
+
request.model,
|
|
168
|
+
lambda value, _: re.sub(r"[^a-zA-Z0-9_-]", "_", value)[:64],
|
|
169
|
+
)
|
|
170
|
+
conversation = transformed[1:] if initial else transformed
|
|
171
|
+
calls = [
|
|
172
|
+
b.id for m in conversation if isinstance(m, AssistantMessage) for b in m.tool_calls
|
|
173
|
+
]
|
|
174
|
+
if len(set(calls)) != len(calls):
|
|
175
|
+
raise ConfigurationError("Tool call IDs collide after Anthropic normalization")
|
|
176
|
+
messages: list[dict[str, Any]] = []
|
|
177
|
+
levels: dict[int, str] = {}
|
|
178
|
+
# Later system messages go directly before the next assistant message or at the
|
|
179
|
+
# end, because tool_result blocks must immediately follow their tool_use.
|
|
180
|
+
held: list[dict[str, Any]] = []
|
|
181
|
+
index = 0
|
|
182
|
+
while index < len(conversation):
|
|
183
|
+
message = conversation[index]
|
|
184
|
+
index += 1
|
|
185
|
+
if isinstance(message, SystemMessage):
|
|
186
|
+
blocks: list[dict[str, Any]] = []
|
|
187
|
+
text = render_system_update(message)
|
|
188
|
+
if text:
|
|
189
|
+
blocks.append({"type": "text", "text": text})
|
|
190
|
+
if native:
|
|
191
|
+
blocks += [
|
|
192
|
+
{
|
|
193
|
+
"type": "tool_removal",
|
|
194
|
+
"tool": {"type": "tool_reference", "name": tool_name(name)},
|
|
195
|
+
}
|
|
196
|
+
for name in message.tools_removed
|
|
197
|
+
]
|
|
198
|
+
blocks += [
|
|
199
|
+
{
|
|
200
|
+
"type": "tool_addition",
|
|
201
|
+
"tool": {"type": "tool_reference", "name": tool_name(t.name)},
|
|
202
|
+
}
|
|
203
|
+
for t in message.tools_added
|
|
204
|
+
]
|
|
205
|
+
if blocks:
|
|
206
|
+
held.append({"role": "system", "content": blocks})
|
|
207
|
+
elif isinstance(message, UserMessage):
|
|
208
|
+
if isinstance(message.content, str):
|
|
209
|
+
if message.content.strip():
|
|
210
|
+
messages.append({"role": "user", "content": message.content})
|
|
211
|
+
else:
|
|
212
|
+
blocks = [
|
|
213
|
+
b
|
|
214
|
+
for b in user_blocks(message.content)
|
|
215
|
+
if b["type"] != "text" or b["text"].strip()
|
|
216
|
+
]
|
|
217
|
+
if blocks:
|
|
218
|
+
messages.append({"role": "user", "content": blocks})
|
|
219
|
+
elif isinstance(message, AssistantMessage):
|
|
220
|
+
messages += held
|
|
221
|
+
held.clear()
|
|
222
|
+
blocks = []
|
|
223
|
+
for b in message.content:
|
|
224
|
+
if isinstance(b, TextContent):
|
|
225
|
+
if b.text.strip():
|
|
226
|
+
blocks.append({"type": "text", "text": b.text})
|
|
227
|
+
elif isinstance(b, ThinkingContent):
|
|
228
|
+
if b.redacted:
|
|
229
|
+
blocks.append(
|
|
230
|
+
{"type": "redacted_thinking", "data": b.thinking_signature}
|
|
231
|
+
)
|
|
232
|
+
elif b.thinking_signature and b.thinking_signature.strip():
|
|
233
|
+
blocks.append(
|
|
234
|
+
{
|
|
235
|
+
"type": "thinking",
|
|
236
|
+
"thinking": b.thinking,
|
|
237
|
+
"signature": b.thinking_signature,
|
|
238
|
+
}
|
|
239
|
+
)
|
|
240
|
+
elif b.thinking.strip():
|
|
241
|
+
blocks.append(
|
|
242
|
+
{"type": "thinking", "thinking": b.thinking, "signature": ""}
|
|
243
|
+
if compat.get("allowEmptySignature") is True
|
|
244
|
+
else {"type": "text", "text": b.thinking}
|
|
245
|
+
)
|
|
246
|
+
elif isinstance(b, ToolCall):
|
|
247
|
+
blocks.append(
|
|
248
|
+
{
|
|
249
|
+
"type": "tool_use",
|
|
250
|
+
"id": b.id,
|
|
251
|
+
"name": tool_name(b.name),
|
|
252
|
+
"input": b.arguments,
|
|
253
|
+
}
|
|
254
|
+
)
|
|
255
|
+
if not blocks:
|
|
256
|
+
continue
|
|
257
|
+
if (
|
|
258
|
+
managed
|
|
259
|
+
and message.api == self.api
|
|
260
|
+
and message.provider == self.name
|
|
261
|
+
and message.provider_thinking_level in _EFFORTS
|
|
262
|
+
):
|
|
263
|
+
levels[len(messages)] = message.provider_thinking_level
|
|
264
|
+
messages.append({"role": "assistant", "content": blocks})
|
|
265
|
+
elif isinstance(message, ToolResultMessage):
|
|
266
|
+
results = [message]
|
|
267
|
+
while index < len(conversation):
|
|
268
|
+
following = conversation[index]
|
|
269
|
+
if not isinstance(following, ToolResultMessage):
|
|
270
|
+
break
|
|
271
|
+
results.append(following)
|
|
272
|
+
index += 1
|
|
273
|
+
messages.append(
|
|
274
|
+
{
|
|
275
|
+
"role": "user",
|
|
276
|
+
"content": [
|
|
277
|
+
{
|
|
278
|
+
"type": "tool_result",
|
|
279
|
+
"tool_use_id": r.call_id,
|
|
280
|
+
"content": content_blocks(r.content),
|
|
281
|
+
"is_error": r.is_error,
|
|
282
|
+
}
|
|
283
|
+
for r in results
|
|
284
|
+
],
|
|
285
|
+
}
|
|
286
|
+
)
|
|
287
|
+
elif isinstance(message, CustomMessage):
|
|
288
|
+
raise UnsupportedCapabilityError("Convert custom messages before provider boundary")
|
|
289
|
+
messages += held
|
|
290
|
+
if cache and messages and messages[-1]["role"] in {"user", "system"}:
|
|
291
|
+
last = messages[-1]
|
|
292
|
+
if isinstance(last["content"], str):
|
|
293
|
+
last["content"] = [
|
|
294
|
+
{"type": "text", "text": last["content"], "cache_control": deepcopy(cache)}
|
|
295
|
+
]
|
|
296
|
+
elif last["content"] and last["content"][-1]["type"] in {
|
|
297
|
+
"text",
|
|
298
|
+
"image",
|
|
299
|
+
"tool_result",
|
|
300
|
+
"tool_addition",
|
|
301
|
+
"tool_removal",
|
|
302
|
+
}:
|
|
303
|
+
last["content"][-1]["cache_control"] = deepcopy(cache)
|
|
304
|
+
|
|
305
|
+
reasoning = options.get("reasoning")
|
|
306
|
+
if reasoning == "off":
|
|
307
|
+
reasoning = None
|
|
308
|
+
if reasoning is not None and reasoning not in {
|
|
309
|
+
"minimal",
|
|
310
|
+
"low",
|
|
311
|
+
"medium",
|
|
312
|
+
"high",
|
|
313
|
+
"xhigh",
|
|
314
|
+
"max",
|
|
315
|
+
}:
|
|
316
|
+
raise ConfigurationError("Unsupported reasoning level")
|
|
317
|
+
# Managed-effort models carry the effort as a marker after the history.
|
|
318
|
+
active_effort = effort_for(reasoning, level_map) if managed and reasoning else "high"
|
|
319
|
+
if managed:
|
|
320
|
+
marked = []
|
|
321
|
+
for position, value in enumerate(messages):
|
|
322
|
+
if position in levels:
|
|
323
|
+
marked.append(
|
|
324
|
+
{
|
|
325
|
+
"role": "system",
|
|
326
|
+
"content": [],
|
|
327
|
+
"output_config": {"effort": levels[position]},
|
|
328
|
+
}
|
|
329
|
+
)
|
|
330
|
+
marked.append(value)
|
|
331
|
+
marked.append(
|
|
332
|
+
{"role": "system", "content": [], "output_config": {"effort": active_effort}}
|
|
333
|
+
)
|
|
334
|
+
messages = marked
|
|
335
|
+
body: dict[str, Any] = {
|
|
336
|
+
"model": request.model,
|
|
337
|
+
"stream": True,
|
|
338
|
+
"max_tokens": fit(min(options.get("max_tokens", model_max_tokens), model_max_tokens)),
|
|
339
|
+
"messages": messages,
|
|
340
|
+
}
|
|
341
|
+
system = system_message_text(initial) if initial else ""
|
|
342
|
+
if oauth:
|
|
343
|
+
body["system"] = [
|
|
344
|
+
{
|
|
345
|
+
"type": "text",
|
|
346
|
+
"text": "You are Claude Code, Anthropic's official CLI for Claude.",
|
|
347
|
+
}
|
|
348
|
+
]
|
|
349
|
+
if system:
|
|
350
|
+
body["system"].append({"type": "text", "text": system})
|
|
351
|
+
elif system:
|
|
352
|
+
body["system"] = [{"type": "text", "text": system}]
|
|
353
|
+
for block in body.get("system", []):
|
|
354
|
+
if cache:
|
|
355
|
+
block["cache_control"] = deepcopy(cache)
|
|
356
|
+
|
|
357
|
+
tool_cache = cache if compat.get("supportsCacheControlOnTools", True) else None
|
|
358
|
+
|
|
359
|
+
def declarations(
|
|
360
|
+
tools: list[ToolDeclaration], cached: dict[str, Any] | None
|
|
361
|
+
) -> list[dict[str, Any]]:
|
|
362
|
+
result = [
|
|
363
|
+
{
|
|
364
|
+
"name": tool_name(t.name),
|
|
365
|
+
"description": t.description,
|
|
366
|
+
**(
|
|
367
|
+
{"eager_input_streaming": True}
|
|
368
|
+
if compat.get("supportsEagerToolInputStreaming", True)
|
|
369
|
+
else {}
|
|
370
|
+
),
|
|
371
|
+
"input_schema": {
|
|
372
|
+
"type": "object",
|
|
373
|
+
"properties": deepcopy(t.input_schema.get("properties", {})),
|
|
374
|
+
"required": deepcopy(t.input_schema.get("required", [])),
|
|
375
|
+
},
|
|
376
|
+
}
|
|
377
|
+
for t in tools
|
|
378
|
+
]
|
|
379
|
+
if cached and result:
|
|
380
|
+
result[-1]["cache_control"] = deepcopy(cached)
|
|
381
|
+
return result
|
|
382
|
+
|
|
383
|
+
current = current_tools(transcript)
|
|
384
|
+
if native:
|
|
385
|
+
# Initial tools stay active with the cache breakpoint; later ones are deferred
|
|
386
|
+
# and surfaced by tool_addition; removed ones stay declared. The list only grows.
|
|
387
|
+
names = {t.name for t in initial_tools}
|
|
388
|
+
body["tools"] = [
|
|
389
|
+
*declarations(initial_tools, tool_cache),
|
|
390
|
+
deepcopy(_DEFERRED_PLACEHOLDER),
|
|
391
|
+
*(
|
|
392
|
+
{**t, "defer_loading": True}
|
|
393
|
+
for t in declarations(
|
|
394
|
+
[t for t in declared_tools(transcript) if t.name not in names], None
|
|
395
|
+
)
|
|
396
|
+
),
|
|
397
|
+
]
|
|
398
|
+
elif current:
|
|
399
|
+
body["tools"] = declarations(current, tool_cache)
|
|
400
|
+
|
|
401
|
+
thinking_enabled = False
|
|
402
|
+
display = options.get("thinking_display", "summarized")
|
|
403
|
+
if managed:
|
|
404
|
+
# Adaptive with block binding, so a prefix mismatch drops a thinking block
|
|
405
|
+
# instead of failing every later request.
|
|
406
|
+
thinking_enabled = reasoning is not None
|
|
407
|
+
body["thinking"] = {
|
|
408
|
+
"type": "adaptive",
|
|
409
|
+
"display": display,
|
|
410
|
+
"block_binding": {"prefix_mismatch_behavior": "drop_block"},
|
|
411
|
+
}
|
|
412
|
+
body["output_config"] = {"effort": "high"}
|
|
413
|
+
elif model.reasoning:
|
|
414
|
+
if reasoning is not None:
|
|
415
|
+
thinking_enabled = True
|
|
416
|
+
if adaptive:
|
|
417
|
+
body["thinking"] = {"type": "adaptive", "display": display}
|
|
418
|
+
body["output_config"] = {"effort": effort_for(reasoning, level_map)}
|
|
419
|
+
else:
|
|
420
|
+
level = "high" if reasoning in {"xhigh", "max"} else reasoning
|
|
421
|
+
budget = options.get("thinking_budgets", {}).get(level, _BUDGETS[level])
|
|
422
|
+
if type(budget) is not int or budget < 0:
|
|
423
|
+
raise ConfigurationError("Invalid thinking budget")
|
|
424
|
+
ceiling = fit(
|
|
425
|
+
model_max_tokens
|
|
426
|
+
if "max_tokens" not in options
|
|
427
|
+
else min(options["max_tokens"] + budget, model_max_tokens)
|
|
428
|
+
)
|
|
429
|
+
budget = min(budget, max(0, ceiling - 1024))
|
|
430
|
+
if budget < 1024:
|
|
431
|
+
raise ConfigurationError(
|
|
432
|
+
"Thinking requires room for at least 1024 thinking and 1024 answer tokens"
|
|
433
|
+
)
|
|
434
|
+
body["thinking"] = {
|
|
435
|
+
"type": "enabled",
|
|
436
|
+
"budget_tokens": budget,
|
|
437
|
+
"display": display,
|
|
438
|
+
}
|
|
439
|
+
body["max_tokens"] = ceiling
|
|
440
|
+
elif level_map.get("off", "") is not None:
|
|
441
|
+
body["thinking"] = {"type": "disabled"}
|
|
442
|
+
if (
|
|
443
|
+
"temperature" in options
|
|
444
|
+
and not thinking_enabled
|
|
445
|
+
and not managed
|
|
446
|
+
and compat.get("supportsTemperature", True)
|
|
447
|
+
):
|
|
448
|
+
body["temperature"] = options["temperature"]
|
|
449
|
+
metadata = options.get("metadata")
|
|
450
|
+
if isinstance(metadata, dict) and isinstance(metadata.get("user_id"), str):
|
|
451
|
+
body["metadata"] = {"user_id": metadata["user_id"]}
|
|
452
|
+
if "tool_choice" in options:
|
|
453
|
+
choice = options["tool_choice"]
|
|
454
|
+
body["tool_choice"] = {"type": choice} if isinstance(choice, str) else deepcopy(choice)
|
|
455
|
+
fallbacks = compat.get("allowedFallbackModels") or []
|
|
456
|
+
if fallbacks:
|
|
457
|
+
body["fallbacks"] = [{"model": f["model"]} for f in fallbacks]
|
|
458
|
+
# Python extensions with no Pi equivalent; raw thinking/output_config override above.
|
|
459
|
+
for key in ("top_p", "top_k", "stop_sequences", "thinking", "output_config"):
|
|
460
|
+
if key in options:
|
|
461
|
+
body[key] = deepcopy(options[key])
|
|
462
|
+
|
|
463
|
+
configured = [
|
|
464
|
+
v for k, v in options.get("headers", {}).items() if k.lower() == "anthropic-beta"
|
|
465
|
+
]
|
|
466
|
+
if configured:
|
|
467
|
+
betas = list(dict.fromkeys(f.strip() for f in configured[-1].split(",") if f.strip()))
|
|
468
|
+
else:
|
|
469
|
+
betas = []
|
|
470
|
+
if oauth:
|
|
471
|
+
betas += ["claude-code-20250219", "oauth-2025-04-20"]
|
|
472
|
+
if current and compat.get("supportsEagerToolInputStreaming", True) is False:
|
|
473
|
+
betas.append(_FINE_GRAINED_TOOL_STREAMING)
|
|
474
|
+
if thinking_enabled and not adaptive and options.get("interleaved_thinking", True):
|
|
475
|
+
betas.append(_INTERLEAVED_THINKING)
|
|
476
|
+
if fallbacks:
|
|
477
|
+
betas.append(_SERVER_SIDE_FALLBACK)
|
|
478
|
+
if managed:
|
|
479
|
+
betas += [_MID_CONVERSATION_OUTPUT_CONFIG, _THINKING_BINDING_CONTROLS]
|
|
480
|
+
if native:
|
|
481
|
+
betas.append(_MID_CONVERSATION_TOOL_CHANGES)
|
|
482
|
+
betas = list(dict.fromkeys(betas))
|
|
483
|
+
return body, betas, active_effort if managed else None
|
|
484
|
+
|
|
485
|
+
@event_contract
|
|
486
|
+
async def stream(
|
|
487
|
+
self, request: ModelRequest, cancel: CancelToken
|
|
488
|
+
) -> AsyncGenerator[ModelEvent, None]:
|
|
489
|
+
key = await self.credential(request, cancel)
|
|
490
|
+
oauth = self.auth_mode == "oauth" or (
|
|
491
|
+
self.auth_mode == "auto"
|
|
492
|
+
and (self.credentials is not None or key.startswith("sk-ant-oat"))
|
|
493
|
+
)
|
|
494
|
+
if request.options.get("transport", "sse") not in {"sse", "auto"}:
|
|
495
|
+
raise UnsupportedCapabilityError("Anthropic Messages supports SSE transport")
|
|
496
|
+
built, betas, effort = self._build(request, oauth)
|
|
497
|
+
headers = {
|
|
498
|
+
**{
|
|
499
|
+
k: v
|
|
500
|
+
for k, v in request.options.get("headers", {}).items()
|
|
501
|
+
if k.lower() != "anthropic-beta"
|
|
502
|
+
},
|
|
503
|
+
**({"anthropic-beta": ",".join(betas)} if betas else {}),
|
|
504
|
+
"anthropic-version": "2023-06-01",
|
|
505
|
+
"anthropic-dangerous-direct-browser-access": "true",
|
|
506
|
+
"content-type": "application/json",
|
|
507
|
+
}
|
|
508
|
+
if oauth:
|
|
509
|
+
headers.update(
|
|
510
|
+
{
|
|
511
|
+
"authorization": f"Bearer {key}",
|
|
512
|
+
"user-agent": "claude-cli/2.1.280",
|
|
513
|
+
"x-app": "cli",
|
|
514
|
+
"anthropic-dangerous-direct-browser-access": "true",
|
|
515
|
+
}
|
|
516
|
+
)
|
|
517
|
+
else:
|
|
518
|
+
headers["x-api-key"] = key
|
|
519
|
+
body = await self.payload(request, built)
|
|
520
|
+
blocks: dict[int, Any] = {}
|
|
521
|
+
# Wire index -> content index; a leading server-side fallback block is skipped.
|
|
522
|
+
positions: dict[int, int] = {}
|
|
523
|
+
skipped: set[int] = set()
|
|
524
|
+
arguments = {}
|
|
525
|
+
usage = {}
|
|
526
|
+
reason = None
|
|
527
|
+
response_meta = {}
|
|
528
|
+
ended = False
|
|
529
|
+
closed_blocks = set()
|
|
530
|
+
names = {
|
|
531
|
+
t.name.lower(): t.name
|
|
532
|
+
for t in declared_tools(with_request_tools(request.messages, request.tools))
|
|
533
|
+
}
|
|
534
|
+
events = self.transport.stream(
|
|
535
|
+
self.base_url + "/v1/messages?beta=true",
|
|
536
|
+
body,
|
|
537
|
+
headers,
|
|
538
|
+
cancel,
|
|
539
|
+
on_response=request.on_response,
|
|
540
|
+
)
|
|
541
|
+
announced = False
|
|
542
|
+
try:
|
|
543
|
+
async for event in events:
|
|
544
|
+
if not announced:
|
|
545
|
+
announced = True
|
|
546
|
+
yield ModelEvent("start")
|
|
547
|
+
await invoke(request.on_provider_stream_event, deepcopy(event))
|
|
548
|
+
kind = event.get("type")
|
|
549
|
+
if ended:
|
|
550
|
+
raise ProviderProtocolError("Anthropic event after message_stop")
|
|
551
|
+
if kind == "error":
|
|
552
|
+
raise ProviderProtocolError("Anthropic stream reported an error")
|
|
553
|
+
if kind == "message_start":
|
|
554
|
+
usage.update(event["message"].get("usage", {}))
|
|
555
|
+
response_meta = event["message"]
|
|
556
|
+
elif kind == "content_block_start":
|
|
557
|
+
index = event["index"]
|
|
558
|
+
block = event["content_block"]
|
|
559
|
+
typ = block["type"]
|
|
560
|
+
if type(index) is not int or index in blocks or index in skipped:
|
|
561
|
+
raise ProviderProtocolError("Duplicate Anthropic block")
|
|
562
|
+
if typ == "fallback":
|
|
563
|
+
if blocks:
|
|
564
|
+
raise ProviderProtocolError(
|
|
565
|
+
"Anthropic performed an unsupported mid-output model fallback"
|
|
566
|
+
)
|
|
567
|
+
skipped.add(index)
|
|
568
|
+
continue
|
|
569
|
+
if index != len(blocks) + len(skipped):
|
|
570
|
+
raise ProviderProtocolError("Out-of-order Anthropic block")
|
|
571
|
+
positions[index] = len(blocks)
|
|
572
|
+
if typ == "text":
|
|
573
|
+
blocks[index] = TextContent(block.get("text", ""))
|
|
574
|
+
yield ModelEvent.boundary("start", positions[index], TextContent(""))
|
|
575
|
+
if blocks[index].text:
|
|
576
|
+
yield ModelEvent.text(blocks[index].text, positions[index])
|
|
577
|
+
elif typ == "thinking":
|
|
578
|
+
blocks[index] = ThinkingContent(
|
|
579
|
+
block.get("thinking", ""), block.get("signature", "")
|
|
580
|
+
)
|
|
581
|
+
yield ModelEvent.boundary("start", positions[index], ThinkingContent(""))
|
|
582
|
+
if blocks[index].thinking:
|
|
583
|
+
yield ModelEvent.thinking(blocks[index].thinking, positions[index])
|
|
584
|
+
elif typ == "redacted_thinking":
|
|
585
|
+
blocks[index] = ThinkingContent("[Reasoning redacted]", block["data"], True)
|
|
586
|
+
yield ModelEvent.boundary("start", positions[index], blocks[index])
|
|
587
|
+
elif typ == "tool_use":
|
|
588
|
+
blocks[index] = ToolCall(
|
|
589
|
+
block["id"],
|
|
590
|
+
names.get(block["name"].lower(), block["name"])
|
|
591
|
+
if oauth
|
|
592
|
+
else block["name"],
|
|
593
|
+
block.get("input", {}),
|
|
594
|
+
)
|
|
595
|
+
arguments[index] = ""
|
|
596
|
+
yield ModelEvent.boundary("start", positions[index], blocks[index])
|
|
597
|
+
else:
|
|
598
|
+
raise UnsupportedCapabilityError(f"Unsupported Anthropic block: {typ}")
|
|
599
|
+
elif kind == "content_block_delta":
|
|
600
|
+
index = event["index"]
|
|
601
|
+
if index in skipped:
|
|
602
|
+
continue
|
|
603
|
+
if index in closed_blocks:
|
|
604
|
+
raise ProviderProtocolError("Delta after content block stop")
|
|
605
|
+
block = blocks[index]
|
|
606
|
+
delta = event["delta"]
|
|
607
|
+
typ = delta["type"]
|
|
608
|
+
if typ == "text_delta" and isinstance(block, TextContent):
|
|
609
|
+
block.text += delta["text"]
|
|
610
|
+
yield ModelEvent.text(delta["text"], positions[index])
|
|
611
|
+
elif typ == "thinking_delta" and isinstance(block, ThinkingContent):
|
|
612
|
+
block.thinking += delta["thinking"]
|
|
613
|
+
yield ModelEvent.thinking(delta["thinking"], positions[index])
|
|
614
|
+
elif typ == "signature_delta" and isinstance(block, ThinkingContent):
|
|
615
|
+
block.thinking_signature = (block.thinking_signature or "") + delta[
|
|
616
|
+
"signature"
|
|
617
|
+
]
|
|
618
|
+
elif typ == "input_json_delta" and isinstance(block, ToolCall):
|
|
619
|
+
arguments[index] += delta["partial_json"]
|
|
620
|
+
yield ModelEvent.toolcall(delta["partial_json"], positions[index])
|
|
621
|
+
else:
|
|
622
|
+
raise UnsupportedCapabilityError(f"Unsupported Anthropic delta: {typ}")
|
|
623
|
+
elif kind == "content_block_stop":
|
|
624
|
+
index = event["index"]
|
|
625
|
+
if index in skipped:
|
|
626
|
+
continue
|
|
627
|
+
if index not in blocks or index in closed_blocks:
|
|
628
|
+
raise ProviderProtocolError("Invalid content block stop")
|
|
629
|
+
closed_blocks.add(index)
|
|
630
|
+
if index in arguments and arguments[index]:
|
|
631
|
+
blocks[index].arguments = json.loads(arguments[index])
|
|
632
|
+
if not isinstance(blocks[index].arguments, dict):
|
|
633
|
+
raise ProviderProtocolError("Tool arguments must be an object")
|
|
634
|
+
yield ModelEvent.boundary("end", positions[index], blocks[index])
|
|
635
|
+
elif kind == "message_delta":
|
|
636
|
+
usage.update(event.get("usage", {}))
|
|
637
|
+
reason = event["delta"].get("stop_reason", reason)
|
|
638
|
+
elif kind == "message_stop":
|
|
639
|
+
ended = True
|
|
640
|
+
finally:
|
|
641
|
+
await events.aclose()
|
|
642
|
+
if not ended or reason is None or closed_blocks != set(blocks):
|
|
643
|
+
raise ProviderProtocolError("Incomplete Anthropic stream")
|
|
644
|
+
mapped = {
|
|
645
|
+
"end_turn": "stop",
|
|
646
|
+
"stop_sequence": "stop",
|
|
647
|
+
"pause_turn": "stop",
|
|
648
|
+
"tool_use": "tool_use",
|
|
649
|
+
"max_tokens": "length",
|
|
650
|
+
"refusal": "error",
|
|
651
|
+
"sensitive": "error",
|
|
652
|
+
}.get(reason)
|
|
653
|
+
if mapped is None:
|
|
654
|
+
raise ProviderProtocolError("Unknown Anthropic stop reason")
|
|
655
|
+
if mapped == "error":
|
|
656
|
+
raise ProviderProtocolError(f"Anthropic stopped: {reason}")
|
|
657
|
+
yield ModelEvent.done(
|
|
658
|
+
AssistantMessage(
|
|
659
|
+
list(blocks.values()),
|
|
660
|
+
mapped,
|
|
661
|
+
self.name,
|
|
662
|
+
request.model,
|
|
663
|
+
normalize_usage(usage, self.name),
|
|
664
|
+
api="anthropic-messages",
|
|
665
|
+
provider_thinking_level=effort,
|
|
666
|
+
thinking_level=request.options.get("reasoning"),
|
|
667
|
+
response_id=response_meta.get("id"),
|
|
668
|
+
response_model=response_meta.get("model")
|
|
669
|
+
if response_meta.get("model") != request.model
|
|
670
|
+
else None,
|
|
671
|
+
raw_stop_reason=reason,
|
|
672
|
+
)
|
|
673
|
+
)
|