agenthub-python 0.4.1__py3-none-any.whl → 0.4.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agenthub/{glm5_1 → ant_messages}/__init__.py +2 -2
- agenthub/{claude4_6 → ant_messages}/client.py +114 -145
- agenthub/auto_client.py +35 -36
- agenthub/claude5/client.py +24 -3
- agenthub/deepseek_v4/client.py +5 -0
- agenthub/{claude4_6 → gemini3_7}/__init__.py +2 -2
- agenthub/{gemini3_6 → gemini3_7}/client.py +95 -20
- agenthub/{glm5_2 → glm5_3}/__init__.py +2 -2
- agenthub/{glm5_2 → glm5_3}/client.py +48 -23
- agenthub/{gpt5_5 → gpt5_6}/__init__.py +2 -2
- agenthub/{gpt5_5 → gpt5_6}/client.py +49 -37
- agenthub/kimi_k3/client.py +34 -10
- agenthub/{gemini3_6 → minimax_m3}/__init__.py +2 -2
- agenthub/minimax_m3/client.py +315 -0
- agenthub/{gemini3 → openai_chat}/__init__.py +2 -2
- agenthub/{openai → openai_chat}/client.py +4 -1
- agenthub/openai_embedding/client.py +6 -0
- agenthub/openai_responses/__init__.py +18 -0
- agenthub/openai_responses/client.py +374 -0
- agenthub/registry.py +100 -29
- agenthub/types.py +3 -0
- {agenthub_python-0.4.1.dist-info → agenthub_python-0.4.2.dist-info}/METADATA +30 -41
- agenthub_python-0.4.2.dist-info/RECORD +36 -0
- {agenthub_python-0.4.1.dist-info → agenthub_python-0.4.2.dist-info}/WHEEL +1 -1
- agenthub/gemini3/client.py +0 -449
- agenthub/glm5_1/client.py +0 -393
- agenthub/kimi_k2_6/__init__.py +0 -18
- agenthub/kimi_k2_6/client.py +0 -429
- agenthub/openai/__init__.py +0 -18
- agenthub_python-0.4.1.dist-info/RECORD +0 -38
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
# See the License for the specific language governing permissions and
|
|
13
13
|
# limitations under the License.
|
|
14
14
|
|
|
15
|
-
from .client import
|
|
15
|
+
from .client import AntMessagesClient
|
|
16
16
|
|
|
17
17
|
|
|
18
|
-
__all__ = ["
|
|
18
|
+
__all__ = ["AntMessagesClient"]
|
|
@@ -12,14 +12,11 @@
|
|
|
12
12
|
# See the License for the specific language governing permissions and
|
|
13
13
|
# limitations under the License.
|
|
14
14
|
|
|
15
|
-
import base64
|
|
16
|
-
import mimetypes
|
|
17
15
|
import os
|
|
18
16
|
import re
|
|
19
17
|
from typing import Any, AsyncIterator
|
|
20
18
|
|
|
21
|
-
import
|
|
22
|
-
from anthropic import AsyncAnthropic, AsyncAnthropicBedrock
|
|
19
|
+
from anthropic import AsyncAnthropic
|
|
23
20
|
from anthropic.types.beta import BetaMessageParam, BetaRawMessageStreamEvent
|
|
24
21
|
|
|
25
22
|
from ..base_client import LLMClient
|
|
@@ -36,93 +33,60 @@ from ..types import (
|
|
|
36
33
|
UniMessage,
|
|
37
34
|
UsageMetadata,
|
|
38
35
|
)
|
|
36
|
+
from ..utils import fix_openrouter_usage_metadata
|
|
39
37
|
|
|
40
38
|
|
|
41
39
|
REDACTED_THINKING = "_REDACTED_THINKING"
|
|
42
40
|
|
|
43
41
|
|
|
44
|
-
class
|
|
45
|
-
"""
|
|
42
|
+
class AntMessagesClient(LLMClient):
|
|
43
|
+
"""Anthropic Messages-compatible client implementation."""
|
|
46
44
|
|
|
47
45
|
def __init__(self, model: str, api_key: str | None = None, base_url: str | None = None):
|
|
48
|
-
"""Initialize
|
|
46
|
+
"""Initialize Anthropic Messages-compatible client with model, API key, and base URL."""
|
|
49
47
|
self._model = model
|
|
50
48
|
api_key = api_key or os.getenv("ANTHROPIC_API_KEY")
|
|
51
49
|
base_url = base_url or os.getenv("ANTHROPIC_BASE_URL")
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
self._client = AsyncAnthropicBedrock(
|
|
56
|
-
aws_secret_key=secret_key, aws_access_key=access_key, aws_region=region
|
|
57
|
-
)
|
|
58
|
-
self._use_bedrock = True
|
|
59
|
-
else:
|
|
60
|
-
self._client = AsyncAnthropic(api_key=api_key, base_url=base_url)
|
|
61
|
-
self._use_bedrock = False
|
|
62
|
-
|
|
50
|
+
# send the credential through both header conventions: Anthropic and DeepSeek read
|
|
51
|
+
# x-api-key while gateways such as OpenRouter and Z.AI read Authorization: Bearer
|
|
52
|
+
self._client = AsyncAnthropic(api_key=api_key, auth_token=api_key, base_url=base_url)
|
|
63
53
|
self._history: list[UniMessage] = []
|
|
64
54
|
|
|
65
|
-
|
|
66
|
-
"""Convert image URL to image source.
|
|
67
|
-
|
|
68
|
-
Bedrock does not support image url sources, so we need to fetch the image bytes and encode them.
|
|
69
|
-
|
|
70
|
-
Args:
|
|
71
|
-
url: Image URL to convert
|
|
72
|
-
|
|
73
|
-
Returns:
|
|
74
|
-
Image source
|
|
75
|
-
"""
|
|
55
|
+
def _convert_image_url_to_source(self, url: str) -> dict[str, Any]:
|
|
56
|
+
"""Convert image URL to an Anthropic image source block."""
|
|
76
57
|
if url.startswith("data:"):
|
|
77
58
|
match = re.match(r"data:([^;]+);base64,(.+)", url)
|
|
78
|
-
if match:
|
|
79
|
-
media_type = match.group(1)
|
|
80
|
-
base64_data = match.group(2)
|
|
81
|
-
source = {
|
|
82
|
-
"type": "image",
|
|
83
|
-
"source": {"type": "base64", "media_type": media_type, "data": base64_data},
|
|
84
|
-
}
|
|
85
|
-
else:
|
|
59
|
+
if not match:
|
|
86
60
|
raise ValueError(f"Invalid base64 image: {url}")
|
|
87
|
-
elif self._use_bedrock:
|
|
88
|
-
async with httpx.AsyncClient() as client:
|
|
89
|
-
response = await client.get(url)
|
|
90
|
-
response.raise_for_status()
|
|
91
|
-
image_bytes = response.content
|
|
92
|
-
mime_type = mimetypes.guess_type(url)[0] or "image/jpeg"
|
|
93
|
-
source = {
|
|
94
|
-
"type": "image",
|
|
95
|
-
"source": {
|
|
96
|
-
"type": "base64",
|
|
97
|
-
"media_type": mime_type,
|
|
98
|
-
"data": base64.b64encode(image_bytes).decode("utf-8"),
|
|
99
|
-
},
|
|
100
|
-
}
|
|
101
|
-
else:
|
|
102
|
-
source = {"type": "image", "source": {"type": "url", "url": url}}
|
|
103
61
|
|
|
104
|
-
|
|
62
|
+
return {
|
|
63
|
+
"type": "image",
|
|
64
|
+
"source": {"type": "base64", "media_type": match.group(1), "data": match.group(2)},
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
return {"type": "image", "source": {"type": "url", "url": url}}
|
|
105
68
|
|
|
106
69
|
def _convert_thinking_level_to_thinking_config(self, thinking_level: ThinkingLevel) -> dict[str, Any]:
|
|
107
|
-
"""Convert ThinkingLevel enum to
|
|
70
|
+
"""Convert ThinkingLevel enum to the Messages API thinking config."""
|
|
71
|
+
# NONE is explicit rather than omitted because some servers (e.g. Z.AI) think by default
|
|
108
72
|
mapping = {
|
|
109
|
-
ThinkingLevel.NONE: {},
|
|
73
|
+
ThinkingLevel.NONE: {"thinking": {"type": "disabled"}},
|
|
110
74
|
ThinkingLevel.LOW: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "low"}},
|
|
111
75
|
ThinkingLevel.MEDIUM: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "medium"}},
|
|
112
76
|
ThinkingLevel.HIGH: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "high"}},
|
|
113
|
-
ThinkingLevel.XHIGH: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "
|
|
77
|
+
ThinkingLevel.XHIGH: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "xhigh"}},
|
|
114
78
|
}
|
|
115
79
|
return mapping.get(thinking_level)
|
|
116
80
|
|
|
117
81
|
def _convert_tool_choice(self, tool_choice: ToolChoice) -> dict[str, str]:
|
|
118
|
-
"""Convert ToolChoice to
|
|
82
|
+
"""Convert ToolChoice to the Messages API tool_choice format."""
|
|
119
83
|
if isinstance(tool_choice, list):
|
|
120
84
|
if len(tool_choice) > 1:
|
|
121
85
|
raise UnsupportedParameterError(
|
|
122
|
-
self.__class__.__name__, "tool_choice", "
|
|
86
|
+
self.__class__.__name__, "tool_choice", "The Messages API does not support multiple tool choices."
|
|
123
87
|
)
|
|
124
88
|
|
|
125
|
-
return {"type": "
|
|
89
|
+
return {"type": "tool", "name": tool_choice[0]}
|
|
126
90
|
elif tool_choice == "none":
|
|
127
91
|
return {"type": "none"}
|
|
128
92
|
elif tool_choice == "auto":
|
|
@@ -132,70 +96,70 @@ class Claude4_6Client(LLMClient):
|
|
|
132
96
|
|
|
133
97
|
def transform_uni_config_to_model_config(self, config: UniConfig) -> dict[str, Any]:
|
|
134
98
|
"""
|
|
135
|
-
Transform universal configuration to
|
|
99
|
+
Transform universal configuration to Anthropic Messages-compatible configuration.
|
|
136
100
|
|
|
137
101
|
Args:
|
|
138
102
|
config: Universal configuration dict
|
|
139
103
|
|
|
140
104
|
Returns:
|
|
141
|
-
|
|
105
|
+
Anthropic Messages API configuration dictionary
|
|
142
106
|
"""
|
|
143
|
-
|
|
107
|
+
ant_config = {"model": self._model, "stream": True}
|
|
144
108
|
|
|
145
109
|
if config.get("system_prompt") is not None:
|
|
146
|
-
|
|
110
|
+
ant_config["system"] = config["system_prompt"]
|
|
147
111
|
|
|
148
112
|
if config.get("max_tokens") is not None:
|
|
149
|
-
|
|
113
|
+
ant_config["max_tokens"] = config["max_tokens"]
|
|
150
114
|
else:
|
|
151
|
-
|
|
115
|
+
ant_config["max_tokens"] = 64000 # the Messages API requires max_tokens to be specified
|
|
152
116
|
|
|
153
117
|
if config.get("temperature") is not None:
|
|
154
|
-
|
|
118
|
+
ant_config["temperature"] = config["temperature"]
|
|
155
119
|
|
|
156
|
-
# NOTE: Claude always provides thinking summary
|
|
157
120
|
if config.get("thinking_level") is not None:
|
|
158
|
-
|
|
159
|
-
|
|
121
|
+
ant_config.update(self._convert_thinking_level_to_thinking_config(config["thinking_level"]))
|
|
122
|
+
if config.get("thinking_summary") and ant_config.get("thinking", {}).get("type") == "adaptive":
|
|
123
|
+
ant_config["thinking"]["display"] = "summarized"
|
|
160
124
|
|
|
161
|
-
# Convert tools to
|
|
125
|
+
# Convert tools to the Messages API tool schema
|
|
162
126
|
if config.get("tools") is not None:
|
|
163
|
-
|
|
127
|
+
ant_tools = []
|
|
164
128
|
for tool in config["tools"]:
|
|
165
|
-
|
|
129
|
+
ant_tool = {}
|
|
166
130
|
for key, value in tool.items():
|
|
167
|
-
|
|
131
|
+
ant_tool[key.replace("parameters", "input_schema")] = value
|
|
168
132
|
|
|
169
|
-
|
|
133
|
+
ant_tools.append(ant_tool)
|
|
170
134
|
|
|
171
|
-
|
|
135
|
+
ant_config["tools"] = ant_tools
|
|
172
136
|
|
|
173
137
|
# Convert tool_choice
|
|
174
138
|
if config.get("tool_choice") is not None:
|
|
175
|
-
|
|
139
|
+
ant_config["tool_choice"] = self._convert_tool_choice(config["tool_choice"])
|
|
140
|
+
|
|
141
|
+
if config.get("fast_mode"):
|
|
142
|
+
ant_config["speed"] = "fast"
|
|
143
|
+
ant_config["betas"] = ["fast-mode-2026-02-01"]
|
|
176
144
|
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
if prompt_caching == PromptCaching.ENABLE:
|
|
182
|
-
claude_config["cache_control"] = {"type": "ephemeral"}
|
|
183
|
-
elif prompt_caching == PromptCaching.ENHANCE:
|
|
184
|
-
claude_config["cache_control"] = {"type": "ephemeral", "ttl": "1h"}
|
|
145
|
+
if config.get("prompt_caching") is not None and config["prompt_caching"] != PromptCaching.ENABLE:
|
|
146
|
+
raise UnsupportedParameterError(
|
|
147
|
+
self.__class__.__name__, "prompt_caching", "prompt_caching must be ENABLE for the Messages API."
|
|
148
|
+
)
|
|
185
149
|
|
|
186
|
-
return
|
|
150
|
+
return ant_config
|
|
187
151
|
|
|
188
|
-
|
|
152
|
+
def transform_uni_message_to_model_input(self, messages: list[UniMessage]) -> list[BetaMessageParam]:
|
|
189
153
|
"""
|
|
190
|
-
Transform universal message format to
|
|
154
|
+
Transform universal message format to the Messages API BetaMessageParam format.
|
|
191
155
|
|
|
192
156
|
Args:
|
|
193
157
|
messages: List of universal message dictionaries
|
|
194
158
|
|
|
195
159
|
Returns:
|
|
196
|
-
List of
|
|
160
|
+
List of Messages API BetaMessageParam objects
|
|
197
161
|
"""
|
|
198
|
-
|
|
162
|
+
ant_messages: list[BetaMessageParam] = []
|
|
199
163
|
|
|
200
164
|
for msg in messages:
|
|
201
165
|
content_blocks = []
|
|
@@ -203,18 +167,19 @@ class Claude4_6Client(LLMClient):
|
|
|
203
167
|
if item["type"] == "text":
|
|
204
168
|
content_blocks.append({"type": "text", "text": item["text"]})
|
|
205
169
|
elif item["type"] == "image_url":
|
|
206
|
-
content_blocks.append(
|
|
170
|
+
content_blocks.append(self._convert_image_url_to_source(item["image_url"]))
|
|
207
171
|
elif item["type"] == "thinking":
|
|
208
172
|
if item["thinking"] == REDACTED_THINKING:
|
|
209
173
|
content_blocks.append({"type": "redacted_thinking", "data": item["fidelity"]["signature"]})
|
|
210
174
|
else:
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
175
|
+
# third-party servers accept thinking without a signature, but the
|
|
176
|
+
# official API requires the one it emitted
|
|
177
|
+
thinking_block = {"type": "thinking", "thinking": item["thinking"]}
|
|
178
|
+
signature = (item.get("fidelity") or {}).get("signature")
|
|
179
|
+
if signature is not None:
|
|
180
|
+
thinking_block["signature"] = signature
|
|
181
|
+
|
|
182
|
+
content_blocks.append(thinking_block)
|
|
218
183
|
elif item["type"] == "tool_call":
|
|
219
184
|
content_blocks.append(
|
|
220
185
|
{
|
|
@@ -231,7 +196,7 @@ class Claude4_6Client(LLMClient):
|
|
|
231
196
|
tool_result = [{"type": "text", "text": item["text"]}]
|
|
232
197
|
if "images" in item:
|
|
233
198
|
for image_url in item["images"]:
|
|
234
|
-
tool_result.append(
|
|
199
|
+
tool_result.append(self._convert_image_url_to_source(image_url))
|
|
235
200
|
|
|
236
201
|
content_blocks.append(
|
|
237
202
|
{"type": "tool_result", "content": tool_result, "tool_use_id": item["tool_call_id"]}
|
|
@@ -239,18 +204,18 @@ class Claude4_6Client(LLMClient):
|
|
|
239
204
|
else:
|
|
240
205
|
raise ValueError(f"Unknown item: {item}")
|
|
241
206
|
|
|
242
|
-
|
|
207
|
+
ant_messages.append({"role": msg["role"], "content": content_blocks})
|
|
243
208
|
|
|
244
|
-
return
|
|
209
|
+
return ant_messages
|
|
245
210
|
|
|
246
211
|
def transform_model_output_to_uni_event(self, model_output: BetaRawMessageStreamEvent) -> UniEvent:
|
|
247
212
|
"""
|
|
248
|
-
Transform
|
|
213
|
+
Transform a Messages API streaming event to universal event format.
|
|
249
214
|
|
|
250
|
-
NOTE:
|
|
215
|
+
NOTE: the Messages API always has only one content item per event.
|
|
251
216
|
|
|
252
217
|
Args:
|
|
253
|
-
model_output:
|
|
218
|
+
model_output: Messages API streaming event
|
|
254
219
|
|
|
255
220
|
Returns:
|
|
256
221
|
Universal event dictionary
|
|
@@ -260,8 +225,8 @@ class Claude4_6Client(LLMClient):
|
|
|
260
225
|
usage_metadata: UsageMetadata | None = None
|
|
261
226
|
finish_reason: FinishReason | None = None
|
|
262
227
|
|
|
263
|
-
|
|
264
|
-
if
|
|
228
|
+
ant_event_type = model_output.type
|
|
229
|
+
if ant_event_type == "content_block_start":
|
|
265
230
|
event_type = "start"
|
|
266
231
|
block = model_output.content_block
|
|
267
232
|
if block.type == "tool_use":
|
|
@@ -273,7 +238,7 @@ class Claude4_6Client(LLMClient):
|
|
|
273
238
|
{"type": "thinking", "thinking": REDACTED_THINKING, "fidelity": {"signature": block.data}}
|
|
274
239
|
)
|
|
275
240
|
|
|
276
|
-
elif
|
|
241
|
+
elif ant_event_type == "content_block_delta":
|
|
277
242
|
event_type = "delta"
|
|
278
243
|
delta = model_output.delta
|
|
279
244
|
if delta.type == "thinking_delta":
|
|
@@ -287,10 +252,10 @@ class Claude4_6Client(LLMClient):
|
|
|
287
252
|
elif delta.type == "signature_delta":
|
|
288
253
|
content_items.append({"type": "thinking", "thinking": "", "fidelity": {"signature": delta.signature}})
|
|
289
254
|
|
|
290
|
-
elif
|
|
255
|
+
elif ant_event_type == "content_block_stop":
|
|
291
256
|
event_type = "stop"
|
|
292
257
|
|
|
293
|
-
elif
|
|
258
|
+
elif ant_event_type == "message_start":
|
|
294
259
|
event_type = "start"
|
|
295
260
|
message = model_output.message
|
|
296
261
|
if getattr(message, "usage", None):
|
|
@@ -302,7 +267,7 @@ class Claude4_6Client(LLMClient):
|
|
|
302
267
|
"response_tokens": None,
|
|
303
268
|
}
|
|
304
269
|
|
|
305
|
-
elif
|
|
270
|
+
elif ant_event_type == "message_delta":
|
|
306
271
|
event_type = "stop"
|
|
307
272
|
delta = model_output.delta
|
|
308
273
|
if getattr(delta, "stop_reason", None):
|
|
@@ -314,19 +279,28 @@ class Claude4_6Client(LLMClient):
|
|
|
314
279
|
}
|
|
315
280
|
finish_reason = stop_reason_mapping.get(delta.stop_reason, "unknown")
|
|
316
281
|
|
|
317
|
-
|
|
318
|
-
|
|
282
|
+
usage = getattr(model_output, "usage", None)
|
|
283
|
+
if usage:
|
|
284
|
+
# gateways report zero usage in message_start and the full counts here, so the
|
|
285
|
+
# delta also carries the input-side fields (None on servers that omit them)
|
|
286
|
+
if usage.input_tokens is not None:
|
|
287
|
+
prompt_tokens = usage.input_tokens + (usage.cache_creation_input_tokens or 0)
|
|
288
|
+
else:
|
|
289
|
+
prompt_tokens = None
|
|
290
|
+
|
|
291
|
+
output_details = getattr(usage, "output_tokens_details", None)
|
|
292
|
+
thinking_tokens = getattr(output_details, "thinking_tokens", None) if output_details else None
|
|
319
293
|
usage_metadata = {
|
|
320
|
-
"cached_tokens":
|
|
321
|
-
"prompt_tokens":
|
|
322
|
-
"thoughts_tokens":
|
|
323
|
-
"response_tokens":
|
|
294
|
+
"cached_tokens": usage.cache_read_input_tokens,
|
|
295
|
+
"prompt_tokens": prompt_tokens,
|
|
296
|
+
"thoughts_tokens": thinking_tokens,
|
|
297
|
+
"response_tokens": usage.output_tokens - (thinking_tokens or 0),
|
|
324
298
|
}
|
|
325
299
|
|
|
326
|
-
elif
|
|
300
|
+
elif ant_event_type == "message_stop":
|
|
327
301
|
event_type = "stop"
|
|
328
302
|
|
|
329
|
-
elif
|
|
303
|
+
elif ant_event_type in ["text", "thinking", "signature", "input_json"]:
|
|
330
304
|
event_type = "unused"
|
|
331
305
|
|
|
332
306
|
else:
|
|
@@ -345,32 +319,17 @@ class Claude4_6Client(LLMClient):
|
|
|
345
319
|
messages: list[UniMessage],
|
|
346
320
|
config: UniConfig,
|
|
347
321
|
) -> AsyncIterator[UniEvent]:
|
|
348
|
-
"""Stream generate using
|
|
322
|
+
"""Stream generate using an Anthropic Messages-compatible API with unified conversion methods."""
|
|
349
323
|
# Use unified config conversion
|
|
350
|
-
|
|
324
|
+
ant_config = self.transform_uni_config_to_model_config(config)
|
|
351
325
|
|
|
352
326
|
# Use unified message conversion
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
# Add cache_control to last user message's last item if using bedrock and enabled prompt caching
|
|
356
|
-
# TODO: remove after bedrock supports cache_control in config
|
|
357
|
-
if self._use_bedrock:
|
|
358
|
-
prompt_caching = config.get("prompt_caching", PromptCaching.ENABLE)
|
|
359
|
-
if prompt_caching != PromptCaching.DISABLE and claude_messages:
|
|
360
|
-
try:
|
|
361
|
-
last_user_message = next(filter(lambda x: x["role"] == "user", claude_messages[::-1]))
|
|
362
|
-
last_content_item = last_user_message["content"][-1]
|
|
363
|
-
last_content_item["cache_control"] = {
|
|
364
|
-
"type": "ephemeral",
|
|
365
|
-
"ttl": "1h" if prompt_caching == PromptCaching.ENHANCE else "5m",
|
|
366
|
-
}
|
|
367
|
-
except StopIteration:
|
|
368
|
-
pass
|
|
327
|
+
ant_messages = self.transform_uni_message_to_model_input(messages)
|
|
369
328
|
|
|
370
329
|
# Stream generate
|
|
371
330
|
partial_tool_call = {}
|
|
372
331
|
partial_usage = {}
|
|
373
|
-
stream = await self._client.beta.messages.create(**
|
|
332
|
+
stream = await self._client.beta.messages.create(**ant_config, messages=ant_messages)
|
|
374
333
|
async for event in stream:
|
|
375
334
|
event = self.transform_model_output_to_uni_event(event)
|
|
376
335
|
if event["event_type"] == "start":
|
|
@@ -425,18 +384,28 @@ class Claude4_6Client(LLMClient):
|
|
|
425
384
|
}
|
|
426
385
|
partial_tool_call = {}
|
|
427
386
|
|
|
428
|
-
if
|
|
429
|
-
# finish partial_usage
|
|
387
|
+
if event["usage_metadata"] is not None:
|
|
388
|
+
# finish partial_usage: the message_delta counts win over message_start
|
|
389
|
+
delta_usage = event["usage_metadata"]
|
|
390
|
+
usage_metadata = {
|
|
391
|
+
"prompt_tokens": (
|
|
392
|
+
delta_usage["prompt_tokens"]
|
|
393
|
+
if delta_usage["prompt_tokens"] is not None
|
|
394
|
+
else partial_usage.get("prompt_tokens")
|
|
395
|
+
),
|
|
396
|
+
"cached_tokens": (
|
|
397
|
+
delta_usage["cached_tokens"]
|
|
398
|
+
if delta_usage["cached_tokens"] is not None
|
|
399
|
+
else partial_usage.get("cached_tokens")
|
|
400
|
+
),
|
|
401
|
+
"thoughts_tokens": delta_usage["thoughts_tokens"],
|
|
402
|
+
"response_tokens": delta_usage["response_tokens"],
|
|
403
|
+
}
|
|
430
404
|
yield {
|
|
431
405
|
"role": "assistant",
|
|
432
406
|
"event_type": "stop",
|
|
433
407
|
"content_items": [],
|
|
434
|
-
"usage_metadata":
|
|
435
|
-
"prompt_tokens": partial_usage["prompt_tokens"],
|
|
436
|
-
"thoughts_tokens": None,
|
|
437
|
-
"response_tokens": event["usage_metadata"]["response_tokens"],
|
|
438
|
-
"cached_tokens": partial_usage["cached_tokens"],
|
|
439
|
-
},
|
|
408
|
+
"usage_metadata": fix_openrouter_usage_metadata(usage_metadata, str(self._client.base_url)),
|
|
440
409
|
"finish_reason": event["finish_reason"],
|
|
441
410
|
}
|
|
442
411
|
partial_usage = {}
|
agenthub/auto_client.py
CHANGED
|
@@ -46,66 +46,65 @@ class AutoLLMClient(LLMClient):
|
|
|
46
46
|
self, model: str, api_key: str | None = None, base_url: str | None = None, client_type: str | None = None
|
|
47
47
|
) -> LLMClient:
|
|
48
48
|
"""Create the appropriate client for the given model."""
|
|
49
|
-
client_type = (client_type or os.getenv("CLIENT_TYPE"
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
prefix in client_type for prefix in ("gemini-3.6", "gemini-3.5-flash-lite")
|
|
53
|
-
): # e.g., gemini-3.6-flash; gemini-3.5-flash-lite shares the sampling-parameter deprecation
|
|
54
|
-
from .gemini3_6 import Gemini3_6Client
|
|
49
|
+
client_type = (client_type or os.getenv("CLIENT_TYPE") or model).lower()
|
|
50
|
+
if client_type == "minimax-m3":
|
|
51
|
+
from .minimax_m3 import MiniMaxM3Client
|
|
55
52
|
|
|
56
|
-
return
|
|
57
|
-
|
|
53
|
+
return MiniMaxM3Client(model=model, api_key=api_key, base_url=base_url)
|
|
54
|
+
# every Gemini generation shares the unified client ("gemini-3" also matches the
|
|
55
|
+
# gemini-3.7/gemini-3.6/gemini-3.5-flash-lite client types)
|
|
56
|
+
if any(
|
|
58
57
|
prefix in client_type for prefix in ("gemini-3", "gemini-embedding")
|
|
59
|
-
): # e.g., gemini-3-flash-preview, gemini-embedding-2
|
|
60
|
-
from .
|
|
58
|
+
): # e.g., gemini-3.7-flash, gemini-3-flash-preview, gemini-embedding-2
|
|
59
|
+
from .gemini3_7 import Gemini3_7Client
|
|
61
60
|
|
|
62
|
-
return
|
|
61
|
+
return Gemini3_7Client(model=model, api_key=api_key, base_url=base_url)
|
|
63
62
|
elif "claude" in client_type and (
|
|
64
|
-
"4-7" in client_type or "4-8" in client_type or "-5" in client_type
|
|
65
|
-
): # e.g., claude-
|
|
63
|
+
"4-6" in client_type or "4-7" in client_type or "4-8" in client_type or "-5" in client_type
|
|
64
|
+
): # the whole Claude 4.6+ series shares the unified client, e.g., claude-sonnet-4-6
|
|
66
65
|
from .claude5 import Claude5Client
|
|
67
66
|
|
|
68
67
|
return Claude5Client(model=model, api_key=api_key, base_url=base_url)
|
|
69
|
-
elif "
|
|
70
|
-
from .
|
|
71
|
-
|
|
72
|
-
return Claude4_6Client(model=model, api_key=api_key, base_url=base_url)
|
|
73
|
-
elif "gpt-5.4" in client_type or "gpt-5.5" in client_type: # e.g., gpt-5.5
|
|
74
|
-
from .gpt5_5 import GPT5_5Client
|
|
75
|
-
|
|
76
|
-
return GPT5_5Client(model=model, api_key=api_key, base_url=base_url)
|
|
77
|
-
elif "glm-5.2" in client_type:
|
|
78
|
-
from .glm5_2 import GLM5_2Client
|
|
68
|
+
elif "gpt-5.4" in client_type or "gpt-5.5" in client_type or "gpt-5.6" in client_type: # e.g., gpt-5.6
|
|
69
|
+
from .gpt5_6 import GPT5_6Client
|
|
79
70
|
|
|
80
|
-
return
|
|
81
|
-
elif "glm-5" in client_type
|
|
82
|
-
from .
|
|
71
|
+
return GPT5_6Client(model=model, api_key=api_key, base_url=base_url)
|
|
72
|
+
elif "glm-5" in client_type: # the whole GLM series shares the unified client
|
|
73
|
+
from .glm5_3 import GLM5_3Client
|
|
83
74
|
|
|
84
|
-
return
|
|
85
|
-
elif "kimi-k3" in client_type:
|
|
75
|
+
return GLM5_3Client(model=model, api_key=api_key, base_url=base_url)
|
|
76
|
+
elif "kimi-k3" in client_type or "kimi-k2.5" in client_type or "kimi-k2.6" in client_type:
|
|
77
|
+
# the whole Kimi K2.5+ series shares the unified client
|
|
86
78
|
from .kimi_k3 import KimiK3Client
|
|
87
79
|
|
|
88
80
|
return KimiK3Client(model=model, api_key=api_key, base_url=base_url)
|
|
89
|
-
elif "kimi-k2.5" in client_type or "kimi-k2.6" in client_type:
|
|
90
|
-
from .kimi_k2_6 import KimiK2_6Client
|
|
91
|
-
|
|
92
|
-
return KimiK2_6Client(model=model, api_key=api_key, base_url=base_url)
|
|
93
81
|
elif "deepseek-v4" in client_type:
|
|
94
82
|
from .deepseek_v4 import DeepSeekV4Client
|
|
95
83
|
|
|
96
84
|
return DeepSeekV4Client(model=model, api_key=api_key, base_url=base_url)
|
|
85
|
+
elif "ant-messages" in client_type:
|
|
86
|
+
from .ant_messages import AntMessagesClient
|
|
87
|
+
|
|
88
|
+
return AntMessagesClient(model=model, api_key=api_key, base_url=base_url)
|
|
89
|
+
elif "openai-responses" in client_type:
|
|
90
|
+
from .openai_responses import OpenaiResponsesClient
|
|
91
|
+
|
|
92
|
+
return OpenaiResponsesClient(model=model, api_key=api_key, base_url=base_url)
|
|
97
93
|
elif "openai" in client_type and "embedding" in client_type:
|
|
98
94
|
from .openai_embedding import OpenaiEmbeddingClient
|
|
99
95
|
|
|
100
96
|
return OpenaiEmbeddingClient(model=model, api_key=api_key, base_url=base_url)
|
|
101
|
-
elif "openai" in client_type and "embedding" not in client_type:
|
|
102
|
-
from .
|
|
97
|
+
elif "openai" in client_type and "embedding" not in client_type: # openai-chat, plus bare "openai" as alias
|
|
98
|
+
from .openai_chat import OpenaiChatClient
|
|
103
99
|
|
|
104
|
-
return
|
|
100
|
+
return OpenaiChatClient(model=model, api_key=api_key, base_url=base_url)
|
|
105
101
|
else:
|
|
106
102
|
raise ValueError(
|
|
107
103
|
f"{client_type} is not supported. "
|
|
108
|
-
"Supported client types:
|
|
104
|
+
"Supported client types: minimax-m3, gemini-3.7, gemini-3.6, gemini-3, "
|
|
105
|
+
"claude-5, claude-4-8, claude-4-7, claude-4-6, gpt-5.6, gpt-5.5, gpt-5.4, "
|
|
106
|
+
"glm-5.3, glm-5.2, glm-5.1, kimi-k3, kimi-k2.6, kimi-k2.5, deepseek-v4, "
|
|
107
|
+
"openai-embedding, ant-messages, openai-responses, openai-chat."
|
|
109
108
|
)
|
|
110
109
|
|
|
111
110
|
def transform_uni_config_to_model_config(self, config: UniConfig) -> Any:
|
agenthub/claude5/client.py
CHANGED
|
@@ -42,7 +42,7 @@ REDACTED_THINKING = "_REDACTED_THINKING"
|
|
|
42
42
|
|
|
43
43
|
|
|
44
44
|
class Claude5Client(LLMClient):
|
|
45
|
-
"""Claude 5-specific LLM client implementation."""
|
|
45
|
+
"""Claude 5-specific LLM client implementation (also serves Claude 4.6 through 4.8)."""
|
|
46
46
|
|
|
47
47
|
def __init__(self, model: str, api_key: str | None = None, base_url: str | None = None):
|
|
48
48
|
"""Initialize Claude 5 client with model and API key."""
|
|
@@ -110,7 +110,11 @@ class Claude5Client(LLMClient):
|
|
|
110
110
|
ThinkingLevel.LOW: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "low"}},
|
|
111
111
|
ThinkingLevel.MEDIUM: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "medium"}},
|
|
112
112
|
ThinkingLevel.HIGH: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "high"}},
|
|
113
|
-
|
|
113
|
+
# Claude 4.6 has no xhigh effort, so XHIGH degrades to the closest supported level
|
|
114
|
+
ThinkingLevel.XHIGH: {
|
|
115
|
+
"thinking": {"type": "adaptive"},
|
|
116
|
+
"output_config": {"effort": "high" if "4-6" in self._model else "xhigh"},
|
|
117
|
+
},
|
|
114
118
|
}
|
|
115
119
|
return mapping.get(thinking_level)
|
|
116
120
|
|
|
@@ -152,7 +156,10 @@ class Claude5Client(LLMClient):
|
|
|
152
156
|
|
|
153
157
|
if config.get("temperature") is not None and config["temperature"] != 1.0:
|
|
154
158
|
raise UnsupportedParameterError(
|
|
155
|
-
self.__class__.__name__,
|
|
159
|
+
self.__class__.__name__,
|
|
160
|
+
"temperature",
|
|
161
|
+
"Claude models do not support setting temperature; the API dropped it "
|
|
162
|
+
"starting with the 4.7 generation and the unified client rejects it for the whole family.",
|
|
156
163
|
)
|
|
157
164
|
|
|
158
165
|
if config.get("thinking_level") is not None:
|
|
@@ -176,6 +183,20 @@ class Claude5Client(LLMClient):
|
|
|
176
183
|
if config.get("tool_choice") is not None:
|
|
177
184
|
claude_config["tool_choice"] = self._convert_tool_choice(config["tool_choice"])
|
|
178
185
|
|
|
186
|
+
if config.get("fast_mode"):
|
|
187
|
+
if self._use_bedrock:
|
|
188
|
+
raise UnsupportedParameterError(
|
|
189
|
+
self.__class__.__name__, "fast_mode", "Bedrock does not support fast mode."
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
if "4-6" in self._model:
|
|
193
|
+
raise UnsupportedParameterError(
|
|
194
|
+
self.__class__.__name__, "fast_mode", "Claude 4.6 does not support fast mode."
|
|
195
|
+
)
|
|
196
|
+
|
|
197
|
+
claude_config["speed"] = "fast"
|
|
198
|
+
claude_config["betas"] = ["fast-mode-2026-02-01"]
|
|
199
|
+
|
|
179
200
|
# Add cache_control if prompt caching is enabled
|
|
180
201
|
# TODO: wait for bedrock to support cache_control in config
|
|
181
202
|
if not self._use_bedrock:
|
agenthub/deepseek_v4/client.py
CHANGED
|
@@ -109,6 +109,11 @@ class DeepSeekV4Client(LLMClient):
|
|
|
109
109
|
if config.get("tool_choice") is not None:
|
|
110
110
|
deepseek_config["tool_choice"] = self._convert_tool_choice(config["tool_choice"])
|
|
111
111
|
|
|
112
|
+
if config.get("fast_mode"):
|
|
113
|
+
raise UnsupportedParameterError(
|
|
114
|
+
self.__class__.__name__, "fast_mode", "DeepSeek V4 does not support fast mode."
|
|
115
|
+
)
|
|
116
|
+
|
|
112
117
|
if config.get("prompt_caching") is not None and config["prompt_caching"] != PromptCaching.ENABLE:
|
|
113
118
|
raise UnsupportedParameterError(
|
|
114
119
|
self.__class__.__name__, "prompt_caching", "prompt_caching must be ENABLE for DeepSeek."
|
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
# See the License for the specific language governing permissions and
|
|
13
13
|
# limitations under the License.
|
|
14
14
|
|
|
15
|
-
from .client import
|
|
15
|
+
from .client import Gemini3_7Client
|
|
16
16
|
|
|
17
17
|
|
|
18
|
-
__all__ = ["
|
|
18
|
+
__all__ = ["Gemini3_7Client"]
|