agenthub-python 0.4.1__py3-none-any.whl → 0.4.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -12,7 +12,7 @@
12
12
  # See the License for the specific language governing permissions and
13
13
  # limitations under the License.
14
14
 
15
- from .client import GLM5_1Client
15
+ from .client import AntMessagesClient
16
16
 
17
17
 
18
- __all__ = ["GLM5_1Client"]
18
+ __all__ = ["AntMessagesClient"]
@@ -12,14 +12,11 @@
12
12
  # See the License for the specific language governing permissions and
13
13
  # limitations under the License.
14
14
 
15
- import base64
16
- import mimetypes
17
15
  import os
18
16
  import re
19
17
  from typing import Any, AsyncIterator
20
18
 
21
- import httpx
22
- from anthropic import AsyncAnthropic, AsyncAnthropicBedrock
19
+ from anthropic import AsyncAnthropic
23
20
  from anthropic.types.beta import BetaMessageParam, BetaRawMessageStreamEvent
24
21
 
25
22
  from ..base_client import LLMClient
@@ -36,93 +33,60 @@ from ..types import (
36
33
  UniMessage,
37
34
  UsageMetadata,
38
35
  )
36
+ from ..utils import fix_openrouter_usage_metadata
39
37
 
40
38
 
41
39
  REDACTED_THINKING = "_REDACTED_THINKING"
42
40
 
43
41
 
44
- class Claude4_6Client(LLMClient):
45
- """Claude 4.6-specific LLM client implementation."""
42
+ class AntMessagesClient(LLMClient):
43
+ """Anthropic Messages-compatible client implementation."""
46
44
 
47
45
  def __init__(self, model: str, api_key: str | None = None, base_url: str | None = None):
48
- """Initialize Claude 4.6 client with model and API key."""
46
+ """Initialize Anthropic Messages-compatible client with model, API key, and base URL."""
49
47
  self._model = model
50
48
  api_key = api_key or os.getenv("ANTHROPIC_API_KEY")
51
49
  base_url = base_url or os.getenv("ANTHROPIC_BASE_URL")
52
- if base_url and base_url.startswith("bedrock://"): # example: bedrock://us-east-1
53
- region = base_url.replace("bedrock://", "")
54
- access_key, secret_key = api_key.split(",")
55
- self._client = AsyncAnthropicBedrock(
56
- aws_secret_key=secret_key, aws_access_key=access_key, aws_region=region
57
- )
58
- self._use_bedrock = True
59
- else:
60
- self._client = AsyncAnthropic(api_key=api_key, base_url=base_url)
61
- self._use_bedrock = False
62
-
50
+ # send the credential through both header conventions: Anthropic and DeepSeek read
51
+ # x-api-key while gateways such as OpenRouter and Z.AI read Authorization: Bearer
52
+ self._client = AsyncAnthropic(api_key=api_key, auth_token=api_key, base_url=base_url)
63
53
  self._history: list[UniMessage] = []
64
54
 
65
- async def _convert_image_url_to_source(self, url: str) -> dict[str, Any]:
66
- """Convert image URL to image source.
67
-
68
- Bedrock does not support image url sources, so we need to fetch the image bytes and encode them.
69
-
70
- Args:
71
- url: Image URL to convert
72
-
73
- Returns:
74
- Image source
75
- """
55
+ def _convert_image_url_to_source(self, url: str) -> dict[str, Any]:
56
+ """Convert image URL to an Anthropic image source block."""
76
57
  if url.startswith("data:"):
77
58
  match = re.match(r"data:([^;]+);base64,(.+)", url)
78
- if match:
79
- media_type = match.group(1)
80
- base64_data = match.group(2)
81
- source = {
82
- "type": "image",
83
- "source": {"type": "base64", "media_type": media_type, "data": base64_data},
84
- }
85
- else:
59
+ if not match:
86
60
  raise ValueError(f"Invalid base64 image: {url}")
87
- elif self._use_bedrock:
88
- async with httpx.AsyncClient() as client:
89
- response = await client.get(url)
90
- response.raise_for_status()
91
- image_bytes = response.content
92
- mime_type = mimetypes.guess_type(url)[0] or "image/jpeg"
93
- source = {
94
- "type": "image",
95
- "source": {
96
- "type": "base64",
97
- "media_type": mime_type,
98
- "data": base64.b64encode(image_bytes).decode("utf-8"),
99
- },
100
- }
101
- else:
102
- source = {"type": "image", "source": {"type": "url", "url": url}}
103
61
 
104
- return source
62
+ return {
63
+ "type": "image",
64
+ "source": {"type": "base64", "media_type": match.group(1), "data": match.group(2)},
65
+ }
66
+
67
+ return {"type": "image", "source": {"type": "url", "url": url}}
105
68
 
106
69
  def _convert_thinking_level_to_thinking_config(self, thinking_level: ThinkingLevel) -> dict[str, Any]:
107
- """Convert ThinkingLevel enum to Claude's adaptive thinking config."""
70
+ """Convert ThinkingLevel enum to the Messages API thinking config."""
71
+ # NONE is explicit rather than omitted because some servers (e.g. Z.AI) think by default
108
72
  mapping = {
109
- ThinkingLevel.NONE: {}, # omit thinking config
73
+ ThinkingLevel.NONE: {"thinking": {"type": "disabled"}},
110
74
  ThinkingLevel.LOW: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "low"}},
111
75
  ThinkingLevel.MEDIUM: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "medium"}},
112
76
  ThinkingLevel.HIGH: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "high"}},
113
- ThinkingLevel.XHIGH: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "high"}},
77
+ ThinkingLevel.XHIGH: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "xhigh"}},
114
78
  }
115
79
  return mapping.get(thinking_level)
116
80
 
117
81
  def _convert_tool_choice(self, tool_choice: ToolChoice) -> dict[str, str]:
118
- """Convert ToolChoice to Claude's tool_choice format."""
82
+ """Convert ToolChoice to the Messages API tool_choice format."""
119
83
  if isinstance(tool_choice, list):
120
84
  if len(tool_choice) > 1:
121
85
  raise UnsupportedParameterError(
122
- self.__class__.__name__, "tool_choice", "Claude supports only one tool choice."
86
+ self.__class__.__name__, "tool_choice", "The Messages API does not support multiple tool choices."
123
87
  )
124
88
 
125
- return {"type": "any", "name": tool_choice[0]}
89
+ return {"type": "tool", "name": tool_choice[0]}
126
90
  elif tool_choice == "none":
127
91
  return {"type": "none"}
128
92
  elif tool_choice == "auto":
@@ -132,70 +96,70 @@ class Claude4_6Client(LLMClient):
132
96
 
133
97
  def transform_uni_config_to_model_config(self, config: UniConfig) -> dict[str, Any]:
134
98
  """
135
- Transform universal configuration to Claude-specific configuration.
99
+ Transform universal configuration to Anthropic Messages-compatible configuration.
136
100
 
137
101
  Args:
138
102
  config: Universal configuration dict
139
103
 
140
104
  Returns:
141
- Claude configuration dictionary
105
+ Anthropic Messages API configuration dictionary
142
106
  """
143
- claude_config = {"model": self._model, "stream": True}
107
+ ant_config = {"model": self._model, "stream": True}
144
108
 
145
109
  if config.get("system_prompt") is not None:
146
- claude_config["system"] = config["system_prompt"]
110
+ ant_config["system"] = config["system_prompt"]
147
111
 
148
112
  if config.get("max_tokens") is not None:
149
- claude_config["max_tokens"] = config["max_tokens"]
113
+ ant_config["max_tokens"] = config["max_tokens"]
150
114
  else:
151
- claude_config["max_tokens"] = 64000 # Claude requires max_tokens to be specified
115
+ ant_config["max_tokens"] = 64000 # the Messages API requires max_tokens to be specified
152
116
 
153
117
  if config.get("temperature") is not None:
154
- claude_config["temperature"] = config["temperature"]
118
+ ant_config["temperature"] = config["temperature"]
155
119
 
156
- # NOTE: Claude always provides thinking summary
157
120
  if config.get("thinking_level") is not None:
158
- claude_config["temperature"] = 1.0 # `temperature` may only be set to 1 when thinking is enabled
159
- claude_config.update(self._convert_thinking_level_to_thinking_config(config["thinking_level"]))
121
+ ant_config.update(self._convert_thinking_level_to_thinking_config(config["thinking_level"]))
122
+ if config.get("thinking_summary") and ant_config.get("thinking", {}).get("type") == "adaptive":
123
+ ant_config["thinking"]["display"] = "summarized"
160
124
 
161
- # Convert tools to Claude's tool schema
125
+ # Convert tools to the Messages API tool schema
162
126
  if config.get("tools") is not None:
163
- claude_tools = []
127
+ ant_tools = []
164
128
  for tool in config["tools"]:
165
- claude_tool = {}
129
+ ant_tool = {}
166
130
  for key, value in tool.items():
167
- claude_tool[key.replace("parameters", "input_schema")] = value
131
+ ant_tool[key.replace("parameters", "input_schema")] = value
168
132
 
169
- claude_tools.append(claude_tool)
133
+ ant_tools.append(ant_tool)
170
134
 
171
- claude_config["tools"] = claude_tools
135
+ ant_config["tools"] = ant_tools
172
136
 
173
137
  # Convert tool_choice
174
138
  if config.get("tool_choice") is not None:
175
- claude_config["tool_choice"] = self._convert_tool_choice(config["tool_choice"])
139
+ ant_config["tool_choice"] = self._convert_tool_choice(config["tool_choice"])
140
+
141
+ if config.get("fast_mode"):
142
+ ant_config["speed"] = "fast"
143
+ ant_config["betas"] = ["fast-mode-2026-02-01"]
176
144
 
177
- # Add cache_control if prompt caching is enabled
178
- # TODO: wait for bedrock to support cache_control in config
179
- if not self._use_bedrock:
180
- prompt_caching = config.get("prompt_caching", PromptCaching.ENABLE)
181
- if prompt_caching == PromptCaching.ENABLE:
182
- claude_config["cache_control"] = {"type": "ephemeral"}
183
- elif prompt_caching == PromptCaching.ENHANCE:
184
- claude_config["cache_control"] = {"type": "ephemeral", "ttl": "1h"}
145
+ if config.get("prompt_caching") is not None and config["prompt_caching"] != PromptCaching.ENABLE:
146
+ raise UnsupportedParameterError(
147
+ self.__class__.__name__, "prompt_caching", "prompt_caching must be ENABLE for the Messages API."
148
+ )
185
149
 
186
- return claude_config
150
+ return ant_config
187
151
 
188
- async def transform_uni_message_to_model_input(self, messages: list[UniMessage]) -> list[BetaMessageParam]:
152
+ def transform_uni_message_to_model_input(self, messages: list[UniMessage]) -> list[BetaMessageParam]:
189
153
  """
190
- Transform universal message format to Claude's BetaMessageParam format.
154
+ Transform universal message format to the Messages API BetaMessageParam format.
191
155
 
192
156
  Args:
193
157
  messages: List of universal message dictionaries
194
158
 
195
159
  Returns:
196
- List of Claude BetaMessageParam objects
160
+ List of Messages API BetaMessageParam objects
197
161
  """
198
- claude_messages: list[BetaMessageParam] = []
162
+ ant_messages: list[BetaMessageParam] = []
199
163
 
200
164
  for msg in messages:
201
165
  content_blocks = []
@@ -203,18 +167,19 @@ class Claude4_6Client(LLMClient):
203
167
  if item["type"] == "text":
204
168
  content_blocks.append({"type": "text", "text": item["text"]})
205
169
  elif item["type"] == "image_url":
206
- content_blocks.append(await self._convert_image_url_to_source(item["image_url"]))
170
+ content_blocks.append(self._convert_image_url_to_source(item["image_url"]))
207
171
  elif item["type"] == "thinking":
208
172
  if item["thinking"] == REDACTED_THINKING:
209
173
  content_blocks.append({"type": "redacted_thinking", "data": item["fidelity"]["signature"]})
210
174
  else:
211
- content_blocks.append(
212
- {
213
- "type": "thinking",
214
- "thinking": item["thinking"],
215
- "signature": item["fidelity"]["signature"],
216
- }
217
- )
175
+ # third-party servers accept thinking without a signature, but the
176
+ # official API requires the one it emitted
177
+ thinking_block = {"type": "thinking", "thinking": item["thinking"]}
178
+ signature = (item.get("fidelity") or {}).get("signature")
179
+ if signature is not None:
180
+ thinking_block["signature"] = signature
181
+
182
+ content_blocks.append(thinking_block)
218
183
  elif item["type"] == "tool_call":
219
184
  content_blocks.append(
220
185
  {
@@ -231,7 +196,7 @@ class Claude4_6Client(LLMClient):
231
196
  tool_result = [{"type": "text", "text": item["text"]}]
232
197
  if "images" in item:
233
198
  for image_url in item["images"]:
234
- tool_result.append(await self._convert_image_url_to_source(image_url))
199
+ tool_result.append(self._convert_image_url_to_source(image_url))
235
200
 
236
201
  content_blocks.append(
237
202
  {"type": "tool_result", "content": tool_result, "tool_use_id": item["tool_call_id"]}
@@ -239,18 +204,18 @@ class Claude4_6Client(LLMClient):
239
204
  else:
240
205
  raise ValueError(f"Unknown item: {item}")
241
206
 
242
- claude_messages.append({"role": msg["role"], "content": content_blocks})
207
+ ant_messages.append({"role": msg["role"], "content": content_blocks})
243
208
 
244
- return claude_messages
209
+ return ant_messages
245
210
 
246
211
  def transform_model_output_to_uni_event(self, model_output: BetaRawMessageStreamEvent) -> UniEvent:
247
212
  """
248
- Transform Claude model output to universal event format.
213
+ Transform a Messages API streaming event to universal event format.
249
214
 
250
- NOTE: Claude always has only one content item per event.
215
+ NOTE: the Messages API always has only one content item per event.
251
216
 
252
217
  Args:
253
- model_output: Claude streaming event
218
+ model_output: Messages API streaming event
254
219
 
255
220
  Returns:
256
221
  Universal event dictionary
@@ -260,8 +225,8 @@ class Claude4_6Client(LLMClient):
260
225
  usage_metadata: UsageMetadata | None = None
261
226
  finish_reason: FinishReason | None = None
262
227
 
263
- claude_event_type = model_output.type
264
- if claude_event_type == "content_block_start":
228
+ ant_event_type = model_output.type
229
+ if ant_event_type == "content_block_start":
265
230
  event_type = "start"
266
231
  block = model_output.content_block
267
232
  if block.type == "tool_use":
@@ -273,7 +238,7 @@ class Claude4_6Client(LLMClient):
273
238
  {"type": "thinking", "thinking": REDACTED_THINKING, "fidelity": {"signature": block.data}}
274
239
  )
275
240
 
276
- elif claude_event_type == "content_block_delta":
241
+ elif ant_event_type == "content_block_delta":
277
242
  event_type = "delta"
278
243
  delta = model_output.delta
279
244
  if delta.type == "thinking_delta":
@@ -287,10 +252,10 @@ class Claude4_6Client(LLMClient):
287
252
  elif delta.type == "signature_delta":
288
253
  content_items.append({"type": "thinking", "thinking": "", "fidelity": {"signature": delta.signature}})
289
254
 
290
- elif claude_event_type == "content_block_stop":
255
+ elif ant_event_type == "content_block_stop":
291
256
  event_type = "stop"
292
257
 
293
- elif claude_event_type == "message_start":
258
+ elif ant_event_type == "message_start":
294
259
  event_type = "start"
295
260
  message = model_output.message
296
261
  if getattr(message, "usage", None):
@@ -302,7 +267,7 @@ class Claude4_6Client(LLMClient):
302
267
  "response_tokens": None,
303
268
  }
304
269
 
305
- elif claude_event_type == "message_delta":
270
+ elif ant_event_type == "message_delta":
306
271
  event_type = "stop"
307
272
  delta = model_output.delta
308
273
  if getattr(delta, "stop_reason", None):
@@ -314,19 +279,28 @@ class Claude4_6Client(LLMClient):
314
279
  }
315
280
  finish_reason = stop_reason_mapping.get(delta.stop_reason, "unknown")
316
281
 
317
- if getattr(model_output, "usage", None):
318
- # In message_delta, we only update response_tokens
282
+ usage = getattr(model_output, "usage", None)
283
+ if usage:
284
+ # gateways report zero usage in message_start and the full counts here, so the
285
+ # delta also carries the input-side fields (None on servers that omit them)
286
+ if usage.input_tokens is not None:
287
+ prompt_tokens = usage.input_tokens + (usage.cache_creation_input_tokens or 0)
288
+ else:
289
+ prompt_tokens = None
290
+
291
+ output_details = getattr(usage, "output_tokens_details", None)
292
+ thinking_tokens = getattr(output_details, "thinking_tokens", None) if output_details else None
319
293
  usage_metadata = {
320
- "cached_tokens": None,
321
- "prompt_tokens": None,
322
- "thoughts_tokens": None,
323
- "response_tokens": model_output.usage.output_tokens,
294
+ "cached_tokens": usage.cache_read_input_tokens,
295
+ "prompt_tokens": prompt_tokens,
296
+ "thoughts_tokens": thinking_tokens,
297
+ "response_tokens": usage.output_tokens - (thinking_tokens or 0),
324
298
  }
325
299
 
326
- elif claude_event_type == "message_stop":
300
+ elif ant_event_type == "message_stop":
327
301
  event_type = "stop"
328
302
 
329
- elif claude_event_type in ["text", "thinking", "signature", "input_json"]:
303
+ elif ant_event_type in ["text", "thinking", "signature", "input_json"]:
330
304
  event_type = "unused"
331
305
 
332
306
  else:
@@ -345,32 +319,17 @@ class Claude4_6Client(LLMClient):
345
319
  messages: list[UniMessage],
346
320
  config: UniConfig,
347
321
  ) -> AsyncIterator[UniEvent]:
348
- """Stream generate using Claude SDK with unified conversion methods."""
322
+ """Stream generate using an Anthropic Messages-compatible API with unified conversion methods."""
349
323
  # Use unified config conversion
350
- claude_config = self.transform_uni_config_to_model_config(config)
324
+ ant_config = self.transform_uni_config_to_model_config(config)
351
325
 
352
326
  # Use unified message conversion
353
- claude_messages = await self.transform_uni_message_to_model_input(messages)
354
-
355
- # Add cache_control to last user message's last item if using bedrock and enabled prompt caching
356
- # TODO: remove after bedrock supports cache_control in config
357
- if self._use_bedrock:
358
- prompt_caching = config.get("prompt_caching", PromptCaching.ENABLE)
359
- if prompt_caching != PromptCaching.DISABLE and claude_messages:
360
- try:
361
- last_user_message = next(filter(lambda x: x["role"] == "user", claude_messages[::-1]))
362
- last_content_item = last_user_message["content"][-1]
363
- last_content_item["cache_control"] = {
364
- "type": "ephemeral",
365
- "ttl": "1h" if prompt_caching == PromptCaching.ENHANCE else "5m",
366
- }
367
- except StopIteration:
368
- pass
327
+ ant_messages = self.transform_uni_message_to_model_input(messages)
369
328
 
370
329
  # Stream generate
371
330
  partial_tool_call = {}
372
331
  partial_usage = {}
373
- stream = await self._client.beta.messages.create(**claude_config, messages=claude_messages)
332
+ stream = await self._client.beta.messages.create(**ant_config, messages=ant_messages)
374
333
  async for event in stream:
375
334
  event = self.transform_model_output_to_uni_event(event)
376
335
  if event["event_type"] == "start":
@@ -425,18 +384,28 @@ class Claude4_6Client(LLMClient):
425
384
  }
426
385
  partial_tool_call = {}
427
386
 
428
- if "prompt_tokens" in partial_usage and event["usage_metadata"] is not None:
429
- # finish partial_usage
387
+ if event["usage_metadata"] is not None:
388
+ # finish partial_usage: the message_delta counts win over message_start
389
+ delta_usage = event["usage_metadata"]
390
+ usage_metadata = {
391
+ "prompt_tokens": (
392
+ delta_usage["prompt_tokens"]
393
+ if delta_usage["prompt_tokens"] is not None
394
+ else partial_usage.get("prompt_tokens")
395
+ ),
396
+ "cached_tokens": (
397
+ delta_usage["cached_tokens"]
398
+ if delta_usage["cached_tokens"] is not None
399
+ else partial_usage.get("cached_tokens")
400
+ ),
401
+ "thoughts_tokens": delta_usage["thoughts_tokens"],
402
+ "response_tokens": delta_usage["response_tokens"],
403
+ }
430
404
  yield {
431
405
  "role": "assistant",
432
406
  "event_type": "stop",
433
407
  "content_items": [],
434
- "usage_metadata": {
435
- "prompt_tokens": partial_usage["prompt_tokens"],
436
- "thoughts_tokens": None,
437
- "response_tokens": event["usage_metadata"]["response_tokens"],
438
- "cached_tokens": partial_usage["cached_tokens"],
439
- },
408
+ "usage_metadata": fix_openrouter_usage_metadata(usage_metadata, str(self._client.base_url)),
440
409
  "finish_reason": event["finish_reason"],
441
410
  }
442
411
  partial_usage = {}
agenthub/auto_client.py CHANGED
@@ -46,66 +46,65 @@ class AutoLLMClient(LLMClient):
46
46
  self, model: str, api_key: str | None = None, base_url: str | None = None, client_type: str | None = None
47
47
  ) -> LLMClient:
48
48
  """Create the appropriate client for the given model."""
49
- client_type = (client_type or os.getenv("CLIENT_TYPE", model)).lower()
50
- # gemini-3.6 must be matched before the broader gemini-3 prefix below
51
- if any(
52
- prefix in client_type for prefix in ("gemini-3.6", "gemini-3.5-flash-lite")
53
- ): # e.g., gemini-3.6-flash; gemini-3.5-flash-lite shares the sampling-parameter deprecation
54
- from .gemini3_6 import Gemini3_6Client
49
+ client_type = (client_type or os.getenv("CLIENT_TYPE") or model).lower()
50
+ if client_type == "minimax-m3":
51
+ from .minimax_m3 import MiniMaxM3Client
55
52
 
56
- return Gemini3_6Client(model=model, api_key=api_key, base_url=base_url)
57
- elif any(
53
+ return MiniMaxM3Client(model=model, api_key=api_key, base_url=base_url)
54
+ # every Gemini generation shares the unified client ("gemini-3" also matches the
55
+ # gemini-3.7/gemini-3.6/gemini-3.5-flash-lite client types)
56
+ if any(
58
57
  prefix in client_type for prefix in ("gemini-3", "gemini-embedding")
59
- ): # e.g., gemini-3-flash-preview, gemini-embedding-2
60
- from .gemini3 import Gemini3Client
58
+ ): # e.g., gemini-3.7-flash, gemini-3-flash-preview, gemini-embedding-2
59
+ from .gemini3_7 import Gemini3_7Client
61
60
 
62
- return Gemini3Client(model=model, api_key=api_key, base_url=base_url)
61
+ return Gemini3_7Client(model=model, api_key=api_key, base_url=base_url)
63
62
  elif "claude" in client_type and (
64
- "4-7" in client_type or "4-8" in client_type or "-5" in client_type
65
- ): # e.g., claude-opus-4-7
63
+ "4-6" in client_type or "4-7" in client_type or "4-8" in client_type or "-5" in client_type
64
+ ): # the whole Claude 4.6+ series shares the unified client, e.g., claude-sonnet-4-6
66
65
  from .claude5 import Claude5Client
67
66
 
68
67
  return Claude5Client(model=model, api_key=api_key, base_url=base_url)
69
- elif "claude" in client_type and "4-6" in client_type: # e.g., claude-sonnet-4-6
70
- from .claude4_6 import Claude4_6Client
71
-
72
- return Claude4_6Client(model=model, api_key=api_key, base_url=base_url)
73
- elif "gpt-5.4" in client_type or "gpt-5.5" in client_type: # e.g., gpt-5.5
74
- from .gpt5_5 import GPT5_5Client
75
-
76
- return GPT5_5Client(model=model, api_key=api_key, base_url=base_url)
77
- elif "glm-5.2" in client_type:
78
- from .glm5_2 import GLM5_2Client
68
+ elif "gpt-5.4" in client_type or "gpt-5.5" in client_type or "gpt-5.6" in client_type: # e.g., gpt-5.6
69
+ from .gpt5_6 import GPT5_6Client
79
70
 
80
- return GLM5_2Client(model=model, api_key=api_key, base_url=base_url)
81
- elif "glm-5" in client_type or "glm-5.1" in client_type:
82
- from .glm5_1 import GLM5_1Client
71
+ return GPT5_6Client(model=model, api_key=api_key, base_url=base_url)
72
+ elif "glm-5" in client_type: # the whole GLM series shares the unified client
73
+ from .glm5_3 import GLM5_3Client
83
74
 
84
- return GLM5_1Client(model=model, api_key=api_key, base_url=base_url)
85
- elif "kimi-k3" in client_type:
75
+ return GLM5_3Client(model=model, api_key=api_key, base_url=base_url)
76
+ elif "kimi-k3" in client_type or "kimi-k2.5" in client_type or "kimi-k2.6" in client_type:
77
+ # the whole Kimi K2.5+ series shares the unified client
86
78
  from .kimi_k3 import KimiK3Client
87
79
 
88
80
  return KimiK3Client(model=model, api_key=api_key, base_url=base_url)
89
- elif "kimi-k2.5" in client_type or "kimi-k2.6" in client_type:
90
- from .kimi_k2_6 import KimiK2_6Client
91
-
92
- return KimiK2_6Client(model=model, api_key=api_key, base_url=base_url)
93
81
  elif "deepseek-v4" in client_type:
94
82
  from .deepseek_v4 import DeepSeekV4Client
95
83
 
96
84
  return DeepSeekV4Client(model=model, api_key=api_key, base_url=base_url)
85
+ elif "ant-messages" in client_type:
86
+ from .ant_messages import AntMessagesClient
87
+
88
+ return AntMessagesClient(model=model, api_key=api_key, base_url=base_url)
89
+ elif "openai-responses" in client_type:
90
+ from .openai_responses import OpenaiResponsesClient
91
+
92
+ return OpenaiResponsesClient(model=model, api_key=api_key, base_url=base_url)
97
93
  elif "openai" in client_type and "embedding" in client_type:
98
94
  from .openai_embedding import OpenaiEmbeddingClient
99
95
 
100
96
  return OpenaiEmbeddingClient(model=model, api_key=api_key, base_url=base_url)
101
- elif "openai" in client_type and "embedding" not in client_type:
102
- from .openai import OpenaiClient
97
+ elif "openai" in client_type and "embedding" not in client_type: # openai-chat, plus bare "openai" as alias
98
+ from .openai_chat import OpenaiChatClient
103
99
 
104
- return OpenaiClient(model=model, api_key=api_key, base_url=base_url)
100
+ return OpenaiChatClient(model=model, api_key=api_key, base_url=base_url)
105
101
  else:
106
102
  raise ValueError(
107
103
  f"{client_type} is not supported. "
108
- "Supported client types: gemini-3.6, gemini-3, claude-5, claude-4-8, claude-4-7, claude-4-6, gpt-5.5, gpt-5.4, glm-5.2, glm-5.1, kimi-k3, kimi-k2.6, kimi-k2.5, deepseek-v4, openai-embedding, openai."
104
+ "Supported client types: minimax-m3, gemini-3.7, gemini-3.6, gemini-3, "
105
+ "claude-5, claude-4-8, claude-4-7, claude-4-6, gpt-5.6, gpt-5.5, gpt-5.4, "
106
+ "glm-5.3, glm-5.2, glm-5.1, kimi-k3, kimi-k2.6, kimi-k2.5, deepseek-v4, "
107
+ "openai-embedding, ant-messages, openai-responses, openai-chat."
109
108
  )
110
109
 
111
110
  def transform_uni_config_to_model_config(self, config: UniConfig) -> Any:
@@ -42,7 +42,7 @@ REDACTED_THINKING = "_REDACTED_THINKING"
42
42
 
43
43
 
44
44
  class Claude5Client(LLMClient):
45
- """Claude 5-specific LLM client implementation."""
45
+ """Claude 5-specific LLM client implementation (also serves Claude 4.6 through 4.8)."""
46
46
 
47
47
  def __init__(self, model: str, api_key: str | None = None, base_url: str | None = None):
48
48
  """Initialize Claude 5 client with model and API key."""
@@ -110,7 +110,11 @@ class Claude5Client(LLMClient):
110
110
  ThinkingLevel.LOW: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "low"}},
111
111
  ThinkingLevel.MEDIUM: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "medium"}},
112
112
  ThinkingLevel.HIGH: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "high"}},
113
- ThinkingLevel.XHIGH: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "xhigh"}},
113
+ # Claude 4.6 has no xhigh effort, so XHIGH degrades to the closest supported level
114
+ ThinkingLevel.XHIGH: {
115
+ "thinking": {"type": "adaptive"},
116
+ "output_config": {"effort": "high" if "4-6" in self._model else "xhigh"},
117
+ },
114
118
  }
115
119
  return mapping.get(thinking_level)
116
120
 
@@ -152,7 +156,10 @@ class Claude5Client(LLMClient):
152
156
 
153
157
  if config.get("temperature") is not None and config["temperature"] != 1.0:
154
158
  raise UnsupportedParameterError(
155
- self.__class__.__name__, "temperature", "Claude 4.8 does not support setting temperature."
159
+ self.__class__.__name__,
160
+ "temperature",
161
+ "Claude models do not support setting temperature; the API dropped it "
162
+ "starting with the 4.7 generation and the unified client rejects it for the whole family.",
156
163
  )
157
164
 
158
165
  if config.get("thinking_level") is not None:
@@ -176,6 +183,20 @@ class Claude5Client(LLMClient):
176
183
  if config.get("tool_choice") is not None:
177
184
  claude_config["tool_choice"] = self._convert_tool_choice(config["tool_choice"])
178
185
 
186
+ if config.get("fast_mode"):
187
+ if self._use_bedrock:
188
+ raise UnsupportedParameterError(
189
+ self.__class__.__name__, "fast_mode", "Bedrock does not support fast mode."
190
+ )
191
+
192
+ if "4-6" in self._model:
193
+ raise UnsupportedParameterError(
194
+ self.__class__.__name__, "fast_mode", "Claude 4.6 does not support fast mode."
195
+ )
196
+
197
+ claude_config["speed"] = "fast"
198
+ claude_config["betas"] = ["fast-mode-2026-02-01"]
199
+
179
200
  # Add cache_control if prompt caching is enabled
180
201
  # TODO: wait for bedrock to support cache_control in config
181
202
  if not self._use_bedrock:
@@ -109,6 +109,11 @@ class DeepSeekV4Client(LLMClient):
109
109
  if config.get("tool_choice") is not None:
110
110
  deepseek_config["tool_choice"] = self._convert_tool_choice(config["tool_choice"])
111
111
 
112
+ if config.get("fast_mode"):
113
+ raise UnsupportedParameterError(
114
+ self.__class__.__name__, "fast_mode", "DeepSeek V4 does not support fast mode."
115
+ )
116
+
112
117
  if config.get("prompt_caching") is not None and config["prompt_caching"] != PromptCaching.ENABLE:
113
118
  raise UnsupportedParameterError(
114
119
  self.__class__.__name__, "prompt_caching", "prompt_caching must be ENABLE for DeepSeek."
@@ -12,7 +12,7 @@
12
12
  # See the License for the specific language governing permissions and
13
13
  # limitations under the License.
14
14
 
15
- from .client import Claude4_6Client
15
+ from .client import Gemini3_7Client
16
16
 
17
17
 
18
- __all__ = ["Claude4_6Client"]
18
+ __all__ = ["Gemini3_7Client"]