agenthub-python 0.4.3__tar.gz → 0.4.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/PKG-INFO +2 -1
  2. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/__init__.py +8 -1
  3. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/ant_messages/client.py +27 -4
  4. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/auto_client.py +69 -18
  5. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/base_client.py +9 -0
  6. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/claude5/client.py +36 -5
  7. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/deepseek_v4/client.py +27 -5
  8. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/errors.py +14 -0
  9. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/gemini3_7/client.py +37 -11
  10. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/glm5_3/client.py +19 -2
  11. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/gpt5_6/client.py +25 -3
  12. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/integration/playground.py +251 -46
  13. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/kimi_k3/client.py +19 -2
  14. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/minimax_m3/client.py +35 -20
  15. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/openai_chat/client.py +17 -2
  16. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/openai_embedding/client.py +17 -2
  17. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/openai_responses/client.py +27 -17
  18. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/registry.py +50 -23
  19. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/types.py +1 -0
  20. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/utils.py +16 -0
  21. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/pyproject.toml +2 -1
  22. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/pyproject.toml.orig +2 -2
  23. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/README.md +0 -0
  24. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/abort_signal.py +0 -0
  25. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/ant_messages/__init__.py +0 -0
  26. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/claude5/__init__.py +0 -0
  27. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/deepseek_v4/__init__.py +0 -0
  28. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/gemini3_7/__init__.py +0 -0
  29. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/glm5_3/__init__.py +0 -0
  30. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/gpt5_6/__init__.py +0 -0
  31. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/integration/__init__.py +0 -0
  32. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/integration/tracer.py +0 -0
  33. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/kimi_k3/__init__.py +0 -0
  34. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/minimax_m3/__init__.py +0 -0
  35. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/openai_chat/__init__.py +0 -0
  36. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/openai_embedding/__init__.py +0 -0
  37. {agenthub_python-0.4.3 → agenthub_python-0.4.5}/agenthub/openai_responses/__init__.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agenthub-python
3
- Version: 0.4.3
3
+ Version: 0.4.5
4
4
  Summary: AgentHub is the LLM API Hub for the Agent era, built for high-precision autonomous agents.
5
5
  Keywords: agent,llm,gemini,claude,gpt
6
6
  Author: PrismShadow
@@ -14,6 +14,7 @@ Requires-Dist: httpx[socks] ; extra == 'dev'
14
14
  Requires-Dist: pytest>=8.4.2 ; extra == 'dev'
15
15
  Requires-Dist: pytest-asyncio>=0.23.0 ; extra == 'dev'
16
16
  Requires-Dist: pytest-xdist>=3.6.0 ; extra == 'dev'
17
+ Requires-Dist: pytest-rerunfailures>=15.0 ; extra == 'dev'
17
18
  Requires-Dist: ruff>=0.14.3 ; extra == 'dev'
18
19
  Requires-Dist: pillow>=10.0.0 ; extra == 'dev'
19
20
  Requires-Python: >=3.11
@@ -13,7 +13,13 @@
13
13
  # limitations under the License.
14
14
 
15
15
  from .auto_client import AutoLLMClient
16
- from .errors import AgentHubError, EmptyResponseError, ToolCallArgumentParseError, UnsupportedParameterError
16
+ from .errors import (
17
+ AgentHubError,
18
+ EmptyResponseError,
19
+ ToolCallArgumentParseError,
20
+ UnsupportedOperationError,
21
+ UnsupportedParameterError,
22
+ )
17
23
  from .registry import Currency, Modality, ModelPricing, SupportedModel, list_supported_models
18
24
  from .types import PromptCaching, ThinkingLevel
19
25
 
@@ -29,6 +35,7 @@ __all__ = [
29
35
  "SupportedModel",
30
36
  "ThinkingLevel",
31
37
  "ToolCallArgumentParseError",
38
+ "UnsupportedOperationError",
32
39
  "UnsupportedParameterError",
33
40
  "list_supported_models",
34
41
  ]
@@ -33,7 +33,7 @@ from ..types import (
33
33
  UniMessage,
34
34
  UsageMetadata,
35
35
  )
36
- from ..utils import fix_openrouter_usage_metadata
36
+ from ..utils import fix_openrouter_usage_metadata, is_debug_enabled
37
37
 
38
38
 
39
39
  REDACTED_THINKING = "_REDACTED_THINKING"
@@ -42,14 +42,22 @@ REDACTED_THINKING = "_REDACTED_THINKING"
42
42
  class AntMessagesClient(LLMClient):
43
43
  """Anthropic Messages-compatible client implementation."""
44
44
 
45
- def __init__(self, model: str, api_key: str | None = None, base_url: str | None = None):
45
+ def __init__(
46
+ self,
47
+ model: str,
48
+ api_key: str | None = None,
49
+ base_url: str | None = None,
50
+ default_headers: dict[str, str] | None = None,
51
+ ):
46
52
  """Initialize Anthropic Messages-compatible client with model, API key, and base URL."""
47
53
  self._model = model
48
54
  api_key = api_key or os.getenv("ANTHROPIC_API_KEY")
49
55
  base_url = base_url or os.getenv("ANTHROPIC_BASE_URL")
50
56
  # send the credential through both header conventions: Anthropic and DeepSeek read
51
57
  # x-api-key while gateways such as OpenRouter and Z.AI read Authorization: Bearer
52
- self._client = AsyncAnthropic(api_key=api_key, auth_token=api_key, base_url=base_url)
58
+ self._client = AsyncAnthropic(
59
+ api_key=api_key, auth_token=api_key, base_url=base_url, default_headers=default_headers
60
+ )
53
61
  self._history: list[UniMessage] = []
54
62
 
55
63
  def _convert_image_url_to_source(self, url: str) -> dict[str, Any]:
@@ -75,6 +83,7 @@ class AntMessagesClient(LLMClient):
75
83
  ThinkingLevel.MEDIUM: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "medium"}},
76
84
  ThinkingLevel.HIGH: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "high"}},
77
85
  ThinkingLevel.XHIGH: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "xhigh"}},
86
+ ThinkingLevel.MAX: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "max"}},
78
87
  }
79
88
  return mapping.get(thinking_level)
80
89
 
@@ -305,9 +314,14 @@ class AntMessagesClient(LLMClient):
305
314
  # gateways that relabel it onto another event
306
315
  event_type = "unused"
307
316
 
308
- else:
317
+ elif is_debug_enabled():
309
318
  raise ValueError(f"Unknown output: {model_output}")
310
319
 
320
+ else:
321
+ # a gateway injects its own events (heartbeats, cost tickers) into the stream, and
322
+ # killing a long generation over one costs more than dropping it
323
+ event_type = "unused"
324
+
311
325
  return {
312
326
  "role": "assistant",
313
327
  "event_type": event_type,
@@ -411,3 +425,12 @@ class AntMessagesClient(LLMClient):
411
425
  "finish_reason": event["finish_reason"],
412
426
  }
413
427
  partial_usage = {}
428
+
429
+ async def list_models(self) -> list[str]:
430
+ """
431
+ List the model ids the configured endpoint serves.
432
+
433
+ Returns:
434
+ list[str]: The model ids, in the order the endpoint returned them.
435
+ """
436
+ return [model.id async for model in self._client.models.list()]
@@ -20,6 +20,10 @@ from .base_client import LLMClient
20
20
  from .types import UniConfig, UniEvent, UniMessage
21
21
 
22
22
 
23
+ # The generic protocol clients are named explicitly rather than deduced from a model id.
24
+ _PROTOCOL_CLIENT_TYPES = ("openai-chat", "openai-responses", "ant-messages", "openai-embedding")
25
+
26
+
23
27
  class AutoLLMClient(LLMClient):
24
28
  """
25
29
  Auto-routing LLM client that dispatches to appropriate model-specific client.
@@ -29,7 +33,12 @@ class AutoLLMClient(LLMClient):
29
33
  """
30
34
 
31
35
  def __init__(
32
- self, model: str, api_key: str | None = None, base_url: str | None = None, client_type: str | None = None
36
+ self,
37
+ model: str,
38
+ api_key: str | None = None,
39
+ base_url: str | None = None,
40
+ client_type: str | None = None,
41
+ default_headers: dict[str, str] | None = None,
33
42
  ):
34
43
  """
35
44
  Initialize AutoLLMClient with a specific model.
@@ -39,14 +48,22 @@ class AutoLLMClient(LLMClient):
39
48
  api_key: Optional API key
40
49
  base_url: Optional base URL for API requests
41
50
  client_type: Optional client type override
51
+ default_headers: Optional headers sent with every request, for endpoints that demand their own
42
52
  """
43
- self._client = self._create_client_for_model(model, api_key, base_url, client_type)
53
+ self._client_type = (client_type or os.getenv("CLIENT_TYPE") or model).lower()
54
+ self._client = self._create_client_for_model(model, api_key, base_url, self._client_type, default_headers)
44
55
 
45
- def _create_client_for_model(
46
- self, model: str, api_key: str | None = None, base_url: str | None = None, client_type: str | None = None
47
- ) -> LLMClient:
48
- """Create the appropriate client for the given model."""
49
- client_type = (client_type or os.getenv("CLIENT_TYPE") or model).lower()
56
+ @staticmethod
57
+ def _client_class_for_model(client_type: str) -> type[LLMClient] | None:
58
+ """
59
+ Resolve which client a resolved (lowercased) client type routes to.
60
+
61
+ Args:
62
+ client_type: The resolved client type, which is the model id when none was given.
63
+
64
+ Returns:
65
+ type[LLMClient] | None: The client class, or None when no client claims the type.
66
+ """
50
67
  # every Gemini generation shares the unified client ("gemini-3" also matches the
51
68
  # gemini-3.7/gemini-3.6/gemini-3.5-flash-lite client types)
52
69
  if any(
@@ -54,51 +71,64 @@ class AutoLLMClient(LLMClient):
54
71
  ): # e.g., gemini-3.7-flash, gemini-3-flash-preview, gemini-embedding-2
55
72
  from .gemini3_7 import Gemini3_7Client
56
73
 
57
- return Gemini3_7Client(model=model, api_key=api_key, base_url=base_url)
74
+ return Gemini3_7Client
58
75
  elif "claude" in client_type and (
59
76
  "4-6" in client_type or "4-7" in client_type or "4-8" in client_type or "-5" in client_type
60
77
  ): # the whole Claude 4.6+ series shares the unified client, e.g., claude-sonnet-4-6
61
78
  from .claude5 import Claude5Client
62
79
 
63
- return Claude5Client(model=model, api_key=api_key, base_url=base_url)
80
+ return Claude5Client
64
81
  elif "gpt-5.4" in client_type or "gpt-5.5" in client_type or "gpt-5.6" in client_type: # e.g., gpt-5.6
65
82
  from .gpt5_6 import GPT5_6Client
66
83
 
67
- return GPT5_6Client(model=model, api_key=api_key, base_url=base_url)
84
+ return GPT5_6Client
68
85
  elif "glm-5" in client_type: # the whole GLM series shares the unified client
69
86
  from .glm5_3 import GLM5_3Client
70
87
 
71
- return GLM5_3Client(model=model, api_key=api_key, base_url=base_url)
88
+ return GLM5_3Client
72
89
  elif "kimi-k3" in client_type or "kimi-k2.5" in client_type or "kimi-k2.6" in client_type:
73
90
  # the whole Kimi K2.5+ series shares the unified client
74
91
  from .kimi_k3 import KimiK3Client
75
92
 
76
- return KimiK3Client(model=model, api_key=api_key, base_url=base_url)
93
+ return KimiK3Client
77
94
  elif client_type == "minimax-m3":
78
95
  from .minimax_m3 import MiniMaxM3Client
79
96
 
80
- return MiniMaxM3Client(model=model, api_key=api_key, base_url=base_url)
97
+ return MiniMaxM3Client
81
98
  elif "deepseek-v4" in client_type:
82
99
  from .deepseek_v4 import DeepSeekV4Client
83
100
 
84
- return DeepSeekV4Client(model=model, api_key=api_key, base_url=base_url)
101
+ return DeepSeekV4Client
85
102
  elif "ant-messages" in client_type:
86
103
  from .ant_messages import AntMessagesClient
87
104
 
88
- return AntMessagesClient(model=model, api_key=api_key, base_url=base_url)
105
+ return AntMessagesClient
89
106
  elif "openai-responses" in client_type:
90
107
  from .openai_responses import OpenaiResponsesClient
91
108
 
92
- return OpenaiResponsesClient(model=model, api_key=api_key, base_url=base_url)
109
+ return OpenaiResponsesClient
93
110
  elif "openai" in client_type and "embedding" in client_type:
94
111
  from .openai_embedding import OpenaiEmbeddingClient
95
112
 
96
- return OpenaiEmbeddingClient(model=model, api_key=api_key, base_url=base_url)
113
+ return OpenaiEmbeddingClient
97
114
  elif "openai" in client_type and "embedding" not in client_type: # openai-chat, plus bare "openai" as alias
98
115
  from .openai_chat import OpenaiChatClient
99
116
 
100
- return OpenaiChatClient(model=model, api_key=api_key, base_url=base_url)
117
+ return OpenaiChatClient
101
118
  else:
119
+ return None
120
+
121
+ def _create_client_for_model(
122
+ self,
123
+ model: str,
124
+ api_key: str | None = None,
125
+ base_url: str | None = None,
126
+ client_type: str | None = None,
127
+ default_headers: dict[str, str] | None = None,
128
+ ) -> LLMClient:
129
+ """Create the appropriate client for the given model."""
130
+ client_class = self._client_class_for_model(client_type or model.lower())
131
+ if client_class is None:
102
132
  raise ValueError(
103
133
  f"{client_type} is not supported. "
104
134
  "Supported client types: minimax-m3, gemini-3.7, gemini-3.6, gemini-3, "
@@ -107,6 +137,8 @@ class AutoLLMClient(LLMClient):
107
137
  "openai-embedding, ant-messages, openai-responses, openai-chat."
108
138
  )
109
139
 
140
+ return client_class(model=model, api_key=api_key, base_url=base_url, default_headers=default_headers)
141
+
110
142
  def transform_uni_config_to_model_config(self, config: UniConfig) -> Any:
111
143
  """Delegate to underlying client's transform_uni_config_to_model_config."""
112
144
  return self._client.transform_uni_config_to_model_config(config)
@@ -165,3 +197,22 @@ class AutoLLMClient(LLMClient):
165
197
  def set_history(self, history: list[UniMessage]) -> None:
166
198
  """Set history in the underlying client."""
167
199
  self._client.set_history(history)
200
+
201
+ async def list_models(self) -> list[str]:
202
+ """
203
+ List the model ids the endpoint serves that the routed client can be used for.
204
+
205
+ A protocol client is chosen explicitly and speaks for whatever the endpoint serves, so its
206
+ listing is returned whole. A client deduced from a model id serves only the ids that deduce
207
+ back to it, so a gateway fronting many vendors is filtered down to that client's own models.
208
+
209
+ Returns:
210
+ list[str]: The model ids, in the order the endpoint returned them.
211
+ """
212
+ model_ids = await self._client.list_models()
213
+ protocol_classes = {self._client_class_for_model(name) for name in _PROTOCOL_CLIENT_TYPES}
214
+ if type(self._client) in protocol_classes:
215
+ return model_ids
216
+
217
+ client_class = type(self._client)
218
+ return [model_id for model_id in model_ids if self._client_class_for_model(model_id.lower()) is client_class]
@@ -175,6 +175,15 @@ class LLMClient(ABC):
175
175
  """
176
176
  pass
177
177
 
178
+ @abstractmethod
179
+ async def list_models(self) -> list[str]:
180
+ """
181
+ List the model ids the configured endpoint serves.
182
+
183
+ Returns:
184
+ list[str]: The model ids, in the order the endpoint returned them.
185
+ """
186
+
178
187
  async def streaming_response(
179
188
  self,
180
189
  messages: list[UniMessage],
@@ -23,7 +23,7 @@ from anthropic import AsyncAnthropic, AsyncAnthropicBedrock
23
23
  from anthropic.types.beta import BetaMessageParam, BetaRawMessageStreamEvent
24
24
 
25
25
  from ..base_client import LLMClient
26
- from ..errors import UnsupportedParameterError, parse_tool_call_arguments
26
+ from ..errors import UnsupportedOperationError, UnsupportedParameterError, parse_tool_call_arguments
27
27
  from ..types import (
28
28
  EventType,
29
29
  FinishReason,
@@ -36,6 +36,7 @@ from ..types import (
36
36
  UniMessage,
37
37
  UsageMetadata,
38
38
  )
39
+ from ..utils import is_debug_enabled
39
40
 
40
41
 
41
42
  REDACTED_THINKING = "_REDACTED_THINKING"
@@ -44,7 +45,13 @@ REDACTED_THINKING = "_REDACTED_THINKING"
44
45
  class Claude5Client(LLMClient):
45
46
  """Claude 5-specific LLM client implementation (also serves Claude 4.6 through 4.8)."""
46
47
 
47
- def __init__(self, model: str, api_key: str | None = None, base_url: str | None = None):
48
+ def __init__(
49
+ self,
50
+ model: str,
51
+ api_key: str | None = None,
52
+ base_url: str | None = None,
53
+ default_headers: dict[str, str] | None = None,
54
+ ):
48
55
  """Initialize Claude 5 client with model and API key."""
49
56
  self._model = model
50
57
  api_key = api_key or os.getenv("ANTHROPIC_API_KEY")
@@ -53,11 +60,14 @@ class Claude5Client(LLMClient):
53
60
  region = base_url.replace("bedrock://", "")
54
61
  access_key, secret_key = api_key.split(",")
55
62
  self._client = AsyncAnthropicBedrock(
56
- aws_secret_key=secret_key, aws_access_key=access_key, aws_region=region
63
+ aws_secret_key=secret_key,
64
+ aws_access_key=access_key,
65
+ aws_region=region,
66
+ default_headers=default_headers,
57
67
  )
58
68
  self._use_bedrock = True
59
69
  else:
60
- self._client = AsyncAnthropic(api_key=api_key, base_url=base_url)
70
+ self._client = AsyncAnthropic(api_key=api_key, base_url=base_url, default_headers=default_headers)
61
71
  self._use_bedrock = False
62
72
 
63
73
  self._history: list[UniMessage] = []
@@ -115,6 +125,8 @@ class Claude5Client(LLMClient):
115
125
  "thinking": {"type": "adaptive"},
116
126
  "output_config": {"effort": "high" if "4-6" in self._model else "xhigh"},
117
127
  },
128
+ # every model this client serves is 4.6 or later, and max spans that whole range
129
+ ThinkingLevel.MAX: {"thinking": {"type": "adaptive"}, "output_config": {"effort": "max"}},
118
130
  }
119
131
  return mapping.get(thinking_level)
120
132
 
@@ -354,9 +366,14 @@ class Claude5Client(LLMClient):
354
366
  # gateways that relabel it onto another event
355
367
  event_type = "unused"
356
368
 
357
- else:
369
+ elif is_debug_enabled():
358
370
  raise ValueError(f"Unknown output: {model_output}")
359
371
 
372
+ else:
373
+ # a gateway injects its own events (heartbeats, cost tickers) into the stream, and
374
+ # killing a long generation over one costs more than dropping it
375
+ event_type = "unused"
376
+
360
377
  return {
361
378
  "role": "assistant",
362
379
  "event_type": event_type,
@@ -465,3 +482,17 @@ class Claude5Client(LLMClient):
465
482
  "finish_reason": event["finish_reason"],
466
483
  }
467
484
  partial_usage = {}
485
+
486
+ async def list_models(self) -> list[str]:
487
+ """
488
+ List the model ids the configured endpoint serves.
489
+
490
+ Returns:
491
+ list[str]: The model ids, in the order the endpoint returned them.
492
+ """
493
+ if self._use_bedrock:
494
+ raise UnsupportedOperationError(
495
+ self.__class__.__name__, "list_models", "Bedrock does not support listing models."
496
+ )
497
+
498
+ return [model.id async for model in self._client.models.list()]
@@ -38,12 +38,18 @@ from ..types import (
38
38
  class DeepSeekV4Client(LLMClient):
39
39
  """DeepSeek V4-specific LLM client implementation using OpenAI-compatible Chat Completions."""
40
40
 
41
- def __init__(self, model: str, api_key: str | None = None, base_url: str | None = None):
41
+ def __init__(
42
+ self,
43
+ model: str,
44
+ api_key: str | None = None,
45
+ base_url: str | None = None,
46
+ default_headers: dict[str, str] | None = None,
47
+ ):
42
48
  """Initialize DeepSeek client with model, API key, and base URL."""
43
49
  self._model = model
44
50
  api_key = api_key or os.getenv("DEEPSEEK_API_KEY")
45
51
  base_url = base_url or os.getenv("DEEPSEEK_BASE_URL") or "https://api.deepseek.com"
46
- self._client = AsyncOpenAI(api_key=api_key, base_url=base_url)
52
+ self._client = AsyncOpenAI(api_key=api_key, base_url=base_url, default_headers=default_headers)
47
53
  self._history: list[UniMessage] = []
48
54
 
49
55
  def _convert_thinking_level_to_config(self, thinking_level: ThinkingLevel) -> dict[str, str]:
@@ -54,17 +60,24 @@ class DeepSeekV4Client(LLMClient):
54
60
  ThinkingLevel.MEDIUM: {"type": "enabled"},
55
61
  ThinkingLevel.HIGH: {"type": "enabled"},
56
62
  ThinkingLevel.XHIGH: {"type": "enabled"},
63
+ ThinkingLevel.MAX: {"type": "enabled"},
57
64
  }
58
65
  return mapping[thinking_level]
59
66
 
60
67
  def _convert_reasoning_effort(self, thinking_level: ThinkingLevel) -> str | None:
61
- """Convert ThinkingLevel enum to DeepSeek's reasoning_effort."""
68
+ """Convert ThinkingLevel enum to DeepSeek's reasoning_effort.
69
+
70
+ DeepSeek accepts low/high/max and maps medium and xhigh onto high server-side
71
+ (llmsdk_docs/deepseek_v4/docs/thinking-mode.md), so this sends the value the
72
+ server would settle on anyway.
73
+ """
62
74
  mapping = {
63
75
  ThinkingLevel.NONE: None,
64
- ThinkingLevel.LOW: "high",
76
+ ThinkingLevel.LOW: "low",
65
77
  ThinkingLevel.MEDIUM: "high",
66
78
  ThinkingLevel.HIGH: "high",
67
- ThinkingLevel.XHIGH: "max",
79
+ ThinkingLevel.XHIGH: "high",
80
+ ThinkingLevel.MAX: "max",
68
81
  }
69
82
  return mapping[thinking_level]
70
83
 
@@ -367,3 +380,12 @@ class DeepSeekV4Client(LLMClient):
367
380
  "finish_reason": partial_usage["finish_reason"],
368
381
  }
369
382
  partial_usage = {}
383
+
384
+ async def list_models(self) -> list[str]:
385
+ """
386
+ List the model ids the configured endpoint serves.
387
+
388
+ Returns:
389
+ list[str]: The model ids, in the order the endpoint returned them.
390
+ """
391
+ return [model.id async for model in self._client.models.list()]
@@ -43,6 +43,20 @@ class UnsupportedParameterError(AgentHubError):
43
43
  super().__init__(message)
44
44
 
45
45
 
46
+ class UnsupportedOperationError(AgentHubError):
47
+ """Raised when a client cannot perform an operation at all, whatever it is passed.
48
+
49
+ Distinct from UnsupportedParameterError, which rejects a UniConfig parameter value:
50
+ this one reports a capability the routed client does not have, such as listing models
51
+ through an SDK client that carries no models endpoint.
52
+ """
53
+
54
+ def __init__(self, client: str, operation: str, message: str) -> None:
55
+ self.client = client
56
+ self.operation = operation
57
+ super().__init__(message)
58
+
59
+
46
60
  class EmptyResponseError(AgentHubError):
47
61
  """Raised when a completed response carries no non-thinking content and no tool calls.
48
62
 
@@ -17,7 +17,7 @@ import json
17
17
  import mimetypes
18
18
  import os
19
19
  import re
20
- from typing import AsyncIterator
20
+ from typing import Any, AsyncIterator
21
21
 
22
22
  import httpx
23
23
  from google import genai
@@ -40,6 +40,7 @@ from ..types import (
40
40
  UniMessage,
41
41
  UsageMetadata,
42
42
  )
43
+ from ..utils import is_debug_enabled
43
44
 
44
45
 
45
46
  class Gemini3_7Client(LLMClient):
@@ -52,12 +53,23 @@ class Gemini3_7Client(LLMClient):
52
53
  applies that contract to the whole family: temperature is rejected everywhere.
53
54
  """
54
55
 
55
- def __init__(self, model: str, api_key: str | None = None, base_url: str | None = None):
56
+ def __init__(
57
+ self,
58
+ model: str,
59
+ api_key: str | None = None,
60
+ base_url: str | None = None,
61
+ default_headers: dict[str, str] | None = None,
62
+ ):
56
63
  """Initialize Gemini 3.7 client with model and API key."""
57
64
  self._model = model
58
65
  api_key = api_key or os.getenv("GEMINI_API_KEY")
59
66
  base_url = base_url or os.getenv("GEMINI_BASE_URL")
60
- http_options = {"base_url": base_url} if base_url else None
67
+ # the Gemini SDK carries connection headers inside http_options rather than its own argument
68
+ http_options: dict[str, Any] = {}
69
+ if base_url:
70
+ http_options["base_url"] = base_url
71
+ if default_headers:
72
+ http_options["headers"] = default_headers
61
73
  if api_key and api_key.startswith("{"):
62
74
  service_account_info = json.loads(api_key)
63
75
  credentials = service_account.Credentials.from_service_account_info(
@@ -68,10 +80,10 @@ class Gemini3_7Client(LLMClient):
68
80
  credentials=credentials,
69
81
  project=service_account_info["project_id"],
70
82
  location="global",
71
- http_options=http_options,
83
+ http_options=http_options or None,
72
84
  )
73
85
  else:
74
- self._client = genai.Client(api_key=api_key, http_options=http_options)
86
+ self._client = genai.Client(api_key=api_key, http_options=http_options or None)
75
87
 
76
88
  self._history: list[UniMessage] = []
77
89
 
@@ -142,6 +154,8 @@ class Gemini3_7Client(LLMClient):
142
154
  ThinkingLevel.MEDIUM: types.ThinkingLevel.MEDIUM,
143
155
  ThinkingLevel.HIGH: types.ThinkingLevel.HIGH,
144
156
  ThinkingLevel.XHIGH: types.ThinkingLevel.HIGH,
157
+ # Gemini stops at "high", so both top levels land there before per-model clamping
158
+ ThinkingLevel.MAX: types.ThinkingLevel.HIGH,
145
159
  }
146
160
  level = mapping.get(thinking_level)
147
161
  if level is None:
@@ -201,6 +215,8 @@ class Gemini3_7Client(LLMClient):
201
215
  "sampling parameters starting with the 3.6 generation.",
202
216
  )
203
217
 
218
+ # include_thoughts asks for thought summaries, but whether generateContent returns any
219
+ # is model-dependent (llmsdk_docs/gemini3_7/docs/thinking.md)
204
220
  thinking_summary = config.get("thinking_summary")
205
221
  thinking_level = config.get("thinking_level")
206
222
  if thinking_summary is not None or thinking_level is not None:
@@ -376,11 +392,6 @@ class Gemini3_7Client(LLMClient):
376
392
  usage_metadata: UsageMetadata | None = None
377
393
  finish_reason: FinishReason | None = None
378
394
 
379
- if not model_output.candidates and not model_output.usage_metadata:
380
- # gateways inject heartbeat chunks on long generations; they map to a chunk with
381
- # neither candidates nor usage, so there is nothing to emit
382
- event_type = "unused"
383
-
384
395
  if model_output.candidates:
385
396
  candidate = model_output.candidates[0]
386
397
  content = getattr(candidate, "content", None)
@@ -418,7 +429,7 @@ class Gemini3_7Client(LLMClient):
418
429
  )
419
430
  elif part.text is not None:
420
431
  content_items.append({"type": "text", "text": part.text, **self._part_fidelity(part)})
421
- else:
432
+ elif is_debug_enabled():
422
433
  raise ValueError(f"Unknown output: {part}")
423
434
 
424
435
  if candidate.finish_reason:
@@ -441,6 +452,11 @@ class Gemini3_7Client(LLMClient):
441
452
  "response_tokens": model_output.usage_metadata.candidates_token_count,
442
453
  }
443
454
 
455
+ if not content_items and usage_metadata is None and finish_reason is None:
456
+ # nothing was read out of the chunk, so there is nothing to emit: a gateway
457
+ # heartbeat looks like this, and so does any other chunk we take no value from
458
+ event_type = "unused"
459
+
444
460
  return {
445
461
  "role": "assistant",
446
462
  "event_type": event_type,
@@ -539,3 +555,13 @@ class Gemini3_7Client(LLMClient):
539
555
  }
540
556
 
541
557
  yield event
558
+
559
+ async def list_models(self) -> list[str]:
560
+ """
561
+ List the model ids the configured endpoint serves.
562
+
563
+ Returns:
564
+ list[str]: The model ids, in the order the endpoint returned them.
565
+ """
566
+ # the API returns path-qualified names: models/gemini-3.7-flash, publishers/google/models/...
567
+ return [model.name.split("/")[-1] async for model in await self._client.aio.models.list() if model.name]
@@ -43,12 +43,18 @@ class GLM5_3Client(LLMClient):
43
43
  contract differs per generation, handled model-by-model below.
44
44
  """
45
45
 
46
- def __init__(self, model: str, api_key: str | None = None, base_url: str | None = None):
46
+ def __init__(
47
+ self,
48
+ model: str,
49
+ api_key: str | None = None,
50
+ base_url: str | None = None,
51
+ default_headers: dict[str, str] | None = None,
52
+ ):
47
53
  """Initialize GLM client with model and API key."""
48
54
  self._model = model
49
55
  api_key = api_key or os.getenv("ZAI_API_KEY")
50
56
  base_url = base_url or os.getenv("ZAI_BASE_URL") or "https://api.z.ai/api/paas/v4/"
51
- self._client = AsyncOpenAI(api_key=api_key, base_url=base_url)
57
+ self._client = AsyncOpenAI(api_key=api_key, base_url=base_url, default_headers=default_headers)
52
58
  self._history: list[UniMessage] = []
53
59
 
54
60
  def _convert_thinking_level_to_config(self, thinking_level: ThinkingLevel) -> dict[str, str | bool]:
@@ -81,6 +87,7 @@ class GLM5_3Client(LLMClient):
81
87
  ThinkingLevel.MEDIUM: "high",
82
88
  ThinkingLevel.HIGH: "high",
83
89
  ThinkingLevel.XHIGH: "max",
90
+ ThinkingLevel.MAX: "max",
84
91
  }
85
92
  return mapping.get(thinking_level)
86
93
  if "glm-5.2" in model:
@@ -90,6 +97,7 @@ class GLM5_3Client(LLMClient):
90
97
  ThinkingLevel.MEDIUM: "medium",
91
98
  ThinkingLevel.HIGH: "high",
92
99
  ThinkingLevel.XHIGH: "xhigh",
100
+ ThinkingLevel.MAX: "max",
93
101
  }
94
102
  return mapping.get(thinking_level)
95
103
  return None
@@ -435,3 +443,12 @@ class GLM5_3Client(LLMClient):
435
443
  "finish_reason": partial_usage["finish_reason"],
436
444
  }
437
445
  partial_usage = {}
446
+
447
+ async def list_models(self) -> list[str]:
448
+ """
449
+ List the model ids the configured endpoint serves.
450
+
451
+ Returns:
452
+ list[str]: The model ids, in the order the endpoint returned them.
453
+ """
454
+ return [model.id async for model in self._client.models.list()]