llm-interface 0.1.13__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (24) hide show
  1. {llm_interface-0.1.13 → llm_interface-0.2.2}/PKG-INFO +4 -2
  2. {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/__init__.py +1 -1
  3. llm_interface-0.2.2/llm_interface/anthropic.py +506 -0
  4. {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/errors.py +1 -0
  5. {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/llm_config.py +51 -5
  6. {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/llm_interface.py +339 -139
  7. {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/llm_tool.py +69 -14
  8. {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/ollama.py +28 -0
  9. {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/openai.py +175 -33
  10. llm_interface-0.2.2/llm_interface/openai_responses.py +351 -0
  11. {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/remote_ollama.py +5 -0
  12. {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/testing/mock_llm.py +3 -0
  13. {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/token_usage.py +17 -2
  14. {llm_interface-0.1.13 → llm_interface-0.2.2}/pyproject.toml +1 -1
  15. llm_interface-0.1.13/llm_interface/anthropic.py +0 -317
  16. {llm_interface-0.1.13 → llm_interface-0.2.2}/LICENSE +0 -0
  17. {llm_interface-0.1.13 → llm_interface-0.2.2}/README.md +0 -0
  18. {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/gemini.py +0 -0
  19. {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/openrouter.py +0 -0
  20. {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/pydantic_output_parser.py +0 -0
  21. {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/ssh.py +0 -0
  22. {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/testing/__init__.py +0 -0
  23. {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/testing/helpers.py +0 -0
  24. {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/utils.py +0 -0
@@ -1,8 +1,9 @@
1
- Metadata-Version: 2.3
1
+ Metadata-Version: 2.4
2
2
  Name: llm-interface
3
- Version: 0.1.13
3
+ Version: 0.2.2
4
4
  Summary: A flexible interface for working with various LLM providers
5
5
  License: Apache-2.0
6
+ License-File: LICENSE
6
7
  Author: Niels Provos
7
8
  Author-email: provos@gmail.com
8
9
  Requires-Python: >=3.10,<4.0
@@ -12,6 +13,7 @@ Classifier: Programming Language :: Python :: 3.10
12
13
  Classifier: Programming Language :: Python :: 3.11
13
14
  Classifier: Programming Language :: Python :: 3.12
14
15
  Classifier: Programming Language :: Python :: 3.13
16
+ Classifier: Programming Language :: Python :: 3.14
15
17
  Requires-Dist: anthropic (>=0.34.2)
16
18
  Requires-Dist: diskcache (>=5.6.3)
17
19
  Requires-Dist: google-genai (>=1.2.0,<2.0.0)
@@ -5,7 +5,7 @@ from .llm_interface import LLMInterface, ModelError
5
5
  from .llm_tool import Tool, tool
6
6
  from .token_usage import TokenUsage
7
7
 
8
- __version__ = "0.1.12"
8
+ __version__ = "0.2.2"
9
9
  __all__ = [
10
10
  "LLMInterface",
11
11
  "llm_from_config",
@@ -0,0 +1,506 @@
1
+ # Copyright 2024 Niels Provos
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+ import json
15
+ import logging
16
+ from datetime import datetime
17
+ from typing import Any, Dict, List, Optional
18
+
19
+ import requests
20
+ from anthropic import Anthropic, APIConnectionError, APIError, APITimeoutError
21
+ from ollama import ListResponse
22
+
23
+ from . import errors
24
+ from .utils import encode_image_to_base64
25
+
26
+
27
+ def translate_tools_for_anthropic(tools: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
28
+ """
29
+ Translate a list of tools from Ollama/API format to Anthropic format.
30
+
31
+ Args:
32
+ tools (List[Tool]): List of tool objects from the Ollama/API.
33
+
34
+ Returns:
35
+ List[Dict[str, Any]]: Translated tools ready for Anthropic API consumption.
36
+ """
37
+ anthropic_tools = []
38
+
39
+ for tool in tools:
40
+ # Extract the function from the tool
41
+ function = tool["function"]
42
+
43
+ # Assuming Tool objects have keys 'name', 'description', and 'parameters' which is a dict
44
+ input_schema = {
45
+ "type": "object",
46
+ "properties": function["parameters"]["properties"],
47
+ "required": function["parameters"]["required"],
48
+ }
49
+ translated_tool = {
50
+ "name": function["name"],
51
+ "description": function["description"],
52
+ "input_schema": input_schema,
53
+ }
54
+
55
+ # Strict tool use: mirror OpenAI's `strict` flag onto Anthropic's schema.
56
+ # Anthropic expects `strict` alongside the tool definition and
57
+ # `additionalProperties: False` inside the input_schema itself.
58
+ if function.get("strict"):
59
+ translated_tool["strict"] = True
60
+ input_schema["additionalProperties"] = False
61
+
62
+ anthropic_tools.append(translated_tool)
63
+
64
+ return anthropic_tools
65
+
66
+
67
+ def translate_messages_for_anthropic(
68
+ messages: List[Dict[str, Any]],
69
+ ) -> List[Dict[str, Any]]:
70
+ """
71
+ Translate messages from Ollama/API format to Anthropic format.
72
+
73
+ An assistant message carrying N tool_calls becomes ONE assistant message
74
+ whose content is an optional text block (only when the message has
75
+ non-empty content) followed by N `tool_use` blocks. Consecutive `tool` role
76
+ messages are merged into a single `user` message containing one
77
+ `tool_result` block per call - Anthropic requires every tool_result for a
78
+ turn to be returned together. Plain user/assistant/system-free
79
+ conversations pass through unchanged, so this function is safe to call
80
+ unconditionally.
81
+
82
+ Args:
83
+ messages (List[Dict[str, Any]]): List of message dictionaries in Ollama format
84
+
85
+ Returns:
86
+ List[Dict[str, Any]]: Translated messages in Anthropic format
87
+ """
88
+ translated_messages: List[Dict[str, Any]] = []
89
+ # Reference to the content list of the most recently appended tool_result
90
+ # group, so consecutive tool messages get merged into one user message.
91
+ current_tool_result_group: Optional[List[Dict[str, Any]]] = None
92
+
93
+ for msg in messages:
94
+ if "images" in msg and msg["images"]:
95
+ content = [{"type": "text", "text": msg["content"]}]
96
+ for image in msg["images"]:
97
+ content.append(
98
+ {
99
+ "type": "image",
100
+ "source": {
101
+ "type": "base64",
102
+ "data": encode_image_to_base64(image),
103
+ "media_type": "image/jpeg",
104
+ },
105
+ }
106
+ )
107
+ translated_messages.append({"role": "user", "content": content})
108
+ current_tool_result_group = None
109
+
110
+ elif msg["role"] == "user":
111
+ # Regular user messages pass through unchanged
112
+ translated_messages.append({"role": "user", "content": msg["content"]})
113
+ current_tool_result_group = None
114
+
115
+ elif msg["role"] == "assistant" and msg.get("tool_calls"):
116
+ # Convert every tool call made in this turn into one tool_use block
117
+ # on a single assistant message.
118
+ content = []
119
+ if msg.get("content"):
120
+ content.append({"type": "text", "text": msg["content"]})
121
+
122
+ for tool_call in msg["tool_calls"]:
123
+ function = tool_call["function"]
124
+ tool_input = function["arguments"]
125
+ if isinstance(tool_input, str):
126
+ try:
127
+ tool_input = json.loads(tool_input)
128
+ except json.JSONDecodeError:
129
+ # keep the transcript replayable: Anthropic needs an
130
+ # object here, and the tool result already reports the
131
+ # parse failure to the model
132
+ logging.warning(
133
+ "Unparseable tool arguments for %s; sending raw string",
134
+ function["name"],
135
+ )
136
+ tool_input = {"raw_arguments": tool_input}
137
+ content.append(
138
+ {
139
+ "type": "tool_use",
140
+ "id": tool_call["id"],
141
+ "name": function["name"],
142
+ "input": tool_input,
143
+ }
144
+ )
145
+
146
+ translated_messages.append({"role": "assistant", "content": content})
147
+ current_tool_result_group = None
148
+
149
+ elif msg["role"] == "tool":
150
+ # Convert tool response to Anthropic's tool_result format, merging
151
+ # consecutive tool messages into one user message.
152
+ tool_result: Dict[str, Any] = {
153
+ "type": "tool_result",
154
+ "tool_use_id": msg["tool_call_id"],
155
+ "content": msg["content"],
156
+ }
157
+ if msg.get("is_error"):
158
+ tool_result["is_error"] = True
159
+
160
+ if current_tool_result_group is not None:
161
+ current_tool_result_group.append(tool_result)
162
+ else:
163
+ current_tool_result_group = [tool_result]
164
+ translated_messages.append(
165
+ {"role": "user", "content": current_tool_result_group}
166
+ )
167
+
168
+ elif msg["role"] == "assistant":
169
+ # Assistant messages with plain string content pass through.
170
+ translated_messages.append(msg)
171
+ current_tool_result_group = None
172
+ else:
173
+ raise ValueError(f"Unknown message role: {msg['role']}")
174
+
175
+ return translated_messages
176
+
177
+
178
+ def convert_anthropic_models_to_ollama_response(
179
+ models_data: Dict[str, Any],
180
+ ) -> ListResponse:
181
+ """
182
+ Converts Anthropic model list API response to Ollama format.
183
+
184
+ Args:
185
+ models_data: The response from Anthropic's models API endpoint.
186
+
187
+ Returns:
188
+ An instance of ollama's ListResponse.
189
+ """
190
+ ollama_models = []
191
+ for model_data in models_data["data"]:
192
+ # Convert creation time from ISO format to datetime
193
+ created_at = datetime.fromisoformat(
194
+ model_data["created_at"].replace("Z", "+00:00")
195
+ )
196
+
197
+ model = {
198
+ "model": model_data["id"],
199
+ "modified_at": created_at,
200
+ "digest": "unknown",
201
+ "size": 0,
202
+ "details": {
203
+ "parent_model": "",
204
+ "format": "unknown",
205
+ "family": "claude",
206
+ "families": ["claude"],
207
+ "parameter_size": "unknown",
208
+ "quantization_level": "unknown",
209
+ "display_name": model_data["display_name"],
210
+ },
211
+ }
212
+ ollama_models.append(ListResponse.Model(**model))
213
+
214
+ return ListResponse(models=ollama_models)
215
+
216
+
217
+ def _usage_int(value: Any) -> int:
218
+ """Usage counters that the SDK may leave unset (or that tests mock) become 0."""
219
+ return value if isinstance(value, int) else 0
220
+
221
+
222
+ def _estimate_thinking_tokens(response: Any) -> int:
223
+ """Thinking tokens are billed as output but the API does not count them
224
+ separately. When the response carries thinking blocks, estimate them as the
225
+ output tokens that the visible text and tool-call blocks do not account for."""
226
+ blocks = list(getattr(response, "content", None) or [])
227
+ if not any(getattr(block, "type", None) == "thinking" for block in blocks):
228
+ return 0
229
+ visible_chars = 0
230
+ for block in blocks:
231
+ block_type = getattr(block, "type", None)
232
+ if block_type == "text":
233
+ visible_chars += len(getattr(block, "text", "") or "")
234
+ elif block_type == "tool_use":
235
+ try:
236
+ visible_chars += len(json.dumps(getattr(block, "input", None)))
237
+ except (TypeError, ValueError):
238
+ pass
239
+ output_tokens = _usage_int(getattr(response.usage, "output_tokens", 0))
240
+ return max(0, output_tokens - visible_chars // 4)
241
+
242
+
243
+ class AnthropicWrapper:
244
+ def __init__(
245
+ self,
246
+ api_key: str,
247
+ max_tokens: int = 4096,
248
+ timeout: float = 600.0,
249
+ prompt_caching: bool = True,
250
+ thinking: Optional[Dict[str, Any]] = None,
251
+ effort: Optional[str] = None,
252
+ ):
253
+ """
254
+ Args:
255
+ api_key (str): Anthropic API key.
256
+ max_tokens (int): Default max_tokens for requests. Requests always
257
+ stream, so large values are safe.
258
+ timeout (float): Request timeout in seconds.
259
+ prompt_caching (bool): When True (the default), every request carries
260
+ top-level `cache_control={"type": "ephemeral"}`, which auto-caches
261
+ the last cacheable block - useful for a growing tool-call
262
+ transcript.
263
+ thinking (Optional[Dict[str, Any]]): Extended thinking configuration,
264
+ e.g. {"type": "adaptive"}. When set, `temperature` is never sent
265
+ (current models reject it together with thinking) and
266
+ `budget_tokens` is never used.
267
+ effort (Optional[str]): One of low|medium|high|xhigh|max. Forwarded as
268
+ `output_config.effort`.
269
+ """
270
+ self.client = Anthropic(api_key=api_key, timeout=timeout)
271
+ self.api_key = api_key
272
+ self.max_tokens = max_tokens
273
+ self.prompt_caching = prompt_caching
274
+ self.thinking = thinking
275
+ self.effort = effort
276
+
277
+ def chat(
278
+ self,
279
+ messages: List[Dict[str, str]],
280
+ tools: Optional[List[Dict[str, Any]]] = None,
281
+ **kwargs,
282
+ ) -> Dict[str, Any]:
283
+ """
284
+ Conduct a chat conversation using the Anthropic API.
285
+
286
+ Args:
287
+ messages (list[Mapping[str, str]]): A list of message dictionaries, each containing 'role' and 'content'.
288
+ tools (Optional[List[Dict[str, Any]]]): Tool definitions in Ollama/OpenAI format.
289
+ **kwargs: Additional arguments, including:
290
+ - model (str): The Anthropic model to use.
291
+ - max_tokens (int): Overrides the instance's max_tokens.
292
+ - options (dict): May contain "temperature".
293
+ - response_schema (Type[BaseModel]): When provided, it is passed as
294
+ `output_format` so the streamed final message carries a validated
295
+ `parsed_output`.
296
+ - tool_choice (str | dict): "none" -> {"type": "none"},
297
+ "auto" -> {"type": "auto"}, or a dict passed through as-is.
298
+
299
+ Returns:
300
+ A dictionary containing the Anthropic response formatted to match Ollama's expected output.
301
+ """
302
+ # Extract the system message from the messages and prepare it as a separate argument
303
+ system_message = next(
304
+ (msg["content"] for msg in messages if msg["role"] == "system"), None
305
+ )
306
+
307
+ # Filter out the system message to prevent duplication if it's not needed in the messages parameter
308
+ filtered_messages = [msg for msg in messages if msg["role"] != "system"]
309
+
310
+ # Always translate: this normalizes tool_calls/tool-result/image
311
+ # messages into Anthropic's format, and is a no-op for a plain
312
+ # user/assistant conversation.
313
+ filtered_messages = translate_messages_for_anthropic(filtered_messages)
314
+
315
+ # Common parameters
316
+ params: Dict[str, Any] = {
317
+ "max_tokens": kwargs.get("max_tokens", self.max_tokens),
318
+ "messages": filtered_messages,
319
+ "model": kwargs.get("model", "claude-3-5-sonnet-20240620"),
320
+ }
321
+
322
+ # Only include system parameter if it has a value
323
+ if system_message is not None:
324
+ params["system"] = system_message
325
+
326
+ # Extended thinking / effort.
327
+ if self.thinking is not None:
328
+ params["thinking"] = self.thinking
329
+ if self.effort is not None:
330
+ params["output_config"] = {"effort": self.effort}
331
+
332
+ # Conditionally add temperature if it exists in kwargs. Current models
333
+ # reject `temperature` when adaptive thinking is configured, so only
334
+ # send it when the caller explicitly asked for it AND thinking is off.
335
+ if (
336
+ self.thinking is None
337
+ and "options" in kwargs
338
+ and "temperature" in kwargs["options"]
339
+ ):
340
+ params["temperature"] = kwargs["options"]["temperature"]
341
+
342
+ if tools:
343
+ # Translate tools into Anthropic format
344
+ anthropic_tools = translate_tools_for_anthropic(tools)
345
+ params["tools"] = anthropic_tools
346
+
347
+ # tool_choice: "none"/"auto" strings map to Anthropic's dict form; a
348
+ # dict is passed through untouched.
349
+ tool_choice = kwargs.get("tool_choice")
350
+ if tool_choice == "none":
351
+ params["tool_choice"] = {"type": "none"}
352
+ elif tool_choice == "auto":
353
+ params["tool_choice"] = {"type": "auto"}
354
+ elif isinstance(tool_choice, dict):
355
+ params["tool_choice"] = tool_choice
356
+
357
+ response_schema = kwargs.get("response_schema")
358
+
359
+ # Every request streams: `messages.stream` accepts `output_format`
360
+ # (structured output, validated into `parsed_output`) together with
361
+ # `cache_control`, and streaming avoids the SDK's timeout guard on
362
+ # large `max_tokens` values.
363
+ if response_schema is not None:
364
+ params["output_format"] = response_schema
365
+ if self.prompt_caching:
366
+ # Auto-caches the last cacheable block - what a growing
367
+ # tool-call transcript wants.
368
+ params["cache_control"] = {"type": "ephemeral"}
369
+
370
+ try:
371
+ with self.client.messages.stream(**params) as stream:
372
+ response = stream.get_final_message()
373
+
374
+ # Extract usage information
375
+ usage = response.usage
376
+ usage_info = {
377
+ "prompt_tokens": usage.input_tokens,
378
+ "completion_tokens": usage.output_tokens,
379
+ "cached_tokens": usage.cache_read_input_tokens or 0,
380
+ "cache_creation_tokens": _usage_int(
381
+ getattr(usage, "cache_creation_input_tokens", 0)
382
+ ),
383
+ "total_tokens": usage.input_tokens + usage.output_tokens,
384
+ "reasoning_tokens": _estimate_thinking_tokens(response),
385
+ }
386
+
387
+ if response.stop_reason == "refusal":
388
+ stop_details = response.stop_details
389
+ explanation = (
390
+ stop_details.explanation if stop_details is not None else None
391
+ )
392
+ return {
393
+ "refusal": explanation or "refused",
394
+ "content": None,
395
+ "done": False,
396
+ }
397
+
398
+ if response.stop_reason == "max_tokens":
399
+ return {
400
+ "error": "Response exceeded the maximum allowed length.",
401
+ "error_type": errors.LENGTH,
402
+ "content": None,
403
+ "done": False,
404
+ "usage": None,
405
+ }
406
+
407
+ # Handle tool calls if present. This shape is returned whether or
408
+ # not `response_schema` was requested (parsed_output is unused
409
+ # here and stays None on the response in that case).
410
+ tool_use_blocks = [
411
+ block for block in response.content if block.type == "tool_use"
412
+ ]
413
+ if tool_use_blocks:
414
+ return {
415
+ "message": {
416
+ "content": "",
417
+ "tool_calls": [
418
+ {
419
+ "id": tool_block.id,
420
+ "name": tool_block.name,
421
+ "arguments": tool_block.input, # Anthropic uses 'input' instead of 'arguments'
422
+ }
423
+ for tool_block in tool_use_blocks
424
+ ],
425
+ },
426
+ "usage": usage_info,
427
+ "done": response.stop_reason == "end_turn",
428
+ }
429
+
430
+ if response_schema is not None:
431
+ # `output_format` makes the final message a ParsedMessage with
432
+ # `parsed_output`; `generate_pydantic` accepts a BaseModel here.
433
+ return {
434
+ "message": {"content": getattr(response, "parsed_output", None)},
435
+ "usage": usage_info,
436
+ "done": response.stop_reason == "end_turn",
437
+ }
438
+
439
+ # Extract content blocks as text and simulate Ollama-like response
440
+ content = "".join(
441
+ block.text for block in response.content if block.type == "text"
442
+ )
443
+
444
+ return {
445
+ "message": {"content": content},
446
+ "usage": usage_info,
447
+ "done": response.stop_reason == "end_turn",
448
+ }
449
+
450
+ except APITimeoutError as e:
451
+ error_message = f"Anthropic API timeout: {str(e)}"
452
+ logging.error(error_message)
453
+ return {
454
+ "error": error_message,
455
+ "error_type": errors.TIMEOUT,
456
+ "content": None,
457
+ "done": False,
458
+ "usage": None,
459
+ }
460
+ except APIConnectionError as e:
461
+ error_message = f"Anthropic API connection error: {str(e)}"
462
+ logging.error(error_message)
463
+ return {
464
+ "error": error_message,
465
+ "error_type": errors.CONNECTION,
466
+ "content": None,
467
+ "done": False,
468
+ "usage": None,
469
+ }
470
+ except APIError as e:
471
+ error_message = f"Anthropic API error: {str(e)}"
472
+ logging.error(error_message)
473
+ return {
474
+ "error": error_message,
475
+ "error_type": errors.PROVIDER_SPECIFIC,
476
+ "content": None,
477
+ "done": False,
478
+ "usage": None,
479
+ }
480
+ except Exception as e:
481
+ # The streaming accumulator validates structured output as blocks
482
+ # complete and raises a pydantic ValidationError when a text block
483
+ # is empty or truncated (for example when thinking exhausted
484
+ # max_tokens). Surface it as a model error so generate_pydantic
485
+ # can retry instead of crashing the caller.
486
+ error_message = f"Anthropic response could not be parsed: {str(e)}"
487
+ logging.error(error_message)
488
+ return {
489
+ "error": error_message,
490
+ "error_type": errors.PROVIDER_SPECIFIC,
491
+ "content": None,
492
+ "done": False,
493
+ "usage": None,
494
+ }
495
+
496
+ def list(self) -> ListResponse:
497
+ """
498
+ Returns a list of available Anthropic models in Ollama format.
499
+ Uses the Anthropic API to get the current list of models.
500
+ """
501
+ headers = {"x-api-key": self.api_key, "anthropic-version": "2023-06-01"}
502
+
503
+ response = requests.get("https://api.anthropic.com/v1/models", headers=headers)
504
+ response.raise_for_status()
505
+
506
+ return convert_anthropic_models_to_ollama_response(response.json())
@@ -4,3 +4,4 @@ CONTENT_FILTER = "content_filter"
4
4
  PROVIDER_SPECIFIC = "provider_specific"
5
5
  HTTP = "http"
6
6
  CONNECTION = "connection"
7
+ RATE_LIMIT = "rate_limit"
@@ -1,12 +1,13 @@
1
1
  import os
2
2
  import re
3
3
  from datetime import datetime
4
- from typing import Literal, Optional
4
+ from typing import Any, Dict, Literal, Optional
5
5
 
6
6
  from .anthropic import AnthropicWrapper
7
7
  from .gemini import GeminiWrapper
8
8
  from .llm_interface import LLMInterface
9
9
  from .openai import OpenAIWrapper
10
+ from .openai_responses import OpenAIResponsesWrapper
10
11
  from .openrouter import OpenRouterWrapper
11
12
  from .remote_ollama import RemoteOllama
12
13
  from .ssh import SSHConnection
@@ -33,6 +34,15 @@ def supports_structured_output(model_name: str) -> bool:
33
34
  Returns:
34
35
  bool: Whether the model supports structured outputs
35
36
  """
37
+ # Check if this is a GPT model with version 5 or higher
38
+ # This handles both base models and dated variants (e.g., gpt-5, gpt-5-mini, gpt-5-2025-01-01)
39
+ if model_name.startswith("gpt-"):
40
+ # Extract the major version after "gpt-" (gpt-5, gpt-5-mini, gpt-5.6-luna)
41
+ parts = model_name[4:].split("-")
42
+ major = re.match(r"(\d+)(?:\.\d+)?$", parts[0]) if parts else None
43
+ if major and int(major.group(1)) >= 5:
44
+ return True
45
+
36
46
  # Models that always support structured outputs (no date requirements)
37
47
  base_models = {
38
48
  "gpt-4o",
@@ -83,6 +93,11 @@ def llm_from_config(
83
93
  timeout: float = 600.0,
84
94
  json_mode: Optional[bool] = None,
85
95
  structured_outputs: Optional[bool] = None,
96
+ thinking: Optional[Dict[str, Any]] = None,
97
+ effort: Optional[str] = None,
98
+ prompt_caching: bool = True,
99
+ max_tool_rounds: int = 5,
100
+ openai_api: Literal["responses", "chat"] = "responses",
86
101
  ) -> LLMInterface:
87
102
  """
88
103
  Creates and configures a language model interface based on specified provider and parameters.
@@ -103,6 +118,19 @@ def llm_from_config(
103
118
  timeout (float): Timeout in seconds for model requests. Defaults to 600.0.
104
119
  json_mode (Optional[bool]): Whether to override JSON mode support. Defaults to None.
105
120
  structured_outputs (Optional[bool]): Whether to override structured output support. Defaults to None.
121
+ thinking (Optional[Dict[str, Any]]): Extended thinking configuration forwarded to
122
+ `AnthropicWrapper` (e.g. {"type": "adaptive"}). Only used by the "anthropic" provider.
123
+ effort (Optional[str]): Effort level forwarded to `AnthropicWrapper` as
124
+ `output_config.effort` (low|medium|high|xhigh|max) or to `OpenAIWrapper`
125
+ as `reasoning_effort` (none|low|medium|high|xhigh). Used by the
126
+ "anthropic" and "openai" providers.
127
+ prompt_caching (bool): Whether `AnthropicWrapper` should send top-level `cache_control`.
128
+ Defaults to True. Only used by the "anthropic" provider.
129
+ max_tool_rounds (int): Maximum number of tool-call round-trips forwarded to
130
+ `LLMInterface`. Defaults to 5. Used by the "openai" and "anthropic" providers.
131
+ openai_api (Literal["responses", "chat"]): Which OpenAI API the "openai"
132
+ provider talks to. "responses" (the default) supports reasoning
133
+ together with function tools; "chat" is the Chat Completions API.
106
134
 
107
135
  Returns:
108
136
  LLMInterface: Configured interface for interacting with the specified LLM.
@@ -132,8 +160,14 @@ def llm_from_config(
132
160
  api_key = os.getenv("OPENAI_API_KEY")
133
161
  if api_key is None:
134
162
  raise ValueError("OPENAI_API_KEY not found in environment variables")
135
- wrapper = OpenAIWrapper(
136
- api_key=api_key, max_tokens=max_tokens, timeout=timeout
163
+ wrapper_class = (
164
+ OpenAIResponsesWrapper if openai_api == "responses" else OpenAIWrapper
165
+ )
166
+ wrapper = wrapper_class(
167
+ api_key=api_key,
168
+ max_tokens=max_tokens,
169
+ timeout=timeout,
170
+ reasoning_effort=effort,
137
171
  )
138
172
 
139
173
  support_structured_outputs = supports_structured_output(model_name)
@@ -149,19 +183,31 @@ def llm_from_config(
149
183
  support_system_prompt=support_system_prompt,
150
184
  use_cache=use_cache,
151
185
  timeout=timeout,
186
+ max_tool_rounds=max_tool_rounds,
152
187
  )
153
188
  case "anthropic":
154
189
  api_key = os.getenv("ANTHROPIC_API_KEY")
155
190
  if api_key is None:
156
191
  raise ValueError("ANTHROPIC_API_KEY not found in environment variables")
157
- wrapper = AnthropicWrapper(api_key=api_key, max_tokens=max_tokens)
192
+ wrapper = AnthropicWrapper(
193
+ api_key=api_key,
194
+ max_tokens=max_tokens,
195
+ timeout=timeout,
196
+ prompt_caching=prompt_caching,
197
+ thinking=thinking,
198
+ effort=effort,
199
+ )
158
200
  llm = LLMInterface(
159
201
  model_name=model_name,
160
202
  log_dir=log_dir,
161
203
  client=wrapper,
162
- support_json_mode=False,
204
+ # Anthropic's structured-output path (client.messages.parse) and
205
+ # response_schema handling route through these flags.
206
+ support_json_mode=True,
207
+ support_structured_outputs=True,
163
208
  use_cache=use_cache,
164
209
  timeout=timeout,
210
+ max_tool_rounds=max_tool_rounds,
165
211
  )
166
212
  case "gemini":
167
213
  api_key = os.getenv("GEMINI_API_KEY")