llm-interface 0.1.13__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {llm_interface-0.1.13 → llm_interface-0.2.2}/PKG-INFO +4 -2
- {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/__init__.py +1 -1
- llm_interface-0.2.2/llm_interface/anthropic.py +506 -0
- {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/errors.py +1 -0
- {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/llm_config.py +51 -5
- {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/llm_interface.py +339 -139
- {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/llm_tool.py +69 -14
- {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/ollama.py +28 -0
- {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/openai.py +175 -33
- llm_interface-0.2.2/llm_interface/openai_responses.py +351 -0
- {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/remote_ollama.py +5 -0
- {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/testing/mock_llm.py +3 -0
- {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/token_usage.py +17 -2
- {llm_interface-0.1.13 → llm_interface-0.2.2}/pyproject.toml +1 -1
- llm_interface-0.1.13/llm_interface/anthropic.py +0 -317
- {llm_interface-0.1.13 → llm_interface-0.2.2}/LICENSE +0 -0
- {llm_interface-0.1.13 → llm_interface-0.2.2}/README.md +0 -0
- {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/gemini.py +0 -0
- {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/openrouter.py +0 -0
- {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/pydantic_output_parser.py +0 -0
- {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/ssh.py +0 -0
- {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/testing/__init__.py +0 -0
- {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/testing/helpers.py +0 -0
- {llm_interface-0.1.13 → llm_interface-0.2.2}/llm_interface/utils.py +0 -0
|
@@ -1,8 +1,9 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
2
|
Name: llm-interface
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: A flexible interface for working with various LLM providers
|
|
5
5
|
License: Apache-2.0
|
|
6
|
+
License-File: LICENSE
|
|
6
7
|
Author: Niels Provos
|
|
7
8
|
Author-email: provos@gmail.com
|
|
8
9
|
Requires-Python: >=3.10,<4.0
|
|
@@ -12,6 +13,7 @@ Classifier: Programming Language :: Python :: 3.10
|
|
|
12
13
|
Classifier: Programming Language :: Python :: 3.11
|
|
13
14
|
Classifier: Programming Language :: Python :: 3.12
|
|
14
15
|
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
15
17
|
Requires-Dist: anthropic (>=0.34.2)
|
|
16
18
|
Requires-Dist: diskcache (>=5.6.3)
|
|
17
19
|
Requires-Dist: google-genai (>=1.2.0,<2.0.0)
|
|
@@ -0,0 +1,506 @@
|
|
|
1
|
+
# Copyright 2024 Niels Provos
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
import json
|
|
15
|
+
import logging
|
|
16
|
+
from datetime import datetime
|
|
17
|
+
from typing import Any, Dict, List, Optional
|
|
18
|
+
|
|
19
|
+
import requests
|
|
20
|
+
from anthropic import Anthropic, APIConnectionError, APIError, APITimeoutError
|
|
21
|
+
from ollama import ListResponse
|
|
22
|
+
|
|
23
|
+
from . import errors
|
|
24
|
+
from .utils import encode_image_to_base64
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def translate_tools_for_anthropic(tools: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
|
28
|
+
"""
|
|
29
|
+
Translate a list of tools from Ollama/API format to Anthropic format.
|
|
30
|
+
|
|
31
|
+
Args:
|
|
32
|
+
tools (List[Tool]): List of tool objects from the Ollama/API.
|
|
33
|
+
|
|
34
|
+
Returns:
|
|
35
|
+
List[Dict[str, Any]]: Translated tools ready for Anthropic API consumption.
|
|
36
|
+
"""
|
|
37
|
+
anthropic_tools = []
|
|
38
|
+
|
|
39
|
+
for tool in tools:
|
|
40
|
+
# Extract the function from the tool
|
|
41
|
+
function = tool["function"]
|
|
42
|
+
|
|
43
|
+
# Assuming Tool objects have keys 'name', 'description', and 'parameters' which is a dict
|
|
44
|
+
input_schema = {
|
|
45
|
+
"type": "object",
|
|
46
|
+
"properties": function["parameters"]["properties"],
|
|
47
|
+
"required": function["parameters"]["required"],
|
|
48
|
+
}
|
|
49
|
+
translated_tool = {
|
|
50
|
+
"name": function["name"],
|
|
51
|
+
"description": function["description"],
|
|
52
|
+
"input_schema": input_schema,
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
# Strict tool use: mirror OpenAI's `strict` flag onto Anthropic's schema.
|
|
56
|
+
# Anthropic expects `strict` alongside the tool definition and
|
|
57
|
+
# `additionalProperties: False` inside the input_schema itself.
|
|
58
|
+
if function.get("strict"):
|
|
59
|
+
translated_tool["strict"] = True
|
|
60
|
+
input_schema["additionalProperties"] = False
|
|
61
|
+
|
|
62
|
+
anthropic_tools.append(translated_tool)
|
|
63
|
+
|
|
64
|
+
return anthropic_tools
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def translate_messages_for_anthropic(
|
|
68
|
+
messages: List[Dict[str, Any]],
|
|
69
|
+
) -> List[Dict[str, Any]]:
|
|
70
|
+
"""
|
|
71
|
+
Translate messages from Ollama/API format to Anthropic format.
|
|
72
|
+
|
|
73
|
+
An assistant message carrying N tool_calls becomes ONE assistant message
|
|
74
|
+
whose content is an optional text block (only when the message has
|
|
75
|
+
non-empty content) followed by N `tool_use` blocks. Consecutive `tool` role
|
|
76
|
+
messages are merged into a single `user` message containing one
|
|
77
|
+
`tool_result` block per call - Anthropic requires every tool_result for a
|
|
78
|
+
turn to be returned together. Plain user/assistant/system-free
|
|
79
|
+
conversations pass through unchanged, so this function is safe to call
|
|
80
|
+
unconditionally.
|
|
81
|
+
|
|
82
|
+
Args:
|
|
83
|
+
messages (List[Dict[str, Any]]): List of message dictionaries in Ollama format
|
|
84
|
+
|
|
85
|
+
Returns:
|
|
86
|
+
List[Dict[str, Any]]: Translated messages in Anthropic format
|
|
87
|
+
"""
|
|
88
|
+
translated_messages: List[Dict[str, Any]] = []
|
|
89
|
+
# Reference to the content list of the most recently appended tool_result
|
|
90
|
+
# group, so consecutive tool messages get merged into one user message.
|
|
91
|
+
current_tool_result_group: Optional[List[Dict[str, Any]]] = None
|
|
92
|
+
|
|
93
|
+
for msg in messages:
|
|
94
|
+
if "images" in msg and msg["images"]:
|
|
95
|
+
content = [{"type": "text", "text": msg["content"]}]
|
|
96
|
+
for image in msg["images"]:
|
|
97
|
+
content.append(
|
|
98
|
+
{
|
|
99
|
+
"type": "image",
|
|
100
|
+
"source": {
|
|
101
|
+
"type": "base64",
|
|
102
|
+
"data": encode_image_to_base64(image),
|
|
103
|
+
"media_type": "image/jpeg",
|
|
104
|
+
},
|
|
105
|
+
}
|
|
106
|
+
)
|
|
107
|
+
translated_messages.append({"role": "user", "content": content})
|
|
108
|
+
current_tool_result_group = None
|
|
109
|
+
|
|
110
|
+
elif msg["role"] == "user":
|
|
111
|
+
# Regular user messages pass through unchanged
|
|
112
|
+
translated_messages.append({"role": "user", "content": msg["content"]})
|
|
113
|
+
current_tool_result_group = None
|
|
114
|
+
|
|
115
|
+
elif msg["role"] == "assistant" and msg.get("tool_calls"):
|
|
116
|
+
# Convert every tool call made in this turn into one tool_use block
|
|
117
|
+
# on a single assistant message.
|
|
118
|
+
content = []
|
|
119
|
+
if msg.get("content"):
|
|
120
|
+
content.append({"type": "text", "text": msg["content"]})
|
|
121
|
+
|
|
122
|
+
for tool_call in msg["tool_calls"]:
|
|
123
|
+
function = tool_call["function"]
|
|
124
|
+
tool_input = function["arguments"]
|
|
125
|
+
if isinstance(tool_input, str):
|
|
126
|
+
try:
|
|
127
|
+
tool_input = json.loads(tool_input)
|
|
128
|
+
except json.JSONDecodeError:
|
|
129
|
+
# keep the transcript replayable: Anthropic needs an
|
|
130
|
+
# object here, and the tool result already reports the
|
|
131
|
+
# parse failure to the model
|
|
132
|
+
logging.warning(
|
|
133
|
+
"Unparseable tool arguments for %s; sending raw string",
|
|
134
|
+
function["name"],
|
|
135
|
+
)
|
|
136
|
+
tool_input = {"raw_arguments": tool_input}
|
|
137
|
+
content.append(
|
|
138
|
+
{
|
|
139
|
+
"type": "tool_use",
|
|
140
|
+
"id": tool_call["id"],
|
|
141
|
+
"name": function["name"],
|
|
142
|
+
"input": tool_input,
|
|
143
|
+
}
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
translated_messages.append({"role": "assistant", "content": content})
|
|
147
|
+
current_tool_result_group = None
|
|
148
|
+
|
|
149
|
+
elif msg["role"] == "tool":
|
|
150
|
+
# Convert tool response to Anthropic's tool_result format, merging
|
|
151
|
+
# consecutive tool messages into one user message.
|
|
152
|
+
tool_result: Dict[str, Any] = {
|
|
153
|
+
"type": "tool_result",
|
|
154
|
+
"tool_use_id": msg["tool_call_id"],
|
|
155
|
+
"content": msg["content"],
|
|
156
|
+
}
|
|
157
|
+
if msg.get("is_error"):
|
|
158
|
+
tool_result["is_error"] = True
|
|
159
|
+
|
|
160
|
+
if current_tool_result_group is not None:
|
|
161
|
+
current_tool_result_group.append(tool_result)
|
|
162
|
+
else:
|
|
163
|
+
current_tool_result_group = [tool_result]
|
|
164
|
+
translated_messages.append(
|
|
165
|
+
{"role": "user", "content": current_tool_result_group}
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
elif msg["role"] == "assistant":
|
|
169
|
+
# Assistant messages with plain string content pass through.
|
|
170
|
+
translated_messages.append(msg)
|
|
171
|
+
current_tool_result_group = None
|
|
172
|
+
else:
|
|
173
|
+
raise ValueError(f"Unknown message role: {msg['role']}")
|
|
174
|
+
|
|
175
|
+
return translated_messages
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def convert_anthropic_models_to_ollama_response(
|
|
179
|
+
models_data: Dict[str, Any],
|
|
180
|
+
) -> ListResponse:
|
|
181
|
+
"""
|
|
182
|
+
Converts Anthropic model list API response to Ollama format.
|
|
183
|
+
|
|
184
|
+
Args:
|
|
185
|
+
models_data: The response from Anthropic's models API endpoint.
|
|
186
|
+
|
|
187
|
+
Returns:
|
|
188
|
+
An instance of ollama's ListResponse.
|
|
189
|
+
"""
|
|
190
|
+
ollama_models = []
|
|
191
|
+
for model_data in models_data["data"]:
|
|
192
|
+
# Convert creation time from ISO format to datetime
|
|
193
|
+
created_at = datetime.fromisoformat(
|
|
194
|
+
model_data["created_at"].replace("Z", "+00:00")
|
|
195
|
+
)
|
|
196
|
+
|
|
197
|
+
model = {
|
|
198
|
+
"model": model_data["id"],
|
|
199
|
+
"modified_at": created_at,
|
|
200
|
+
"digest": "unknown",
|
|
201
|
+
"size": 0,
|
|
202
|
+
"details": {
|
|
203
|
+
"parent_model": "",
|
|
204
|
+
"format": "unknown",
|
|
205
|
+
"family": "claude",
|
|
206
|
+
"families": ["claude"],
|
|
207
|
+
"parameter_size": "unknown",
|
|
208
|
+
"quantization_level": "unknown",
|
|
209
|
+
"display_name": model_data["display_name"],
|
|
210
|
+
},
|
|
211
|
+
}
|
|
212
|
+
ollama_models.append(ListResponse.Model(**model))
|
|
213
|
+
|
|
214
|
+
return ListResponse(models=ollama_models)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _usage_int(value: Any) -> int:
|
|
218
|
+
"""Usage counters that the SDK may leave unset (or that tests mock) become 0."""
|
|
219
|
+
return value if isinstance(value, int) else 0
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _estimate_thinking_tokens(response: Any) -> int:
|
|
223
|
+
"""Thinking tokens are billed as output but the API does not count them
|
|
224
|
+
separately. When the response carries thinking blocks, estimate them as the
|
|
225
|
+
output tokens that the visible text and tool-call blocks do not account for."""
|
|
226
|
+
blocks = list(getattr(response, "content", None) or [])
|
|
227
|
+
if not any(getattr(block, "type", None) == "thinking" for block in blocks):
|
|
228
|
+
return 0
|
|
229
|
+
visible_chars = 0
|
|
230
|
+
for block in blocks:
|
|
231
|
+
block_type = getattr(block, "type", None)
|
|
232
|
+
if block_type == "text":
|
|
233
|
+
visible_chars += len(getattr(block, "text", "") or "")
|
|
234
|
+
elif block_type == "tool_use":
|
|
235
|
+
try:
|
|
236
|
+
visible_chars += len(json.dumps(getattr(block, "input", None)))
|
|
237
|
+
except (TypeError, ValueError):
|
|
238
|
+
pass
|
|
239
|
+
output_tokens = _usage_int(getattr(response.usage, "output_tokens", 0))
|
|
240
|
+
return max(0, output_tokens - visible_chars // 4)
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
class AnthropicWrapper:
|
|
244
|
+
def __init__(
|
|
245
|
+
self,
|
|
246
|
+
api_key: str,
|
|
247
|
+
max_tokens: int = 4096,
|
|
248
|
+
timeout: float = 600.0,
|
|
249
|
+
prompt_caching: bool = True,
|
|
250
|
+
thinking: Optional[Dict[str, Any]] = None,
|
|
251
|
+
effort: Optional[str] = None,
|
|
252
|
+
):
|
|
253
|
+
"""
|
|
254
|
+
Args:
|
|
255
|
+
api_key (str): Anthropic API key.
|
|
256
|
+
max_tokens (int): Default max_tokens for requests. Requests always
|
|
257
|
+
stream, so large values are safe.
|
|
258
|
+
timeout (float): Request timeout in seconds.
|
|
259
|
+
prompt_caching (bool): When True (the default), every request carries
|
|
260
|
+
top-level `cache_control={"type": "ephemeral"}`, which auto-caches
|
|
261
|
+
the last cacheable block - useful for a growing tool-call
|
|
262
|
+
transcript.
|
|
263
|
+
thinking (Optional[Dict[str, Any]]): Extended thinking configuration,
|
|
264
|
+
e.g. {"type": "adaptive"}. When set, `temperature` is never sent
|
|
265
|
+
(current models reject it together with thinking) and
|
|
266
|
+
`budget_tokens` is never used.
|
|
267
|
+
effort (Optional[str]): One of low|medium|high|xhigh|max. Forwarded as
|
|
268
|
+
`output_config.effort`.
|
|
269
|
+
"""
|
|
270
|
+
self.client = Anthropic(api_key=api_key, timeout=timeout)
|
|
271
|
+
self.api_key = api_key
|
|
272
|
+
self.max_tokens = max_tokens
|
|
273
|
+
self.prompt_caching = prompt_caching
|
|
274
|
+
self.thinking = thinking
|
|
275
|
+
self.effort = effort
|
|
276
|
+
|
|
277
|
+
def chat(
|
|
278
|
+
self,
|
|
279
|
+
messages: List[Dict[str, str]],
|
|
280
|
+
tools: Optional[List[Dict[str, Any]]] = None,
|
|
281
|
+
**kwargs,
|
|
282
|
+
) -> Dict[str, Any]:
|
|
283
|
+
"""
|
|
284
|
+
Conduct a chat conversation using the Anthropic API.
|
|
285
|
+
|
|
286
|
+
Args:
|
|
287
|
+
messages (list[Mapping[str, str]]): A list of message dictionaries, each containing 'role' and 'content'.
|
|
288
|
+
tools (Optional[List[Dict[str, Any]]]): Tool definitions in Ollama/OpenAI format.
|
|
289
|
+
**kwargs: Additional arguments, including:
|
|
290
|
+
- model (str): The Anthropic model to use.
|
|
291
|
+
- max_tokens (int): Overrides the instance's max_tokens.
|
|
292
|
+
- options (dict): May contain "temperature".
|
|
293
|
+
- response_schema (Type[BaseModel]): When provided, it is passed as
|
|
294
|
+
`output_format` so the streamed final message carries a validated
|
|
295
|
+
`parsed_output`.
|
|
296
|
+
- tool_choice (str | dict): "none" -> {"type": "none"},
|
|
297
|
+
"auto" -> {"type": "auto"}, or a dict passed through as-is.
|
|
298
|
+
|
|
299
|
+
Returns:
|
|
300
|
+
A dictionary containing the Anthropic response formatted to match Ollama's expected output.
|
|
301
|
+
"""
|
|
302
|
+
# Extract the system message from the messages and prepare it as a separate argument
|
|
303
|
+
system_message = next(
|
|
304
|
+
(msg["content"] for msg in messages if msg["role"] == "system"), None
|
|
305
|
+
)
|
|
306
|
+
|
|
307
|
+
# Filter out the system message to prevent duplication if it's not needed in the messages parameter
|
|
308
|
+
filtered_messages = [msg for msg in messages if msg["role"] != "system"]
|
|
309
|
+
|
|
310
|
+
# Always translate: this normalizes tool_calls/tool-result/image
|
|
311
|
+
# messages into Anthropic's format, and is a no-op for a plain
|
|
312
|
+
# user/assistant conversation.
|
|
313
|
+
filtered_messages = translate_messages_for_anthropic(filtered_messages)
|
|
314
|
+
|
|
315
|
+
# Common parameters
|
|
316
|
+
params: Dict[str, Any] = {
|
|
317
|
+
"max_tokens": kwargs.get("max_tokens", self.max_tokens),
|
|
318
|
+
"messages": filtered_messages,
|
|
319
|
+
"model": kwargs.get("model", "claude-3-5-sonnet-20240620"),
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
# Only include system parameter if it has a value
|
|
323
|
+
if system_message is not None:
|
|
324
|
+
params["system"] = system_message
|
|
325
|
+
|
|
326
|
+
# Extended thinking / effort.
|
|
327
|
+
if self.thinking is not None:
|
|
328
|
+
params["thinking"] = self.thinking
|
|
329
|
+
if self.effort is not None:
|
|
330
|
+
params["output_config"] = {"effort": self.effort}
|
|
331
|
+
|
|
332
|
+
# Conditionally add temperature if it exists in kwargs. Current models
|
|
333
|
+
# reject `temperature` when adaptive thinking is configured, so only
|
|
334
|
+
# send it when the caller explicitly asked for it AND thinking is off.
|
|
335
|
+
if (
|
|
336
|
+
self.thinking is None
|
|
337
|
+
and "options" in kwargs
|
|
338
|
+
and "temperature" in kwargs["options"]
|
|
339
|
+
):
|
|
340
|
+
params["temperature"] = kwargs["options"]["temperature"]
|
|
341
|
+
|
|
342
|
+
if tools:
|
|
343
|
+
# Translate tools into Anthropic format
|
|
344
|
+
anthropic_tools = translate_tools_for_anthropic(tools)
|
|
345
|
+
params["tools"] = anthropic_tools
|
|
346
|
+
|
|
347
|
+
# tool_choice: "none"/"auto" strings map to Anthropic's dict form; a
|
|
348
|
+
# dict is passed through untouched.
|
|
349
|
+
tool_choice = kwargs.get("tool_choice")
|
|
350
|
+
if tool_choice == "none":
|
|
351
|
+
params["tool_choice"] = {"type": "none"}
|
|
352
|
+
elif tool_choice == "auto":
|
|
353
|
+
params["tool_choice"] = {"type": "auto"}
|
|
354
|
+
elif isinstance(tool_choice, dict):
|
|
355
|
+
params["tool_choice"] = tool_choice
|
|
356
|
+
|
|
357
|
+
response_schema = kwargs.get("response_schema")
|
|
358
|
+
|
|
359
|
+
# Every request streams: `messages.stream` accepts `output_format`
|
|
360
|
+
# (structured output, validated into `parsed_output`) together with
|
|
361
|
+
# `cache_control`, and streaming avoids the SDK's timeout guard on
|
|
362
|
+
# large `max_tokens` values.
|
|
363
|
+
if response_schema is not None:
|
|
364
|
+
params["output_format"] = response_schema
|
|
365
|
+
if self.prompt_caching:
|
|
366
|
+
# Auto-caches the last cacheable block - what a growing
|
|
367
|
+
# tool-call transcript wants.
|
|
368
|
+
params["cache_control"] = {"type": "ephemeral"}
|
|
369
|
+
|
|
370
|
+
try:
|
|
371
|
+
with self.client.messages.stream(**params) as stream:
|
|
372
|
+
response = stream.get_final_message()
|
|
373
|
+
|
|
374
|
+
# Extract usage information
|
|
375
|
+
usage = response.usage
|
|
376
|
+
usage_info = {
|
|
377
|
+
"prompt_tokens": usage.input_tokens,
|
|
378
|
+
"completion_tokens": usage.output_tokens,
|
|
379
|
+
"cached_tokens": usage.cache_read_input_tokens or 0,
|
|
380
|
+
"cache_creation_tokens": _usage_int(
|
|
381
|
+
getattr(usage, "cache_creation_input_tokens", 0)
|
|
382
|
+
),
|
|
383
|
+
"total_tokens": usage.input_tokens + usage.output_tokens,
|
|
384
|
+
"reasoning_tokens": _estimate_thinking_tokens(response),
|
|
385
|
+
}
|
|
386
|
+
|
|
387
|
+
if response.stop_reason == "refusal":
|
|
388
|
+
stop_details = response.stop_details
|
|
389
|
+
explanation = (
|
|
390
|
+
stop_details.explanation if stop_details is not None else None
|
|
391
|
+
)
|
|
392
|
+
return {
|
|
393
|
+
"refusal": explanation or "refused",
|
|
394
|
+
"content": None,
|
|
395
|
+
"done": False,
|
|
396
|
+
}
|
|
397
|
+
|
|
398
|
+
if response.stop_reason == "max_tokens":
|
|
399
|
+
return {
|
|
400
|
+
"error": "Response exceeded the maximum allowed length.",
|
|
401
|
+
"error_type": errors.LENGTH,
|
|
402
|
+
"content": None,
|
|
403
|
+
"done": False,
|
|
404
|
+
"usage": None,
|
|
405
|
+
}
|
|
406
|
+
|
|
407
|
+
# Handle tool calls if present. This shape is returned whether or
|
|
408
|
+
# not `response_schema` was requested (parsed_output is unused
|
|
409
|
+
# here and stays None on the response in that case).
|
|
410
|
+
tool_use_blocks = [
|
|
411
|
+
block for block in response.content if block.type == "tool_use"
|
|
412
|
+
]
|
|
413
|
+
if tool_use_blocks:
|
|
414
|
+
return {
|
|
415
|
+
"message": {
|
|
416
|
+
"content": "",
|
|
417
|
+
"tool_calls": [
|
|
418
|
+
{
|
|
419
|
+
"id": tool_block.id,
|
|
420
|
+
"name": tool_block.name,
|
|
421
|
+
"arguments": tool_block.input, # Anthropic uses 'input' instead of 'arguments'
|
|
422
|
+
}
|
|
423
|
+
for tool_block in tool_use_blocks
|
|
424
|
+
],
|
|
425
|
+
},
|
|
426
|
+
"usage": usage_info,
|
|
427
|
+
"done": response.stop_reason == "end_turn",
|
|
428
|
+
}
|
|
429
|
+
|
|
430
|
+
if response_schema is not None:
|
|
431
|
+
# `output_format` makes the final message a ParsedMessage with
|
|
432
|
+
# `parsed_output`; `generate_pydantic` accepts a BaseModel here.
|
|
433
|
+
return {
|
|
434
|
+
"message": {"content": getattr(response, "parsed_output", None)},
|
|
435
|
+
"usage": usage_info,
|
|
436
|
+
"done": response.stop_reason == "end_turn",
|
|
437
|
+
}
|
|
438
|
+
|
|
439
|
+
# Extract content blocks as text and simulate Ollama-like response
|
|
440
|
+
content = "".join(
|
|
441
|
+
block.text for block in response.content if block.type == "text"
|
|
442
|
+
)
|
|
443
|
+
|
|
444
|
+
return {
|
|
445
|
+
"message": {"content": content},
|
|
446
|
+
"usage": usage_info,
|
|
447
|
+
"done": response.stop_reason == "end_turn",
|
|
448
|
+
}
|
|
449
|
+
|
|
450
|
+
except APITimeoutError as e:
|
|
451
|
+
error_message = f"Anthropic API timeout: {str(e)}"
|
|
452
|
+
logging.error(error_message)
|
|
453
|
+
return {
|
|
454
|
+
"error": error_message,
|
|
455
|
+
"error_type": errors.TIMEOUT,
|
|
456
|
+
"content": None,
|
|
457
|
+
"done": False,
|
|
458
|
+
"usage": None,
|
|
459
|
+
}
|
|
460
|
+
except APIConnectionError as e:
|
|
461
|
+
error_message = f"Anthropic API connection error: {str(e)}"
|
|
462
|
+
logging.error(error_message)
|
|
463
|
+
return {
|
|
464
|
+
"error": error_message,
|
|
465
|
+
"error_type": errors.CONNECTION,
|
|
466
|
+
"content": None,
|
|
467
|
+
"done": False,
|
|
468
|
+
"usage": None,
|
|
469
|
+
}
|
|
470
|
+
except APIError as e:
|
|
471
|
+
error_message = f"Anthropic API error: {str(e)}"
|
|
472
|
+
logging.error(error_message)
|
|
473
|
+
return {
|
|
474
|
+
"error": error_message,
|
|
475
|
+
"error_type": errors.PROVIDER_SPECIFIC,
|
|
476
|
+
"content": None,
|
|
477
|
+
"done": False,
|
|
478
|
+
"usage": None,
|
|
479
|
+
}
|
|
480
|
+
except Exception as e:
|
|
481
|
+
# The streaming accumulator validates structured output as blocks
|
|
482
|
+
# complete and raises a pydantic ValidationError when a text block
|
|
483
|
+
# is empty or truncated (for example when thinking exhausted
|
|
484
|
+
# max_tokens). Surface it as a model error so generate_pydantic
|
|
485
|
+
# can retry instead of crashing the caller.
|
|
486
|
+
error_message = f"Anthropic response could not be parsed: {str(e)}"
|
|
487
|
+
logging.error(error_message)
|
|
488
|
+
return {
|
|
489
|
+
"error": error_message,
|
|
490
|
+
"error_type": errors.PROVIDER_SPECIFIC,
|
|
491
|
+
"content": None,
|
|
492
|
+
"done": False,
|
|
493
|
+
"usage": None,
|
|
494
|
+
}
|
|
495
|
+
|
|
496
|
+
def list(self) -> ListResponse:
|
|
497
|
+
"""
|
|
498
|
+
Returns a list of available Anthropic models in Ollama format.
|
|
499
|
+
Uses the Anthropic API to get the current list of models.
|
|
500
|
+
"""
|
|
501
|
+
headers = {"x-api-key": self.api_key, "anthropic-version": "2023-06-01"}
|
|
502
|
+
|
|
503
|
+
response = requests.get("https://api.anthropic.com/v1/models", headers=headers)
|
|
504
|
+
response.raise_for_status()
|
|
505
|
+
|
|
506
|
+
return convert_anthropic_models_to_ollama_response(response.json())
|
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
import os
|
|
2
2
|
import re
|
|
3
3
|
from datetime import datetime
|
|
4
|
-
from typing import Literal, Optional
|
|
4
|
+
from typing import Any, Dict, Literal, Optional
|
|
5
5
|
|
|
6
6
|
from .anthropic import AnthropicWrapper
|
|
7
7
|
from .gemini import GeminiWrapper
|
|
8
8
|
from .llm_interface import LLMInterface
|
|
9
9
|
from .openai import OpenAIWrapper
|
|
10
|
+
from .openai_responses import OpenAIResponsesWrapper
|
|
10
11
|
from .openrouter import OpenRouterWrapper
|
|
11
12
|
from .remote_ollama import RemoteOllama
|
|
12
13
|
from .ssh import SSHConnection
|
|
@@ -33,6 +34,15 @@ def supports_structured_output(model_name: str) -> bool:
|
|
|
33
34
|
Returns:
|
|
34
35
|
bool: Whether the model supports structured outputs
|
|
35
36
|
"""
|
|
37
|
+
# Check if this is a GPT model with version 5 or higher
|
|
38
|
+
# This handles both base models and dated variants (e.g., gpt-5, gpt-5-mini, gpt-5-2025-01-01)
|
|
39
|
+
if model_name.startswith("gpt-"):
|
|
40
|
+
# Extract the major version after "gpt-" (gpt-5, gpt-5-mini, gpt-5.6-luna)
|
|
41
|
+
parts = model_name[4:].split("-")
|
|
42
|
+
major = re.match(r"(\d+)(?:\.\d+)?$", parts[0]) if parts else None
|
|
43
|
+
if major and int(major.group(1)) >= 5:
|
|
44
|
+
return True
|
|
45
|
+
|
|
36
46
|
# Models that always support structured outputs (no date requirements)
|
|
37
47
|
base_models = {
|
|
38
48
|
"gpt-4o",
|
|
@@ -83,6 +93,11 @@ def llm_from_config(
|
|
|
83
93
|
timeout: float = 600.0,
|
|
84
94
|
json_mode: Optional[bool] = None,
|
|
85
95
|
structured_outputs: Optional[bool] = None,
|
|
96
|
+
thinking: Optional[Dict[str, Any]] = None,
|
|
97
|
+
effort: Optional[str] = None,
|
|
98
|
+
prompt_caching: bool = True,
|
|
99
|
+
max_tool_rounds: int = 5,
|
|
100
|
+
openai_api: Literal["responses", "chat"] = "responses",
|
|
86
101
|
) -> LLMInterface:
|
|
87
102
|
"""
|
|
88
103
|
Creates and configures a language model interface based on specified provider and parameters.
|
|
@@ -103,6 +118,19 @@ def llm_from_config(
|
|
|
103
118
|
timeout (float): Timeout in seconds for model requests. Defaults to 600.0.
|
|
104
119
|
json_mode (Optional[bool]): Whether to override JSON mode support. Defaults to None.
|
|
105
120
|
structured_outputs (Optional[bool]): Whether to override structured output support. Defaults to None.
|
|
121
|
+
thinking (Optional[Dict[str, Any]]): Extended thinking configuration forwarded to
|
|
122
|
+
`AnthropicWrapper` (e.g. {"type": "adaptive"}). Only used by the "anthropic" provider.
|
|
123
|
+
effort (Optional[str]): Effort level forwarded to `AnthropicWrapper` as
|
|
124
|
+
`output_config.effort` (low|medium|high|xhigh|max) or to `OpenAIWrapper`
|
|
125
|
+
as `reasoning_effort` (none|low|medium|high|xhigh). Used by the
|
|
126
|
+
"anthropic" and "openai" providers.
|
|
127
|
+
prompt_caching (bool): Whether `AnthropicWrapper` should send top-level `cache_control`.
|
|
128
|
+
Defaults to True. Only used by the "anthropic" provider.
|
|
129
|
+
max_tool_rounds (int): Maximum number of tool-call round-trips forwarded to
|
|
130
|
+
`LLMInterface`. Defaults to 5. Used by the "openai" and "anthropic" providers.
|
|
131
|
+
openai_api (Literal["responses", "chat"]): Which OpenAI API the "openai"
|
|
132
|
+
provider talks to. "responses" (the default) supports reasoning
|
|
133
|
+
together with function tools; "chat" is the Chat Completions API.
|
|
106
134
|
|
|
107
135
|
Returns:
|
|
108
136
|
LLMInterface: Configured interface for interacting with the specified LLM.
|
|
@@ -132,8 +160,14 @@ def llm_from_config(
|
|
|
132
160
|
api_key = os.getenv("OPENAI_API_KEY")
|
|
133
161
|
if api_key is None:
|
|
134
162
|
raise ValueError("OPENAI_API_KEY not found in environment variables")
|
|
135
|
-
|
|
136
|
-
|
|
163
|
+
wrapper_class = (
|
|
164
|
+
OpenAIResponsesWrapper if openai_api == "responses" else OpenAIWrapper
|
|
165
|
+
)
|
|
166
|
+
wrapper = wrapper_class(
|
|
167
|
+
api_key=api_key,
|
|
168
|
+
max_tokens=max_tokens,
|
|
169
|
+
timeout=timeout,
|
|
170
|
+
reasoning_effort=effort,
|
|
137
171
|
)
|
|
138
172
|
|
|
139
173
|
support_structured_outputs = supports_structured_output(model_name)
|
|
@@ -149,19 +183,31 @@ def llm_from_config(
|
|
|
149
183
|
support_system_prompt=support_system_prompt,
|
|
150
184
|
use_cache=use_cache,
|
|
151
185
|
timeout=timeout,
|
|
186
|
+
max_tool_rounds=max_tool_rounds,
|
|
152
187
|
)
|
|
153
188
|
case "anthropic":
|
|
154
189
|
api_key = os.getenv("ANTHROPIC_API_KEY")
|
|
155
190
|
if api_key is None:
|
|
156
191
|
raise ValueError("ANTHROPIC_API_KEY not found in environment variables")
|
|
157
|
-
wrapper = AnthropicWrapper(
|
|
192
|
+
wrapper = AnthropicWrapper(
|
|
193
|
+
api_key=api_key,
|
|
194
|
+
max_tokens=max_tokens,
|
|
195
|
+
timeout=timeout,
|
|
196
|
+
prompt_caching=prompt_caching,
|
|
197
|
+
thinking=thinking,
|
|
198
|
+
effort=effort,
|
|
199
|
+
)
|
|
158
200
|
llm = LLMInterface(
|
|
159
201
|
model_name=model_name,
|
|
160
202
|
log_dir=log_dir,
|
|
161
203
|
client=wrapper,
|
|
162
|
-
|
|
204
|
+
# Anthropic's structured-output path (client.messages.parse) and
|
|
205
|
+
# response_schema handling route through these flags.
|
|
206
|
+
support_json_mode=True,
|
|
207
|
+
support_structured_outputs=True,
|
|
163
208
|
use_cache=use_cache,
|
|
164
209
|
timeout=timeout,
|
|
210
|
+
max_tool_rounds=max_tool_rounds,
|
|
165
211
|
)
|
|
166
212
|
case "gemini":
|
|
167
213
|
api_key = os.getenv("GEMINI_API_KEY")
|