llm.rb 13.0.0 → 14.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +505 -14
- data/README.md +484 -50
- data/bin/llm.rb +148 -0
- data/data/anthropic.json +206 -263
- data/data/bedrock.json +2138 -1860
- data/data/deepinfra.json +1003 -624
- data/data/deepseek.json +38 -34
- data/data/google.json +1079 -371
- data/data/mistral.json +448 -368
- data/data/moonshot.json +384 -0
- data/data/openai.json +974 -1343
- data/data/xai.json +154 -126
- data/data/zai.json +191 -191
- data/lib/llm/agent.rb +123 -20
- data/lib/llm/context.rb +71 -88
- data/lib/llm/cost.rb +23 -17
- data/lib/llm/error.rb +0 -8
- data/lib/llm/function/array.rb +3 -3
- data/lib/llm/function/async/task.rb +2 -0
- data/lib/llm/function/fiber/task.rb +2 -0
- data/lib/llm/function/fork/task.rb +2 -0
- data/lib/llm/function/ractor/task.rb +2 -0
- data/lib/llm/function/sequential/group.rb +4 -1
- data/lib/llm/function/sequential/task.rb +1 -1
- data/lib/llm/function/task.rb +4 -0
- data/lib/llm/function/thread/task.rb +2 -0
- data/lib/llm/function.rb +33 -6
- data/lib/llm/guard/loop.rb +89 -0
- data/lib/llm/guard/null.rb +19 -0
- data/lib/llm/guard.rb +61 -0
- data/lib/llm/provider.rb +36 -0
- data/lib/llm/providers/anthropic/stream_parser.rb +1 -0
- data/lib/llm/providers/anthropic.rb +2 -9
- data/lib/llm/providers/bedrock/request_adapter.rb +1 -1
- data/lib/llm/providers/bedrock/stream_parser.rb +1 -0
- data/lib/llm/providers/bedrock.rb +1 -8
- data/lib/llm/providers/google/stream_parser.rb +1 -0
- data/lib/llm/providers/google.rb +1 -8
- data/lib/llm/providers/mistral.rb +1 -1
- data/lib/llm/providers/moonshot.rb +76 -0
- data/lib/llm/providers/ollama.rb +2 -9
- data/lib/llm/providers/openai/responses/stream_parser.rb +1 -0
- data/lib/llm/providers/openai/responses.rb +7 -9
- data/lib/llm/providers/openai/stream_parser.rb +1 -0
- data/lib/llm/providers/openai.rb +4 -11
- data/lib/llm/repl/bar.rb +4 -3
- data/lib/llm/repl/{transcript.rb → buffer.rb} +69 -29
- data/lib/llm/repl/color.rb +78 -0
- data/lib/llm/repl/command.rb +12 -5
- data/lib/llm/repl/commands/compact.rb +2 -2
- data/lib/llm/repl/commands/help.rb +3 -5
- data/lib/llm/repl/input/char.rb +46 -0
- data/lib/llm/repl/input/row.rb +39 -0
- data/lib/llm/repl/input.rb +251 -66
- data/lib/llm/repl/markdown/table.rb +11 -3
- data/lib/llm/repl/markdown.rb +34 -8
- data/lib/llm/repl/node.rb +37 -0
- data/lib/llm/repl/status.rb +42 -7
- data/lib/llm/repl/stream.rb +18 -6
- data/lib/llm/repl/walker.rb +3 -2
- data/lib/llm/repl/window.rb +54 -35
- data/lib/llm/repl.rb +74 -32
- data/lib/llm/skill.rb +20 -4
- data/lib/llm/stream.rb +8 -7
- data/lib/llm/tool.rb +29 -0
- data/lib/llm/tools/{swap_text.rb → edit-file.rb} +3 -3
- data/lib/llm/tools/git.rb +3 -0
- data/lib/llm/tools/mkdir.rb +3 -0
- data/lib/llm/tools/rg.rb +3 -0
- data/lib/llm/tools/ruby.rb +46 -0
- data/lib/llm/tools/shell.rb +3 -0
- data/lib/llm/tracer/pretty_logger.rb +127 -0
- data/lib/llm/tracer.rb +1 -0
- data/lib/llm/transformer/null.rb +21 -0
- data/lib/llm/transformer.rb +55 -0
- data/lib/llm/version.rb +1 -1
- data/lib/llm.rb +12 -2
- data/llm.gemspec +9 -2
- data/resources/deepdive/advanced/cancellation.md +74 -0
- data/resources/deepdive/advanced/compaction.md +83 -0
- data/resources/deepdive/advanced/context.md +267 -0
- data/resources/deepdive/advanced/guard.md +371 -0
- data/resources/deepdive/advanced/tracer.md +180 -0
- data/resources/deepdive/advanced/transformer.md +67 -0
- data/resources/deepdive/advanced/transports.md +45 -0
- data/resources/deepdive/everything_else/audio.md +122 -0
- data/resources/deepdive/everything_else/cost.md +99 -0
- data/resources/deepdive/everything_else/images.md +89 -0
- data/resources/deepdive/everything_else/object.md +108 -0
- data/resources/deepdive/everything_else/ocr.md +48 -0
- data/resources/deepdive/fundamentals/agents.md +202 -0
- data/resources/deepdive/fundamentals/builtin_tools.md +191 -0
- data/resources/deepdive/fundamentals/concurrency.md +104 -0
- data/resources/deepdive/fundamentals/database.md +449 -0
- data/resources/deepdive/fundamentals/embeddings.md +157 -0
- data/resources/deepdive/fundamentals/repl.md +87 -0
- data/resources/deepdive/fundamentals/schema.md +61 -0
- data/resources/deepdive/fundamentals/skills.md +106 -0
- data/resources/deepdive/fundamentals/stream.md +110 -0
- data/resources/deepdive/fundamentals/tools.md +265 -0
- data/resources/deepdive/protocols/a2a.md +106 -0
- data/resources/deepdive/protocols/mcp.md +111 -0
- data/resources/deepdive.md +58 -1792
- metadata +51 -7
- data/lib/llm/loop_guard.rb +0 -107
data/data/deepinfra.json
CHANGED
|
@@ -7,24 +7,22 @@
|
|
|
7
7
|
"name": "Deep Infra",
|
|
8
8
|
"doc": "https://deepinfra.com/models",
|
|
9
9
|
"models": {
|
|
10
|
-
"
|
|
11
|
-
"id": "
|
|
12
|
-
"name": "
|
|
13
|
-
"description": "
|
|
14
|
-
"family": "
|
|
15
|
-
"attachment":
|
|
10
|
+
"nvidia/Llama-3.3-Nemotron-Super-49B-v1.5": {
|
|
11
|
+
"id": "nvidia/Llama-3.3-Nemotron-Super-49B-v1.5",
|
|
12
|
+
"name": "Llama 3.3 Nemotron Super 49B v1.5",
|
|
13
|
+
"description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
|
|
14
|
+
"family": "nemotron",
|
|
15
|
+
"attachment": false,
|
|
16
16
|
"reasoning": true,
|
|
17
17
|
"reasoning_options": [],
|
|
18
18
|
"tool_call": true,
|
|
19
19
|
"structured_output": true,
|
|
20
20
|
"temperature": true,
|
|
21
|
-
"release_date": "
|
|
22
|
-
"last_updated": "
|
|
21
|
+
"release_date": "2025-07-25",
|
|
22
|
+
"last_updated": "2025-07-25",
|
|
23
23
|
"modalities": {
|
|
24
24
|
"input": [
|
|
25
|
-
"text"
|
|
26
|
-
"image",
|
|
27
|
-
"video"
|
|
25
|
+
"text"
|
|
28
26
|
],
|
|
29
27
|
"output": [
|
|
30
28
|
"text"
|
|
@@ -32,33 +30,34 @@
|
|
|
32
30
|
},
|
|
33
31
|
"open_weights": true,
|
|
34
32
|
"limit": {
|
|
35
|
-
"context":
|
|
36
|
-
"output":
|
|
33
|
+
"context": 131072,
|
|
34
|
+
"output": 131072
|
|
37
35
|
},
|
|
36
|
+
"status": "deprecated",
|
|
38
37
|
"cost": {
|
|
39
|
-
"input": 0.
|
|
40
|
-
"output": 0.
|
|
38
|
+
"input": 0.4,
|
|
39
|
+
"output": 0.4
|
|
41
40
|
}
|
|
42
41
|
},
|
|
43
|
-
"
|
|
44
|
-
"id": "
|
|
45
|
-
"name": "
|
|
46
|
-
"description": "
|
|
47
|
-
"family": "
|
|
42
|
+
"nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning": {
|
|
43
|
+
"id": "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning",
|
|
44
|
+
"name": "Nemotron 3 Nano Omni 30B A3B Reasoning",
|
|
45
|
+
"description": "Open Nemotron omni model combining reasoning with text, vision, and audio",
|
|
46
|
+
"family": "nemotron",
|
|
48
47
|
"attachment": true,
|
|
49
48
|
"reasoning": true,
|
|
50
49
|
"reasoning_options": [],
|
|
51
50
|
"tool_call": true,
|
|
52
51
|
"structured_output": true,
|
|
53
52
|
"temperature": true,
|
|
54
|
-
"
|
|
55
|
-
"
|
|
56
|
-
"last_updated": "2026-04-20",
|
|
53
|
+
"release_date": "2026-04-28",
|
|
54
|
+
"last_updated": "2026-04-28",
|
|
57
55
|
"modalities": {
|
|
58
56
|
"input": [
|
|
59
57
|
"text",
|
|
60
58
|
"image",
|
|
61
|
-
"video"
|
|
59
|
+
"video",
|
|
60
|
+
"audio"
|
|
62
61
|
],
|
|
63
62
|
"output": [
|
|
64
63
|
"text"
|
|
@@ -67,27 +66,30 @@
|
|
|
67
66
|
"open_weights": true,
|
|
68
67
|
"limit": {
|
|
69
68
|
"context": 262144,
|
|
70
|
-
"output":
|
|
69
|
+
"output": 65536
|
|
71
70
|
},
|
|
71
|
+
"status": "deprecated",
|
|
72
72
|
"cost": {
|
|
73
|
-
"input": 0.
|
|
74
|
-
"output":
|
|
75
|
-
"cache_read": 0.05
|
|
73
|
+
"input": 0.2,
|
|
74
|
+
"output": 0.8
|
|
76
75
|
}
|
|
77
76
|
},
|
|
78
|
-
"
|
|
79
|
-
"id": "
|
|
80
|
-
"name": "
|
|
81
|
-
"description": "
|
|
82
|
-
"family": "
|
|
77
|
+
"nvidia/Nemotron-3-Nano-30B-A3B": {
|
|
78
|
+
"id": "nvidia/Nemotron-3-Nano-30B-A3B",
|
|
79
|
+
"name": "Nemotron 3 Nano 30B A3B",
|
|
80
|
+
"description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents",
|
|
81
|
+
"family": "nemotron",
|
|
83
82
|
"attachment": false,
|
|
84
|
-
"reasoning":
|
|
83
|
+
"reasoning": true,
|
|
84
|
+
"reasoning_options": [
|
|
85
|
+
{
|
|
86
|
+
"type": "toggle"
|
|
87
|
+
}
|
|
88
|
+
],
|
|
85
89
|
"tool_call": true,
|
|
86
|
-
"structured_output": true,
|
|
87
90
|
"temperature": true,
|
|
88
|
-
"
|
|
89
|
-
"
|
|
90
|
-
"last_updated": "2025-07-23",
|
|
91
|
+
"release_date": "2025-12-15",
|
|
92
|
+
"last_updated": "2025-12-15",
|
|
91
93
|
"modalities": {
|
|
92
94
|
"input": [
|
|
93
95
|
"text"
|
|
@@ -99,31 +101,35 @@
|
|
|
99
101
|
"open_weights": true,
|
|
100
102
|
"limit": {
|
|
101
103
|
"context": 262144,
|
|
102
|
-
"output":
|
|
104
|
+
"output": 262144
|
|
103
105
|
},
|
|
104
106
|
"cost": {
|
|
105
|
-
"input": 0.
|
|
106
|
-
"output":
|
|
107
|
-
"cache_read": 0.
|
|
107
|
+
"input": 0.05,
|
|
108
|
+
"output": 0.2,
|
|
109
|
+
"cache_read": 0.025
|
|
108
110
|
}
|
|
109
111
|
},
|
|
110
|
-
"
|
|
111
|
-
"id": "
|
|
112
|
-
"name": "
|
|
113
|
-
"description": "
|
|
114
|
-
"family": "
|
|
115
|
-
"attachment":
|
|
112
|
+
"google/gemma-4-26B-A4B-it": {
|
|
113
|
+
"id": "google/gemma-4-26B-A4B-it",
|
|
114
|
+
"name": "Gemma 4 26B A4B IT",
|
|
115
|
+
"description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
|
|
116
|
+
"family": "gemma",
|
|
117
|
+
"attachment": true,
|
|
116
118
|
"reasoning": true,
|
|
117
|
-
"reasoning_options": [
|
|
119
|
+
"reasoning_options": [
|
|
120
|
+
{
|
|
121
|
+
"type": "toggle"
|
|
122
|
+
}
|
|
123
|
+
],
|
|
118
124
|
"tool_call": true,
|
|
119
125
|
"structured_output": true,
|
|
120
126
|
"temperature": true,
|
|
121
|
-
"
|
|
122
|
-
"
|
|
123
|
-
"last_updated": "2025-04",
|
|
127
|
+
"release_date": "2026-04-02",
|
|
128
|
+
"last_updated": "2026-04-02",
|
|
124
129
|
"modalities": {
|
|
125
130
|
"input": [
|
|
126
|
-
"text"
|
|
131
|
+
"text",
|
|
132
|
+
"image"
|
|
127
133
|
],
|
|
128
134
|
"output": [
|
|
129
135
|
"text"
|
|
@@ -131,33 +137,36 @@
|
|
|
131
137
|
},
|
|
132
138
|
"open_weights": true,
|
|
133
139
|
"limit": {
|
|
134
|
-
"context":
|
|
135
|
-
"output":
|
|
140
|
+
"context": 262144,
|
|
141
|
+
"output": 32768
|
|
136
142
|
},
|
|
137
143
|
"cost": {
|
|
138
|
-
"input": 0.
|
|
139
|
-
"output": 0.
|
|
144
|
+
"input": 0.07,
|
|
145
|
+
"output": 0.34
|
|
140
146
|
}
|
|
141
147
|
},
|
|
142
|
-
"
|
|
143
|
-
"id": "
|
|
144
|
-
"name": "
|
|
145
|
-
"description": "
|
|
146
|
-
"family": "
|
|
148
|
+
"google/gemma-4-E4B-it": {
|
|
149
|
+
"id": "google/gemma-4-E4B-it",
|
|
150
|
+
"name": "Gemma 4 E4B IT",
|
|
151
|
+
"description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
|
|
152
|
+
"family": "gemma",
|
|
147
153
|
"attachment": true,
|
|
148
154
|
"reasoning": true,
|
|
149
|
-
"reasoning_options": [
|
|
155
|
+
"reasoning_options": [
|
|
156
|
+
{
|
|
157
|
+
"type": "toggle"
|
|
158
|
+
}
|
|
159
|
+
],
|
|
150
160
|
"tool_call": true,
|
|
151
161
|
"structured_output": true,
|
|
152
162
|
"temperature": true,
|
|
153
|
-
"
|
|
154
|
-
"
|
|
155
|
-
"last_updated": "2026-04-20",
|
|
163
|
+
"release_date": "2026-04-02",
|
|
164
|
+
"last_updated": "2026-04-02",
|
|
156
165
|
"modalities": {
|
|
157
166
|
"input": [
|
|
158
167
|
"text",
|
|
159
168
|
"image",
|
|
160
|
-
"
|
|
169
|
+
"audio"
|
|
161
170
|
],
|
|
162
171
|
"output": [
|
|
163
172
|
"text"
|
|
@@ -165,34 +174,36 @@
|
|
|
165
174
|
},
|
|
166
175
|
"open_weights": true,
|
|
167
176
|
"limit": {
|
|
168
|
-
"context":
|
|
169
|
-
"output":
|
|
177
|
+
"context": 131072,
|
|
178
|
+
"output": 8192
|
|
170
179
|
},
|
|
171
180
|
"cost": {
|
|
172
|
-
"input": 0.
|
|
173
|
-
"output":
|
|
174
|
-
"cache_read": 0.22
|
|
181
|
+
"input": 0.02,
|
|
182
|
+
"output": 0.1
|
|
175
183
|
}
|
|
176
184
|
},
|
|
177
|
-
"
|
|
178
|
-
"id": "
|
|
179
|
-
"name": "
|
|
180
|
-
"description": "
|
|
181
|
-
"family": "
|
|
185
|
+
"google/gemma-4-31B-it": {
|
|
186
|
+
"id": "google/gemma-4-31B-it",
|
|
187
|
+
"name": "Gemma 4 31B IT",
|
|
188
|
+
"description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
|
|
189
|
+
"family": "gemma",
|
|
182
190
|
"attachment": true,
|
|
183
191
|
"reasoning": true,
|
|
184
|
-
"reasoning_options": [
|
|
192
|
+
"reasoning_options": [
|
|
193
|
+
{
|
|
194
|
+
"type": "toggle"
|
|
195
|
+
}
|
|
196
|
+
],
|
|
185
197
|
"tool_call": true,
|
|
186
198
|
"structured_output": true,
|
|
187
199
|
"temperature": true,
|
|
188
|
-
"release_date": "2026-02
|
|
189
|
-
"last_updated": "2026-02
|
|
200
|
+
"release_date": "2026-04-02",
|
|
201
|
+
"last_updated": "2026-04-02",
|
|
190
202
|
"modalities": {
|
|
191
203
|
"input": [
|
|
192
204
|
"text",
|
|
193
205
|
"image",
|
|
194
|
-
"video"
|
|
195
|
-
"audio"
|
|
206
|
+
"video"
|
|
196
207
|
],
|
|
197
208
|
"output": [
|
|
198
209
|
"text"
|
|
@@ -201,31 +212,29 @@
|
|
|
201
212
|
"open_weights": true,
|
|
202
213
|
"limit": {
|
|
203
214
|
"context": 262144,
|
|
204
|
-
"output":
|
|
215
|
+
"output": 32768
|
|
205
216
|
},
|
|
206
217
|
"cost": {
|
|
207
|
-
"input": 0.
|
|
208
|
-
"output":
|
|
218
|
+
"input": 0.13,
|
|
219
|
+
"output": 0.38
|
|
209
220
|
}
|
|
210
221
|
},
|
|
211
|
-
"
|
|
212
|
-
"id": "
|
|
213
|
-
"name": "
|
|
214
|
-
"description": "
|
|
215
|
-
"family": "
|
|
222
|
+
"thinkingmachines/Inkling-Small": {
|
|
223
|
+
"id": "thinkingmachines/Inkling-Small",
|
|
224
|
+
"name": "Inkling Small",
|
|
225
|
+
"description": "Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio",
|
|
226
|
+
"family": "ling",
|
|
216
227
|
"attachment": true,
|
|
217
228
|
"reasoning": true,
|
|
218
229
|
"reasoning_options": [],
|
|
219
230
|
"tool_call": true,
|
|
220
|
-
"structured_output": true,
|
|
221
231
|
"temperature": true,
|
|
222
|
-
"release_date": "2026-
|
|
223
|
-
"last_updated": "2026-
|
|
232
|
+
"release_date": "2026-07-30",
|
|
233
|
+
"last_updated": "2026-07-30",
|
|
224
234
|
"modalities": {
|
|
225
235
|
"input": [
|
|
226
236
|
"text",
|
|
227
237
|
"image",
|
|
228
|
-
"video",
|
|
229
238
|
"audio"
|
|
230
239
|
],
|
|
231
240
|
"output": [
|
|
@@ -234,32 +243,31 @@
|
|
|
234
243
|
},
|
|
235
244
|
"open_weights": true,
|
|
236
245
|
"limit": {
|
|
237
|
-
"context":
|
|
238
|
-
"output":
|
|
246
|
+
"context": 524288,
|
|
247
|
+
"output": 1048576
|
|
239
248
|
},
|
|
240
249
|
"cost": {
|
|
241
|
-
"input": 0.
|
|
242
|
-
"output":
|
|
250
|
+
"input": 0.45,
|
|
251
|
+
"output": 1.2,
|
|
252
|
+
"cache_read": 0.1
|
|
243
253
|
}
|
|
244
254
|
},
|
|
245
|
-
"
|
|
246
|
-
"id": "
|
|
247
|
-
"name": "
|
|
248
|
-
"description": "
|
|
249
|
-
"family": "
|
|
255
|
+
"thinkingmachines/Inkling": {
|
|
256
|
+
"id": "thinkingmachines/Inkling",
|
|
257
|
+
"name": "Inkling",
|
|
258
|
+
"description": "Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio",
|
|
259
|
+
"family": "ling",
|
|
250
260
|
"attachment": true,
|
|
251
261
|
"reasoning": true,
|
|
252
262
|
"reasoning_options": [],
|
|
253
263
|
"tool_call": true,
|
|
254
|
-
"structured_output": true,
|
|
255
264
|
"temperature": true,
|
|
256
|
-
"release_date": "2026-
|
|
257
|
-
"last_updated": "2026-
|
|
265
|
+
"release_date": "2026-07-15",
|
|
266
|
+
"last_updated": "2026-07-15",
|
|
258
267
|
"modalities": {
|
|
259
268
|
"input": [
|
|
260
269
|
"text",
|
|
261
270
|
"image",
|
|
262
|
-
"video",
|
|
263
271
|
"audio"
|
|
264
272
|
],
|
|
265
273
|
"output": [
|
|
@@ -268,27 +276,36 @@
|
|
|
268
276
|
},
|
|
269
277
|
"open_weights": true,
|
|
270
278
|
"limit": {
|
|
271
|
-
"context":
|
|
272
|
-
"output":
|
|
279
|
+
"context": 524288,
|
|
280
|
+
"output": 1048576
|
|
273
281
|
},
|
|
274
282
|
"cost": {
|
|
275
|
-
"input": 0.
|
|
276
|
-
"output":
|
|
283
|
+
"input": 0.95,
|
|
284
|
+
"output": 4.05,
|
|
285
|
+
"cache_read": 0.16
|
|
277
286
|
}
|
|
278
287
|
},
|
|
279
|
-
"
|
|
280
|
-
"id": "
|
|
281
|
-
"name": "
|
|
282
|
-
"description": "
|
|
283
|
-
"family": "
|
|
288
|
+
"zai-org/GLM-5": {
|
|
289
|
+
"id": "zai-org/GLM-5",
|
|
290
|
+
"name": "GLM-5",
|
|
291
|
+
"description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
|
|
292
|
+
"family": "glm",
|
|
284
293
|
"attachment": false,
|
|
285
|
-
"reasoning":
|
|
294
|
+
"reasoning": true,
|
|
295
|
+
"reasoning_options": [
|
|
296
|
+
{
|
|
297
|
+
"type": "toggle"
|
|
298
|
+
}
|
|
299
|
+
],
|
|
286
300
|
"tool_call": true,
|
|
301
|
+
"interleaved": {
|
|
302
|
+
"field": "reasoning_content"
|
|
303
|
+
},
|
|
287
304
|
"structured_output": true,
|
|
288
305
|
"temperature": true,
|
|
289
|
-
"knowledge": "2025-
|
|
290
|
-
"release_date": "
|
|
291
|
-
"last_updated": "
|
|
306
|
+
"knowledge": "2025-12",
|
|
307
|
+
"release_date": "2026-02-12",
|
|
308
|
+
"last_updated": "2026-02-12",
|
|
292
309
|
"modalities": {
|
|
293
310
|
"input": [
|
|
294
311
|
"text"
|
|
@@ -299,26 +316,32 @@
|
|
|
299
316
|
},
|
|
300
317
|
"open_weights": true,
|
|
301
318
|
"limit": {
|
|
302
|
-
"context":
|
|
303
|
-
"output":
|
|
319
|
+
"context": 202752,
|
|
320
|
+
"output": 16384
|
|
304
321
|
},
|
|
305
322
|
"cost": {
|
|
306
|
-
"input": 0.
|
|
307
|
-
"output":
|
|
323
|
+
"input": 0.6,
|
|
324
|
+
"output": 2.08,
|
|
325
|
+
"cache_read": 0.12
|
|
308
326
|
}
|
|
309
327
|
},
|
|
310
|
-
"
|
|
311
|
-
"id": "
|
|
312
|
-
"name": "
|
|
313
|
-
"description": "
|
|
314
|
-
"family": "
|
|
328
|
+
"zai-org/GLM-4.7-Flash": {
|
|
329
|
+
"id": "zai-org/GLM-4.7-Flash",
|
|
330
|
+
"name": "GLM-4.7-Flash",
|
|
331
|
+
"description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
|
|
332
|
+
"family": "glm-flash",
|
|
315
333
|
"attachment": false,
|
|
316
|
-
"reasoning":
|
|
334
|
+
"reasoning": true,
|
|
335
|
+
"reasoning_options": [],
|
|
317
336
|
"tool_call": true,
|
|
337
|
+
"interleaved": {
|
|
338
|
+
"field": "reasoning_content"
|
|
339
|
+
},
|
|
318
340
|
"structured_output": true,
|
|
319
341
|
"temperature": true,
|
|
320
|
-
"
|
|
321
|
-
"
|
|
342
|
+
"knowledge": "2025-04",
|
|
343
|
+
"release_date": "2026-01-19",
|
|
344
|
+
"last_updated": "2026-01-19",
|
|
322
345
|
"modalities": {
|
|
323
346
|
"input": [
|
|
324
347
|
"text"
|
|
@@ -327,50 +350,86 @@
|
|
|
327
350
|
"text"
|
|
328
351
|
]
|
|
329
352
|
},
|
|
330
|
-
"open_weights":
|
|
353
|
+
"open_weights": true,
|
|
331
354
|
"limit": {
|
|
332
|
-
"context":
|
|
333
|
-
"output":
|
|
355
|
+
"context": 202752,
|
|
356
|
+
"output": 16384
|
|
334
357
|
},
|
|
335
358
|
"cost": {
|
|
336
|
-
"input":
|
|
337
|
-
"output":
|
|
338
|
-
"cache_read": 0.
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
359
|
+
"input": 0.06,
|
|
360
|
+
"output": 0.4,
|
|
361
|
+
"cache_read": 0.01
|
|
362
|
+
}
|
|
363
|
+
},
|
|
364
|
+
"zai-org/GLM-5.2": {
|
|
365
|
+
"id": "zai-org/GLM-5.2",
|
|
366
|
+
"name": "GLM-5.2",
|
|
367
|
+
"description": "Open flagship GLM for long-horizon coding agents and million-token context work",
|
|
368
|
+
"family": "glm",
|
|
369
|
+
"attachment": false,
|
|
370
|
+
"reasoning": true,
|
|
371
|
+
"reasoning_options": [
|
|
372
|
+
{
|
|
373
|
+
"type": "toggle"
|
|
374
|
+
},
|
|
375
|
+
{
|
|
376
|
+
"type": "effort",
|
|
377
|
+
"values": [
|
|
378
|
+
"low",
|
|
379
|
+
"medium",
|
|
380
|
+
"high",
|
|
381
|
+
"xhigh"
|
|
382
|
+
]
|
|
383
|
+
}
|
|
384
|
+
],
|
|
385
|
+
"tool_call": true,
|
|
386
|
+
"interleaved": {
|
|
387
|
+
"field": "reasoning_content"
|
|
388
|
+
},
|
|
389
|
+
"structured_output": true,
|
|
390
|
+
"temperature": true,
|
|
391
|
+
"release_date": "2026-06-13",
|
|
392
|
+
"last_updated": "2026-06-13",
|
|
393
|
+
"modalities": {
|
|
394
|
+
"input": [
|
|
395
|
+
"text"
|
|
396
|
+
],
|
|
397
|
+
"output": [
|
|
398
|
+
"text"
|
|
358
399
|
]
|
|
400
|
+
},
|
|
401
|
+
"open_weights": true,
|
|
402
|
+
"limit": {
|
|
403
|
+
"context": 1048576,
|
|
404
|
+
"output": 32768
|
|
405
|
+
},
|
|
406
|
+
"cost": {
|
|
407
|
+
"input": 0.75,
|
|
408
|
+
"output": 2.4,
|
|
409
|
+
"cache_read": 0.14
|
|
359
410
|
}
|
|
360
411
|
},
|
|
361
|
-
"
|
|
362
|
-
"id": "
|
|
363
|
-
"name": "
|
|
364
|
-
"description": "Flagship
|
|
365
|
-
"family": "
|
|
412
|
+
"zai-org/GLM-5.1": {
|
|
413
|
+
"id": "zai-org/GLM-5.1",
|
|
414
|
+
"name": "GLM-5.1",
|
|
415
|
+
"description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
|
|
416
|
+
"family": "glm",
|
|
366
417
|
"attachment": false,
|
|
367
|
-
"reasoning":
|
|
418
|
+
"reasoning": true,
|
|
419
|
+
"reasoning_options": [
|
|
420
|
+
{
|
|
421
|
+
"type": "toggle"
|
|
422
|
+
}
|
|
423
|
+
],
|
|
368
424
|
"tool_call": true,
|
|
425
|
+
"interleaved": {
|
|
426
|
+
"field": "reasoning_content"
|
|
427
|
+
},
|
|
369
428
|
"structured_output": true,
|
|
370
429
|
"temperature": true,
|
|
371
430
|
"knowledge": "2025-04",
|
|
372
|
-
"release_date": "
|
|
373
|
-
"last_updated": "
|
|
431
|
+
"release_date": "2026-04-07",
|
|
432
|
+
"last_updated": "2026-04-07",
|
|
374
433
|
"modalities": {
|
|
375
434
|
"input": [
|
|
376
435
|
"text"
|
|
@@ -379,40 +438,132 @@
|
|
|
379
438
|
"text"
|
|
380
439
|
]
|
|
381
440
|
},
|
|
382
|
-
"open_weights":
|
|
441
|
+
"open_weights": true,
|
|
383
442
|
"limit": {
|
|
384
|
-
"context":
|
|
385
|
-
"output":
|
|
443
|
+
"context": 202752,
|
|
444
|
+
"output": 16384
|
|
386
445
|
},
|
|
387
446
|
"cost": {
|
|
388
|
-
"input": 1.
|
|
389
|
-
"output":
|
|
390
|
-
"cache_read": 0.
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
447
|
+
"input": 1.05,
|
|
448
|
+
"output": 3.5,
|
|
449
|
+
"cache_read": 0.205
|
|
450
|
+
}
|
|
451
|
+
},
|
|
452
|
+
"zai-org/GLM-4.6": {
|
|
453
|
+
"id": "zai-org/GLM-4.6",
|
|
454
|
+
"name": "GLM-4.6",
|
|
455
|
+
"description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
|
|
456
|
+
"family": "glm",
|
|
457
|
+
"attachment": false,
|
|
458
|
+
"reasoning": true,
|
|
459
|
+
"reasoning_options": [
|
|
460
|
+
{
|
|
461
|
+
"type": "toggle"
|
|
462
|
+
}
|
|
463
|
+
],
|
|
464
|
+
"tool_call": true,
|
|
465
|
+
"interleaved": {
|
|
466
|
+
"field": "reasoning_content"
|
|
467
|
+
},
|
|
468
|
+
"structured_output": true,
|
|
469
|
+
"temperature": true,
|
|
470
|
+
"knowledge": "2025-04",
|
|
471
|
+
"release_date": "2025-09-30",
|
|
472
|
+
"last_updated": "2025-09-30",
|
|
473
|
+
"modalities": {
|
|
474
|
+
"input": [
|
|
475
|
+
"text"
|
|
476
|
+
],
|
|
477
|
+
"output": [
|
|
478
|
+
"text"
|
|
410
479
|
]
|
|
480
|
+
},
|
|
481
|
+
"open_weights": true,
|
|
482
|
+
"limit": {
|
|
483
|
+
"context": 202752,
|
|
484
|
+
"output": 131072
|
|
485
|
+
},
|
|
486
|
+
"cost": {
|
|
487
|
+
"input": 0.5,
|
|
488
|
+
"output": 2,
|
|
489
|
+
"cache_read": 0.1
|
|
411
490
|
}
|
|
412
491
|
},
|
|
413
|
-
"
|
|
414
|
-
"id": "
|
|
415
|
-
"name": "
|
|
492
|
+
"zai-org/GLM-4.7": {
|
|
493
|
+
"id": "zai-org/GLM-4.7",
|
|
494
|
+
"name": "GLM-4.7",
|
|
495
|
+
"description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
|
|
496
|
+
"family": "glm",
|
|
497
|
+
"attachment": false,
|
|
498
|
+
"reasoning": true,
|
|
499
|
+
"reasoning_options": [
|
|
500
|
+
{
|
|
501
|
+
"type": "toggle"
|
|
502
|
+
}
|
|
503
|
+
],
|
|
504
|
+
"tool_call": true,
|
|
505
|
+
"interleaved": {
|
|
506
|
+
"field": "reasoning_content"
|
|
507
|
+
},
|
|
508
|
+
"structured_output": true,
|
|
509
|
+
"temperature": true,
|
|
510
|
+
"knowledge": "2025-04",
|
|
511
|
+
"release_date": "2025-12-22",
|
|
512
|
+
"last_updated": "2025-12-22",
|
|
513
|
+
"modalities": {
|
|
514
|
+
"input": [
|
|
515
|
+
"text"
|
|
516
|
+
],
|
|
517
|
+
"output": [
|
|
518
|
+
"text"
|
|
519
|
+
]
|
|
520
|
+
},
|
|
521
|
+
"open_weights": true,
|
|
522
|
+
"limit": {
|
|
523
|
+
"context": 202752,
|
|
524
|
+
"output": 16384
|
|
525
|
+
},
|
|
526
|
+
"cost": {
|
|
527
|
+
"input": 0.4,
|
|
528
|
+
"output": 1.75,
|
|
529
|
+
"cache_read": 0.08
|
|
530
|
+
}
|
|
531
|
+
},
|
|
532
|
+
"tencent/Hy3": {
|
|
533
|
+
"id": "tencent/Hy3",
|
|
534
|
+
"name": "Hy3",
|
|
535
|
+
"description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
|
|
536
|
+
"family": "Hy",
|
|
537
|
+
"attachment": false,
|
|
538
|
+
"reasoning": true,
|
|
539
|
+
"reasoning_options": [],
|
|
540
|
+
"tool_call": true,
|
|
541
|
+
"structured_output": true,
|
|
542
|
+
"temperature": true,
|
|
543
|
+
"release_date": "2026-07-06",
|
|
544
|
+
"last_updated": "2026-07-06",
|
|
545
|
+
"modalities": {
|
|
546
|
+
"input": [
|
|
547
|
+
"text"
|
|
548
|
+
],
|
|
549
|
+
"output": [
|
|
550
|
+
"text"
|
|
551
|
+
]
|
|
552
|
+
},
|
|
553
|
+
"open_weights": true,
|
|
554
|
+
"limit": {
|
|
555
|
+
"context": 262144,
|
|
556
|
+
"output": 64000
|
|
557
|
+
},
|
|
558
|
+
"cost": {
|
|
559
|
+
"input": 0.14,
|
|
560
|
+
"output": 0.58,
|
|
561
|
+
"cache_read": 0.035
|
|
562
|
+
}
|
|
563
|
+
},
|
|
564
|
+
"Qwen/Qwen3.5-27B": {
|
|
565
|
+
"id": "Qwen/Qwen3.5-27B",
|
|
566
|
+
"name": "Qwen3.5 27B",
|
|
416
567
|
"description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
|
|
417
568
|
"family": "qwen",
|
|
418
569
|
"attachment": true,
|
|
@@ -421,8 +572,42 @@
|
|
|
421
572
|
"tool_call": true,
|
|
422
573
|
"structured_output": true,
|
|
423
574
|
"temperature": true,
|
|
424
|
-
"release_date": "2026-
|
|
425
|
-
"last_updated": "2026-
|
|
575
|
+
"release_date": "2026-02-23",
|
|
576
|
+
"last_updated": "2026-02-23",
|
|
577
|
+
"modalities": {
|
|
578
|
+
"input": [
|
|
579
|
+
"text",
|
|
580
|
+
"image",
|
|
581
|
+
"video",
|
|
582
|
+
"audio"
|
|
583
|
+
],
|
|
584
|
+
"output": [
|
|
585
|
+
"text"
|
|
586
|
+
]
|
|
587
|
+
},
|
|
588
|
+
"open_weights": true,
|
|
589
|
+
"limit": {
|
|
590
|
+
"context": 262144,
|
|
591
|
+
"output": 65536
|
|
592
|
+
},
|
|
593
|
+
"cost": {
|
|
594
|
+
"input": 0.26,
|
|
595
|
+
"output": 2.6
|
|
596
|
+
}
|
|
597
|
+
},
|
|
598
|
+
"Qwen/Qwen3.5-9B": {
|
|
599
|
+
"id": "Qwen/Qwen3.5-9B",
|
|
600
|
+
"name": "Qwen3.5 9B",
|
|
601
|
+
"description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
|
|
602
|
+
"family": "qwen",
|
|
603
|
+
"attachment": true,
|
|
604
|
+
"reasoning": true,
|
|
605
|
+
"reasoning_options": [],
|
|
606
|
+
"tool_call": true,
|
|
607
|
+
"structured_output": true,
|
|
608
|
+
"temperature": true,
|
|
609
|
+
"release_date": "2026-02-23",
|
|
610
|
+
"last_updated": "2026-02-23",
|
|
426
611
|
"modalities": {
|
|
427
612
|
"input": [
|
|
428
613
|
"text",
|
|
@@ -436,29 +621,25 @@
|
|
|
436
621
|
"open_weights": true,
|
|
437
622
|
"limit": {
|
|
438
623
|
"context": 262144,
|
|
439
|
-
"output":
|
|
624
|
+
"output": 65536
|
|
440
625
|
},
|
|
441
626
|
"cost": {
|
|
442
|
-
"input": 0.
|
|
443
|
-
"output": 0.
|
|
627
|
+
"input": 0.1,
|
|
628
|
+
"output": 0.15
|
|
444
629
|
}
|
|
445
630
|
},
|
|
446
|
-
"
|
|
447
|
-
"id": "
|
|
448
|
-
"name": "
|
|
449
|
-
"description": "
|
|
450
|
-
"family": "
|
|
631
|
+
"Qwen/Qwen3-235B-A22B-Instruct-2507": {
|
|
632
|
+
"id": "Qwen/Qwen3-235B-A22B-Instruct-2507",
|
|
633
|
+
"name": "Qwen3 235B-A22B Instruct 2507",
|
|
634
|
+
"description": "Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use",
|
|
635
|
+
"family": "qwen",
|
|
451
636
|
"attachment": false,
|
|
452
|
-
"reasoning":
|
|
453
|
-
"reasoning_options": [
|
|
454
|
-
{
|
|
455
|
-
"type": "toggle"
|
|
456
|
-
}
|
|
457
|
-
],
|
|
637
|
+
"reasoning": false,
|
|
458
638
|
"tool_call": true,
|
|
639
|
+
"structured_output": true,
|
|
459
640
|
"temperature": true,
|
|
460
|
-
"release_date": "2025-
|
|
461
|
-
"last_updated": "2025-
|
|
641
|
+
"release_date": "2025-07-21",
|
|
642
|
+
"last_updated": "2025-07-21",
|
|
462
643
|
"modalities": {
|
|
463
644
|
"input": [
|
|
464
645
|
"text"
|
|
@@ -470,26 +651,59 @@
|
|
|
470
651
|
"open_weights": true,
|
|
471
652
|
"limit": {
|
|
472
653
|
"context": 262144,
|
|
473
|
-
"output":
|
|
654
|
+
"output": 16384
|
|
474
655
|
},
|
|
475
656
|
"cost": {
|
|
476
|
-
"input": 0.
|
|
477
|
-
"output": 0.
|
|
657
|
+
"input": 0.09,
|
|
658
|
+
"output": 0.55
|
|
478
659
|
}
|
|
479
660
|
},
|
|
480
|
-
"
|
|
481
|
-
"id": "
|
|
482
|
-
"name": "
|
|
483
|
-
"description": "
|
|
484
|
-
"family": "
|
|
485
|
-
"attachment":
|
|
661
|
+
"Qwen/Qwen3.5-122B-A10B": {
|
|
662
|
+
"id": "Qwen/Qwen3.5-122B-A10B",
|
|
663
|
+
"name": "Qwen3.5 122B-A10B",
|
|
664
|
+
"description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
|
|
665
|
+
"family": "qwen",
|
|
666
|
+
"attachment": true,
|
|
486
667
|
"reasoning": true,
|
|
487
668
|
"reasoning_options": [],
|
|
488
669
|
"tool_call": true,
|
|
489
670
|
"structured_output": true,
|
|
490
671
|
"temperature": true,
|
|
491
|
-
"release_date": "
|
|
492
|
-
"last_updated": "
|
|
672
|
+
"release_date": "2026-02-23",
|
|
673
|
+
"last_updated": "2026-02-23",
|
|
674
|
+
"modalities": {
|
|
675
|
+
"input": [
|
|
676
|
+
"text",
|
|
677
|
+
"image",
|
|
678
|
+
"video",
|
|
679
|
+
"audio"
|
|
680
|
+
],
|
|
681
|
+
"output": [
|
|
682
|
+
"text"
|
|
683
|
+
]
|
|
684
|
+
},
|
|
685
|
+
"open_weights": true,
|
|
686
|
+
"limit": {
|
|
687
|
+
"context": 262144,
|
|
688
|
+
"output": 65536
|
|
689
|
+
},
|
|
690
|
+
"cost": {
|
|
691
|
+
"input": 0.29,
|
|
692
|
+
"output": 2.4
|
|
693
|
+
}
|
|
694
|
+
},
|
|
695
|
+
"Qwen/Qwen3.7-Max": {
|
|
696
|
+
"id": "Qwen/Qwen3.7-Max",
|
|
697
|
+
"name": "Qwen3.7 Max",
|
|
698
|
+
"description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
|
|
699
|
+
"family": "qwen",
|
|
700
|
+
"attachment": false,
|
|
701
|
+
"reasoning": false,
|
|
702
|
+
"tool_call": true,
|
|
703
|
+
"structured_output": true,
|
|
704
|
+
"temperature": true,
|
|
705
|
+
"release_date": "2026-05-21",
|
|
706
|
+
"last_updated": "2026-05-21",
|
|
493
707
|
"modalities": {
|
|
494
708
|
"input": [
|
|
495
709
|
"text"
|
|
@@ -498,30 +712,85 @@
|
|
|
498
712
|
"text"
|
|
499
713
|
]
|
|
500
714
|
},
|
|
715
|
+
"open_weights": false,
|
|
716
|
+
"limit": {
|
|
717
|
+
"context": 256000,
|
|
718
|
+
"output": 65536
|
|
719
|
+
},
|
|
720
|
+
"cost": {
|
|
721
|
+
"input": 2.5,
|
|
722
|
+
"output": 7.5,
|
|
723
|
+
"cache_read": 0.5,
|
|
724
|
+
"tiers": [
|
|
725
|
+
{
|
|
726
|
+
"input": 5,
|
|
727
|
+
"output": 15,
|
|
728
|
+
"cache_read": 1,
|
|
729
|
+
"tier": {
|
|
730
|
+
"type": "context",
|
|
731
|
+
"size": 32000
|
|
732
|
+
}
|
|
733
|
+
},
|
|
734
|
+
{
|
|
735
|
+
"input": 6.25,
|
|
736
|
+
"output": 18.5,
|
|
737
|
+
"cache_read": 1.25,
|
|
738
|
+
"tier": {
|
|
739
|
+
"type": "context",
|
|
740
|
+
"size": 128000
|
|
741
|
+
}
|
|
742
|
+
}
|
|
743
|
+
]
|
|
744
|
+
}
|
|
745
|
+
},
|
|
746
|
+
"Qwen/Qwen3.5-35B-A3B": {
|
|
747
|
+
"id": "Qwen/Qwen3.5-35B-A3B",
|
|
748
|
+
"name": "Qwen 3.5 35B A3B",
|
|
749
|
+
"description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
|
|
750
|
+
"family": "qwen",
|
|
751
|
+
"attachment": true,
|
|
752
|
+
"reasoning": true,
|
|
753
|
+
"reasoning_options": [],
|
|
754
|
+
"tool_call": true,
|
|
755
|
+
"structured_output": true,
|
|
756
|
+
"temperature": true,
|
|
757
|
+
"knowledge": "2025-01",
|
|
758
|
+
"release_date": "2026-02-01",
|
|
759
|
+
"last_updated": "2026-04-20",
|
|
760
|
+
"modalities": {
|
|
761
|
+
"input": [
|
|
762
|
+
"text",
|
|
763
|
+
"image",
|
|
764
|
+
"video"
|
|
765
|
+
],
|
|
766
|
+
"output": [
|
|
767
|
+
"text"
|
|
768
|
+
]
|
|
769
|
+
},
|
|
501
770
|
"open_weights": true,
|
|
502
771
|
"limit": {
|
|
503
|
-
"context":
|
|
504
|
-
"output":
|
|
772
|
+
"context": 262144,
|
|
773
|
+
"output": 81920
|
|
505
774
|
},
|
|
506
|
-
"status": "deprecated",
|
|
507
775
|
"cost": {
|
|
508
|
-
"input": 0.
|
|
509
|
-
"output":
|
|
776
|
+
"input": 0.14,
|
|
777
|
+
"output": 1,
|
|
778
|
+
"cache_read": 0.05
|
|
510
779
|
}
|
|
511
780
|
},
|
|
512
|
-
"
|
|
513
|
-
"id": "
|
|
514
|
-
"name": "
|
|
515
|
-
"description": "
|
|
516
|
-
"family": "
|
|
781
|
+
"Qwen/Qwen3.6-27B": {
|
|
782
|
+
"id": "Qwen/Qwen3.6-27B",
|
|
783
|
+
"name": "Qwen3.6 27B",
|
|
784
|
+
"description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
|
|
785
|
+
"family": "qwen",
|
|
517
786
|
"attachment": true,
|
|
518
787
|
"reasoning": true,
|
|
519
788
|
"reasoning_options": [],
|
|
520
789
|
"tool_call": true,
|
|
521
790
|
"structured_output": true,
|
|
522
791
|
"temperature": true,
|
|
523
|
-
"release_date": "2026-04-
|
|
524
|
-
"last_updated": "2026-04-
|
|
792
|
+
"release_date": "2026-04-22",
|
|
793
|
+
"last_updated": "2026-04-22",
|
|
525
794
|
"modalities": {
|
|
526
795
|
"input": [
|
|
527
796
|
"text",
|
|
@@ -538,27 +807,30 @@
|
|
|
538
807
|
"context": 262144,
|
|
539
808
|
"output": 65536
|
|
540
809
|
},
|
|
541
|
-
"status": "deprecated",
|
|
542
810
|
"cost": {
|
|
543
|
-
"input": 0.
|
|
544
|
-
"output":
|
|
811
|
+
"input": 0.32,
|
|
812
|
+
"output": 3.2
|
|
545
813
|
}
|
|
546
814
|
},
|
|
547
|
-
"
|
|
548
|
-
"id": "
|
|
549
|
-
"name": "
|
|
550
|
-
"description": "
|
|
551
|
-
"family": "
|
|
815
|
+
"Qwen/Qwen3.5-397B-A17B": {
|
|
816
|
+
"id": "Qwen/Qwen3.5-397B-A17B",
|
|
817
|
+
"name": "Qwen 3.5 397B A17B",
|
|
818
|
+
"description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
|
|
819
|
+
"family": "qwen",
|
|
552
820
|
"attachment": true,
|
|
553
|
-
"reasoning":
|
|
821
|
+
"reasoning": true,
|
|
822
|
+
"reasoning_options": [],
|
|
554
823
|
"tool_call": true,
|
|
555
824
|
"structured_output": true,
|
|
556
|
-
"
|
|
557
|
-
"
|
|
825
|
+
"temperature": true,
|
|
826
|
+
"knowledge": "2025-01",
|
|
827
|
+
"release_date": "2026-02-01",
|
|
828
|
+
"last_updated": "2026-04-20",
|
|
558
829
|
"modalities": {
|
|
559
830
|
"input": [
|
|
560
831
|
"text",
|
|
561
|
-
"image"
|
|
832
|
+
"image",
|
|
833
|
+
"video"
|
|
562
834
|
],
|
|
563
835
|
"output": [
|
|
564
836
|
"text"
|
|
@@ -566,29 +838,33 @@
|
|
|
566
838
|
},
|
|
567
839
|
"open_weights": true,
|
|
568
840
|
"limit": {
|
|
569
|
-
"context":
|
|
570
|
-
"output":
|
|
841
|
+
"context": 262144,
|
|
842
|
+
"output": 81920
|
|
571
843
|
},
|
|
572
844
|
"cost": {
|
|
573
|
-
"input": 0.
|
|
574
|
-
"output":
|
|
845
|
+
"input": 0.45,
|
|
846
|
+
"output": 3,
|
|
847
|
+
"cache_read": 0.22
|
|
575
848
|
}
|
|
576
849
|
},
|
|
577
|
-
"
|
|
578
|
-
"id": "
|
|
579
|
-
"name": "
|
|
580
|
-
"description": "
|
|
581
|
-
"family": "
|
|
850
|
+
"Qwen/Qwen3.6-35B-A3B": {
|
|
851
|
+
"id": "Qwen/Qwen3.6-35B-A3B",
|
|
852
|
+
"name": "Qwen3.6 35B A3B",
|
|
853
|
+
"description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
|
|
854
|
+
"family": "qwen",
|
|
582
855
|
"attachment": true,
|
|
583
|
-
"reasoning":
|
|
584
|
-
"
|
|
856
|
+
"reasoning": true,
|
|
857
|
+
"reasoning_options": [],
|
|
858
|
+
"tool_call": true,
|
|
585
859
|
"structured_output": true,
|
|
586
|
-
"
|
|
587
|
-
"
|
|
860
|
+
"temperature": true,
|
|
861
|
+
"release_date": "2026-04-01",
|
|
862
|
+
"last_updated": "2026-04-01",
|
|
588
863
|
"modalities": {
|
|
589
864
|
"input": [
|
|
590
865
|
"text",
|
|
591
|
-
"image"
|
|
866
|
+
"image",
|
|
867
|
+
"video"
|
|
592
868
|
],
|
|
593
869
|
"output": [
|
|
594
870
|
"text"
|
|
@@ -596,59 +872,62 @@
|
|
|
596
872
|
},
|
|
597
873
|
"open_weights": true,
|
|
598
874
|
"limit": {
|
|
599
|
-
"context":
|
|
600
|
-
"output":
|
|
875
|
+
"context": 262144,
|
|
876
|
+
"output": 81920
|
|
601
877
|
},
|
|
602
878
|
"cost": {
|
|
603
|
-
"input": 0.
|
|
604
|
-
"output": 0.
|
|
879
|
+
"input": 0.1,
|
|
880
|
+
"output": 0.95
|
|
605
881
|
}
|
|
606
882
|
},
|
|
607
|
-
"
|
|
608
|
-
"id": "
|
|
609
|
-
"name": "
|
|
610
|
-
"description": "
|
|
611
|
-
"family": "
|
|
612
|
-
"attachment":
|
|
883
|
+
"Qwen/Qwen3.8-Max": {
|
|
884
|
+
"id": "Qwen/Qwen3.8-Max",
|
|
885
|
+
"name": "Qwen3.8 Max",
|
|
886
|
+
"description": "2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows",
|
|
887
|
+
"family": "qwen",
|
|
888
|
+
"attachment": true,
|
|
613
889
|
"reasoning": false,
|
|
614
890
|
"tool_call": true,
|
|
615
891
|
"structured_output": true,
|
|
616
|
-
"
|
|
617
|
-
"
|
|
892
|
+
"temperature": true,
|
|
893
|
+
"release_date": "2026-08-03",
|
|
894
|
+
"last_updated": "2026-08-03",
|
|
618
895
|
"modalities": {
|
|
619
896
|
"input": [
|
|
620
|
-
"text"
|
|
897
|
+
"text",
|
|
898
|
+
"image",
|
|
899
|
+
"video",
|
|
900
|
+
"pdf"
|
|
621
901
|
],
|
|
622
902
|
"output": [
|
|
623
903
|
"text"
|
|
624
904
|
]
|
|
625
905
|
},
|
|
626
|
-
"open_weights":
|
|
906
|
+
"open_weights": false,
|
|
627
907
|
"limit": {
|
|
628
|
-
"context":
|
|
629
|
-
"output":
|
|
908
|
+
"context": 256000,
|
|
909
|
+
"output": 131072
|
|
630
910
|
},
|
|
631
911
|
"cost": {
|
|
632
|
-
"input":
|
|
633
|
-
"output":
|
|
912
|
+
"input": 1.65,
|
|
913
|
+
"output": 4.951,
|
|
914
|
+
"cache_read": 0.206
|
|
634
915
|
}
|
|
635
916
|
},
|
|
636
|
-
"
|
|
637
|
-
"id": "
|
|
638
|
-
"name": "
|
|
639
|
-
"description": "
|
|
917
|
+
"Qwen/Qwen3-32B": {
|
|
918
|
+
"id": "Qwen/Qwen3-32B",
|
|
919
|
+
"name": "Qwen3 32B",
|
|
920
|
+
"description": "Dense open Qwen model for self-hosted chat, reasoning, and coding",
|
|
921
|
+
"family": "qwen",
|
|
640
922
|
"attachment": false,
|
|
641
923
|
"reasoning": true,
|
|
642
924
|
"reasoning_options": [],
|
|
643
925
|
"tool_call": true,
|
|
644
|
-
"interleaved": {
|
|
645
|
-
"field": "reasoning_content"
|
|
646
|
-
},
|
|
647
926
|
"structured_output": true,
|
|
648
927
|
"temperature": true,
|
|
649
|
-
"knowledge": "
|
|
650
|
-
"release_date": "2025-
|
|
651
|
-
"last_updated": "2025-
|
|
928
|
+
"knowledge": "2025-04",
|
|
929
|
+
"release_date": "2025-04",
|
|
930
|
+
"last_updated": "2025-04",
|
|
652
931
|
"modalities": {
|
|
653
932
|
"input": [
|
|
654
933
|
"text"
|
|
@@ -657,47 +936,29 @@
|
|
|
657
936
|
"text"
|
|
658
937
|
]
|
|
659
938
|
},
|
|
660
|
-
"open_weights":
|
|
939
|
+
"open_weights": true,
|
|
661
940
|
"limit": {
|
|
662
|
-
"context":
|
|
663
|
-
"output":
|
|
941
|
+
"context": 40960,
|
|
942
|
+
"output": 16384
|
|
664
943
|
},
|
|
665
944
|
"cost": {
|
|
666
|
-
"input": 0.
|
|
667
|
-
"output":
|
|
668
|
-
"cache_read": 0.35
|
|
945
|
+
"input": 0.08,
|
|
946
|
+
"output": 0.28
|
|
669
947
|
}
|
|
670
948
|
},
|
|
671
|
-
"
|
|
672
|
-
"id": "
|
|
673
|
-
"name": "
|
|
674
|
-
"description": "
|
|
675
|
-
"family": "
|
|
949
|
+
"Qwen/Qwen3-Next-80B-A3B-Instruct": {
|
|
950
|
+
"id": "Qwen/Qwen3-Next-80B-A3B-Instruct",
|
|
951
|
+
"name": "Qwen3-Next 80B-A3B Instruct",
|
|
952
|
+
"description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
|
|
953
|
+
"family": "qwen",
|
|
676
954
|
"attachment": false,
|
|
677
|
-
"reasoning":
|
|
678
|
-
"reasoning_options": [
|
|
679
|
-
{
|
|
680
|
-
"type": "toggle"
|
|
681
|
-
},
|
|
682
|
-
{
|
|
683
|
-
"type": "effort",
|
|
684
|
-
"values": [
|
|
685
|
-
"low",
|
|
686
|
-
"medium",
|
|
687
|
-
"high",
|
|
688
|
-
"xhigh"
|
|
689
|
-
]
|
|
690
|
-
}
|
|
691
|
-
],
|
|
955
|
+
"reasoning": false,
|
|
692
956
|
"tool_call": true,
|
|
693
|
-
"interleaved": {
|
|
694
|
-
"field": "reasoning_content"
|
|
695
|
-
},
|
|
696
957
|
"structured_output": true,
|
|
697
958
|
"temperature": true,
|
|
698
|
-
"knowledge": "2025-
|
|
699
|
-
"release_date": "
|
|
700
|
-
"last_updated": "
|
|
959
|
+
"knowledge": "2025-04",
|
|
960
|
+
"release_date": "2025-09",
|
|
961
|
+
"last_updated": "2025-09",
|
|
701
962
|
"modalities": {
|
|
702
963
|
"input": [
|
|
703
964
|
"text"
|
|
@@ -708,45 +969,27 @@
|
|
|
708
969
|
},
|
|
709
970
|
"open_weights": true,
|
|
710
971
|
"limit": {
|
|
711
|
-
"context":
|
|
712
|
-
"output":
|
|
972
|
+
"context": 262144,
|
|
973
|
+
"output": 32768
|
|
713
974
|
},
|
|
714
975
|
"cost": {
|
|
715
|
-
"input":
|
|
716
|
-
"output":
|
|
717
|
-
"cache_read": 0.1
|
|
976
|
+
"input": 0.09,
|
|
977
|
+
"output": 1.1
|
|
718
978
|
}
|
|
719
979
|
},
|
|
720
|
-
"
|
|
721
|
-
"id": "
|
|
722
|
-
"name": "
|
|
723
|
-
"description": "
|
|
724
|
-
"family": "
|
|
980
|
+
"Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo": {
|
|
981
|
+
"id": "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo",
|
|
982
|
+
"name": "Qwen3 Coder 480B A35B Instruct Turbo",
|
|
983
|
+
"description": "Qwen coding model for software agents, repository edits, and code reasoning",
|
|
984
|
+
"family": "qwen",
|
|
725
985
|
"attachment": false,
|
|
726
|
-
"reasoning":
|
|
727
|
-
"reasoning_options": [
|
|
728
|
-
{
|
|
729
|
-
"type": "toggle"
|
|
730
|
-
},
|
|
731
|
-
{
|
|
732
|
-
"type": "effort",
|
|
733
|
-
"values": [
|
|
734
|
-
"low",
|
|
735
|
-
"medium",
|
|
736
|
-
"high",
|
|
737
|
-
"xhigh"
|
|
738
|
-
]
|
|
739
|
-
}
|
|
740
|
-
],
|
|
986
|
+
"reasoning": false,
|
|
741
987
|
"tool_call": true,
|
|
742
|
-
"interleaved": {
|
|
743
|
-
"field": "reasoning_content"
|
|
744
|
-
},
|
|
745
988
|
"structured_output": true,
|
|
746
989
|
"temperature": true,
|
|
747
|
-
"knowledge": "2025-
|
|
748
|
-
"release_date": "
|
|
749
|
-
"last_updated": "
|
|
990
|
+
"knowledge": "2025-04",
|
|
991
|
+
"release_date": "2025-07-23",
|
|
992
|
+
"last_updated": "2025-07-23",
|
|
750
993
|
"modalities": {
|
|
751
994
|
"input": [
|
|
752
995
|
"text"
|
|
@@ -757,35 +1000,28 @@
|
|
|
757
1000
|
},
|
|
758
1001
|
"open_weights": true,
|
|
759
1002
|
"limit": {
|
|
760
|
-
"context":
|
|
761
|
-
"output":
|
|
1003
|
+
"context": 262144,
|
|
1004
|
+
"output": 66536
|
|
762
1005
|
},
|
|
763
1006
|
"cost": {
|
|
764
|
-
"input": 0.
|
|
765
|
-
"output":
|
|
766
|
-
"cache_read": 0.
|
|
1007
|
+
"input": 0.3,
|
|
1008
|
+
"output": 1,
|
|
1009
|
+
"cache_read": 0.1
|
|
767
1010
|
}
|
|
768
1011
|
},
|
|
769
|
-
"
|
|
770
|
-
"id": "
|
|
771
|
-
"name": "
|
|
772
|
-
"description": "
|
|
1012
|
+
"Qwen/Qwen3-Max": {
|
|
1013
|
+
"id": "Qwen/Qwen3-Max",
|
|
1014
|
+
"name": "Qwen3 Max",
|
|
1015
|
+
"description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use",
|
|
1016
|
+
"family": "qwen",
|
|
773
1017
|
"attachment": false,
|
|
774
|
-
"reasoning":
|
|
775
|
-
"reasoning_options": [
|
|
776
|
-
{
|
|
777
|
-
"type": "toggle"
|
|
778
|
-
}
|
|
779
|
-
],
|
|
1018
|
+
"reasoning": false,
|
|
780
1019
|
"tool_call": true,
|
|
781
|
-
"interleaved": {
|
|
782
|
-
"field": "reasoning_content"
|
|
783
|
-
},
|
|
784
1020
|
"structured_output": true,
|
|
785
1021
|
"temperature": true,
|
|
786
|
-
"knowledge": "
|
|
787
|
-
"release_date": "2025-
|
|
788
|
-
"last_updated": "2025-
|
|
1022
|
+
"knowledge": "2025-04",
|
|
1023
|
+
"release_date": "2025-09-23",
|
|
1024
|
+
"last_updated": "2025-09-23",
|
|
789
1025
|
"modalities": {
|
|
790
1026
|
"input": [
|
|
791
1027
|
"text"
|
|
@@ -796,31 +1032,47 @@
|
|
|
796
1032
|
},
|
|
797
1033
|
"open_weights": false,
|
|
798
1034
|
"limit": {
|
|
799
|
-
"context":
|
|
800
|
-
"output":
|
|
1035
|
+
"context": 256000,
|
|
1036
|
+
"output": 65536
|
|
801
1037
|
},
|
|
802
1038
|
"cost": {
|
|
803
|
-
"input":
|
|
804
|
-
"output":
|
|
805
|
-
"cache_read": 0.
|
|
1039
|
+
"input": 1.2,
|
|
1040
|
+
"output": 6,
|
|
1041
|
+
"cache_read": 0.24,
|
|
1042
|
+
"tiers": [
|
|
1043
|
+
{
|
|
1044
|
+
"input": 2.4,
|
|
1045
|
+
"output": 12,
|
|
1046
|
+
"cache_read": 0.48,
|
|
1047
|
+
"tier": {
|
|
1048
|
+
"type": "context",
|
|
1049
|
+
"size": 32000
|
|
1050
|
+
}
|
|
1051
|
+
},
|
|
1052
|
+
{
|
|
1053
|
+
"input": 3,
|
|
1054
|
+
"output": 15,
|
|
1055
|
+
"cache_read": 0.6,
|
|
1056
|
+
"tier": {
|
|
1057
|
+
"type": "context",
|
|
1058
|
+
"size": 128000
|
|
1059
|
+
}
|
|
1060
|
+
}
|
|
1061
|
+
]
|
|
806
1062
|
}
|
|
807
1063
|
},
|
|
808
|
-
"MiniMaxAI/MiniMax-M2.
|
|
809
|
-
"id": "MiniMaxAI/MiniMax-M2.
|
|
810
|
-
"name": "MiniMax
|
|
811
|
-
"description": "MiniMax
|
|
1064
|
+
"MiniMaxAI/MiniMax-M2.7": {
|
|
1065
|
+
"id": "MiniMaxAI/MiniMax-M2.7",
|
|
1066
|
+
"name": "MiniMax-M2.7",
|
|
1067
|
+
"description": "Open MiniMax flagship for coding agents, office automation, and complex environments",
|
|
812
1068
|
"family": "minimax",
|
|
813
1069
|
"attachment": false,
|
|
814
1070
|
"reasoning": true,
|
|
815
1071
|
"reasoning_options": [],
|
|
816
1072
|
"tool_call": true,
|
|
817
|
-
"interleaved": {
|
|
818
|
-
"field": "reasoning_content"
|
|
819
|
-
},
|
|
820
1073
|
"temperature": true,
|
|
821
|
-
"
|
|
822
|
-
"
|
|
823
|
-
"last_updated": "2026-02-12",
|
|
1074
|
+
"release_date": "2026-03-18",
|
|
1075
|
+
"last_updated": "2026-03-18",
|
|
824
1076
|
"modalities": {
|
|
825
1077
|
"input": [
|
|
826
1078
|
"text"
|
|
@@ -834,25 +1086,28 @@
|
|
|
834
1086
|
"context": 196608,
|
|
835
1087
|
"output": 131072
|
|
836
1088
|
},
|
|
837
|
-
"status": "deprecated",
|
|
838
1089
|
"cost": {
|
|
839
|
-
"input": 0.
|
|
840
|
-
"output": 1
|
|
841
|
-
"cache_read": 0.
|
|
1090
|
+
"input": 0.25,
|
|
1091
|
+
"output": 1,
|
|
1092
|
+
"cache_read": 0.05
|
|
842
1093
|
}
|
|
843
1094
|
},
|
|
844
|
-
"MiniMaxAI/MiniMax-M2.
|
|
845
|
-
"id": "MiniMaxAI/MiniMax-M2.
|
|
846
|
-
"name": "MiniMax
|
|
847
|
-
"description": "
|
|
1095
|
+
"MiniMaxAI/MiniMax-M2.5": {
|
|
1096
|
+
"id": "MiniMaxAI/MiniMax-M2.5",
|
|
1097
|
+
"name": "MiniMax M2.5",
|
|
1098
|
+
"description": "MiniMax model for chat, coding, office work, and agentic tasks",
|
|
848
1099
|
"family": "minimax",
|
|
849
1100
|
"attachment": false,
|
|
850
1101
|
"reasoning": true,
|
|
851
1102
|
"reasoning_options": [],
|
|
852
1103
|
"tool_call": true,
|
|
1104
|
+
"interleaved": {
|
|
1105
|
+
"field": "reasoning_content"
|
|
1106
|
+
},
|
|
853
1107
|
"temperature": true,
|
|
854
|
-
"
|
|
855
|
-
"
|
|
1108
|
+
"knowledge": "2025-06",
|
|
1109
|
+
"release_date": "2026-02-12",
|
|
1110
|
+
"last_updated": "2026-02-12",
|
|
856
1111
|
"modalities": {
|
|
857
1112
|
"input": [
|
|
858
1113
|
"text"
|
|
@@ -866,10 +1121,11 @@
|
|
|
866
1121
|
"context": 196608,
|
|
867
1122
|
"output": 131072
|
|
868
1123
|
},
|
|
1124
|
+
"status": "deprecated",
|
|
869
1125
|
"cost": {
|
|
870
|
-
"input": 0.
|
|
871
|
-
"output": 1,
|
|
872
|
-
"cache_read": 0.
|
|
1126
|
+
"input": 0.15,
|
|
1127
|
+
"output": 1.15,
|
|
1128
|
+
"cache_read": 0.03
|
|
873
1129
|
}
|
|
874
1130
|
},
|
|
875
1131
|
"MiniMaxAI/MiniMax-M3": {
|
|
@@ -881,6 +1137,7 @@
|
|
|
881
1137
|
"reasoning": true,
|
|
882
1138
|
"reasoning_options": [],
|
|
883
1139
|
"tool_call": true,
|
|
1140
|
+
"structured_output": true,
|
|
884
1141
|
"temperature": true,
|
|
885
1142
|
"release_date": "2026-06-01",
|
|
886
1143
|
"last_updated": "2026-06-01",
|
|
@@ -905,23 +1162,18 @@
|
|
|
905
1162
|
"cache_read": 0.06
|
|
906
1163
|
}
|
|
907
1164
|
},
|
|
908
|
-
"
|
|
909
|
-
"id": "
|
|
910
|
-
"name": "
|
|
911
|
-
"description": "
|
|
912
|
-
"family": "
|
|
1165
|
+
"deepseek-ai/DeepSeek-V3": {
|
|
1166
|
+
"id": "deepseek-ai/DeepSeek-V3",
|
|
1167
|
+
"name": "DeepSeek-V3",
|
|
1168
|
+
"description": "Open DeepSeek MoE chat model for coding, math, and general reasoning",
|
|
1169
|
+
"family": "deepseek",
|
|
913
1170
|
"attachment": false,
|
|
914
|
-
"reasoning":
|
|
915
|
-
"
|
|
916
|
-
"tool_call": true,
|
|
917
|
-
"interleaved": {
|
|
918
|
-
"field": "reasoning_content"
|
|
919
|
-
},
|
|
1171
|
+
"reasoning": false,
|
|
1172
|
+
"tool_call": true,
|
|
920
1173
|
"structured_output": true,
|
|
921
1174
|
"temperature": true,
|
|
922
|
-
"
|
|
923
|
-
"
|
|
924
|
-
"last_updated": "2026-01-19",
|
|
1175
|
+
"release_date": "2024-12-26",
|
|
1176
|
+
"last_updated": "2024-12-26",
|
|
925
1177
|
"modalities": {
|
|
926
1178
|
"input": [
|
|
927
1179
|
"text"
|
|
@@ -932,20 +1184,19 @@
|
|
|
932
1184
|
},
|
|
933
1185
|
"open_weights": true,
|
|
934
1186
|
"limit": {
|
|
935
|
-
"context":
|
|
936
|
-
"output":
|
|
1187
|
+
"context": 163840,
|
|
1188
|
+
"output": 8192
|
|
937
1189
|
},
|
|
938
1190
|
"cost": {
|
|
939
|
-
"input": 0.
|
|
940
|
-
"output": 0.
|
|
941
|
-
"cache_read": 0.01
|
|
1191
|
+
"input": 0.32,
|
|
1192
|
+
"output": 0.89
|
|
942
1193
|
}
|
|
943
1194
|
},
|
|
944
|
-
"
|
|
945
|
-
"id": "
|
|
946
|
-
"name": "
|
|
947
|
-
"description": "
|
|
948
|
-
"family": "
|
|
1195
|
+
"deepseek-ai/DeepSeek-V4-Flash-0731": {
|
|
1196
|
+
"id": "deepseek-ai/DeepSeek-V4-Flash-0731",
|
|
1197
|
+
"name": "DeepSeek V4 Flash 0731",
|
|
1198
|
+
"description": "Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding",
|
|
1199
|
+
"family": "deepseek-flash",
|
|
949
1200
|
"attachment": false,
|
|
950
1201
|
"reasoning": true,
|
|
951
1202
|
"reasoning_options": [
|
|
@@ -954,14 +1205,11 @@
|
|
|
954
1205
|
}
|
|
955
1206
|
],
|
|
956
1207
|
"tool_call": true,
|
|
957
|
-
"interleaved": {
|
|
958
|
-
"field": "reasoning_content"
|
|
959
|
-
},
|
|
960
1208
|
"structured_output": true,
|
|
961
1209
|
"temperature": true,
|
|
962
|
-
"knowledge": "2025-
|
|
963
|
-
"release_date": "
|
|
964
|
-
"last_updated": "
|
|
1210
|
+
"knowledge": "2025-05",
|
|
1211
|
+
"release_date": "2026-07-31",
|
|
1212
|
+
"last_updated": "2026-07-31",
|
|
965
1213
|
"modalities": {
|
|
966
1214
|
"input": [
|
|
967
1215
|
"text"
|
|
@@ -972,44 +1220,31 @@
|
|
|
972
1220
|
},
|
|
973
1221
|
"open_weights": true,
|
|
974
1222
|
"limit": {
|
|
975
|
-
"context":
|
|
976
|
-
"output":
|
|
1223
|
+
"context": 1048576,
|
|
1224
|
+
"output": 384000
|
|
977
1225
|
},
|
|
978
1226
|
"cost": {
|
|
979
|
-
"input": 0.
|
|
980
|
-
"output":
|
|
981
|
-
"cache_read": 0.
|
|
1227
|
+
"input": 0.09,
|
|
1228
|
+
"output": 0.18,
|
|
1229
|
+
"cache_read": 0.018
|
|
982
1230
|
}
|
|
983
1231
|
},
|
|
984
|
-
"
|
|
985
|
-
"id": "
|
|
986
|
-
"name": "
|
|
987
|
-
"description": "
|
|
988
|
-
"family": "glm",
|
|
1232
|
+
"deepseek-ai/DeepSeek-R1-0528": {
|
|
1233
|
+
"id": "deepseek-ai/DeepSeek-R1-0528",
|
|
1234
|
+
"name": "DeepSeek-R1-0528",
|
|
1235
|
+
"description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
|
|
989
1236
|
"attachment": false,
|
|
990
1237
|
"reasoning": true,
|
|
991
|
-
"reasoning_options": [
|
|
992
|
-
{
|
|
993
|
-
"type": "toggle"
|
|
994
|
-
},
|
|
995
|
-
{
|
|
996
|
-
"type": "effort",
|
|
997
|
-
"values": [
|
|
998
|
-
"low",
|
|
999
|
-
"medium",
|
|
1000
|
-
"high",
|
|
1001
|
-
"xhigh"
|
|
1002
|
-
]
|
|
1003
|
-
}
|
|
1004
|
-
],
|
|
1238
|
+
"reasoning_options": [],
|
|
1005
1239
|
"tool_call": true,
|
|
1006
1240
|
"interleaved": {
|
|
1007
1241
|
"field": "reasoning_content"
|
|
1008
1242
|
},
|
|
1009
1243
|
"structured_output": true,
|
|
1010
1244
|
"temperature": true,
|
|
1011
|
-
"
|
|
1012
|
-
"
|
|
1245
|
+
"knowledge": "2024-07",
|
|
1246
|
+
"release_date": "2025-05-28",
|
|
1247
|
+
"last_updated": "2025-05-28",
|
|
1013
1248
|
"modalities": {
|
|
1014
1249
|
"input": [
|
|
1015
1250
|
"text"
|
|
@@ -1018,22 +1253,21 @@
|
|
|
1018
1253
|
"text"
|
|
1019
1254
|
]
|
|
1020
1255
|
},
|
|
1021
|
-
"open_weights":
|
|
1256
|
+
"open_weights": false,
|
|
1022
1257
|
"limit": {
|
|
1023
|
-
"context":
|
|
1024
|
-
"output":
|
|
1258
|
+
"context": 163840,
|
|
1259
|
+
"output": 64000
|
|
1025
1260
|
},
|
|
1026
1261
|
"cost": {
|
|
1027
|
-
"input": 0.
|
|
1028
|
-
"output":
|
|
1029
|
-
"cache_read": 0.
|
|
1262
|
+
"input": 0.5,
|
|
1263
|
+
"output": 2.15,
|
|
1264
|
+
"cache_read": 0.35
|
|
1030
1265
|
}
|
|
1031
1266
|
},
|
|
1032
|
-
"
|
|
1033
|
-
"id": "
|
|
1034
|
-
"name": "
|
|
1035
|
-
"description": "
|
|
1036
|
-
"family": "glm",
|
|
1267
|
+
"deepseek-ai/DeepSeek-V3.2": {
|
|
1268
|
+
"id": "deepseek-ai/DeepSeek-V3.2",
|
|
1269
|
+
"name": "DeepSeek-V3.2",
|
|
1270
|
+
"description": "DeepSeek chat model for instruction following, coding, and analysis",
|
|
1037
1271
|
"attachment": false,
|
|
1038
1272
|
"reasoning": true,
|
|
1039
1273
|
"reasoning_options": [
|
|
@@ -1047,9 +1281,9 @@
|
|
|
1047
1281
|
},
|
|
1048
1282
|
"structured_output": true,
|
|
1049
1283
|
"temperature": true,
|
|
1050
|
-
"knowledge": "
|
|
1051
|
-
"release_date": "
|
|
1052
|
-
"last_updated": "
|
|
1284
|
+
"knowledge": "2024-12",
|
|
1285
|
+
"release_date": "2025-12-02",
|
|
1286
|
+
"last_updated": "2025-12-02",
|
|
1053
1287
|
"modalities": {
|
|
1054
1288
|
"input": [
|
|
1055
1289
|
"text"
|
|
@@ -1058,22 +1292,22 @@
|
|
|
1058
1292
|
"text"
|
|
1059
1293
|
]
|
|
1060
1294
|
},
|
|
1061
|
-
"open_weights":
|
|
1295
|
+
"open_weights": false,
|
|
1062
1296
|
"limit": {
|
|
1063
|
-
"context":
|
|
1064
|
-
"output":
|
|
1297
|
+
"context": 163840,
|
|
1298
|
+
"output": 64000
|
|
1065
1299
|
},
|
|
1066
1300
|
"cost": {
|
|
1067
|
-
"input": 0.
|
|
1068
|
-
"output":
|
|
1069
|
-
"cache_read": 0.
|
|
1301
|
+
"input": 0.26,
|
|
1302
|
+
"output": 0.38,
|
|
1303
|
+
"cache_read": 0.13
|
|
1070
1304
|
}
|
|
1071
1305
|
},
|
|
1072
|
-
"
|
|
1073
|
-
"id": "
|
|
1074
|
-
"name": "
|
|
1075
|
-
"description": "
|
|
1076
|
-
"family": "
|
|
1306
|
+
"deepseek-ai/DeepSeek-V3.1": {
|
|
1307
|
+
"id": "deepseek-ai/DeepSeek-V3.1",
|
|
1308
|
+
"name": "DeepSeek-V3.1",
|
|
1309
|
+
"description": "Hybrid-reasoning DeepSeek model with thinking and non-thinking modes",
|
|
1310
|
+
"family": "deepseek",
|
|
1077
1311
|
"attachment": false,
|
|
1078
1312
|
"reasoning": true,
|
|
1079
1313
|
"reasoning_options": [
|
|
@@ -1082,14 +1316,10 @@
|
|
|
1082
1316
|
}
|
|
1083
1317
|
],
|
|
1084
1318
|
"tool_call": true,
|
|
1085
|
-
"interleaved": {
|
|
1086
|
-
"field": "reasoning_content"
|
|
1087
|
-
},
|
|
1088
1319
|
"structured_output": true,
|
|
1089
1320
|
"temperature": true,
|
|
1090
|
-
"
|
|
1091
|
-
"
|
|
1092
|
-
"last_updated": "2025-12-22",
|
|
1321
|
+
"release_date": "2025-08-21",
|
|
1322
|
+
"last_updated": "2025-08-21",
|
|
1093
1323
|
"modalities": {
|
|
1094
1324
|
"input": [
|
|
1095
1325
|
"text"
|
|
@@ -1100,25 +1330,34 @@
|
|
|
1100
1330
|
},
|
|
1101
1331
|
"open_weights": true,
|
|
1102
1332
|
"limit": {
|
|
1103
|
-
"context":
|
|
1104
|
-
"output":
|
|
1333
|
+
"context": 163840,
|
|
1334
|
+
"output": 8192
|
|
1105
1335
|
},
|
|
1106
1336
|
"cost": {
|
|
1107
|
-
"input": 0.
|
|
1108
|
-
"output":
|
|
1109
|
-
"cache_read": 0.
|
|
1337
|
+
"input": 0.25,
|
|
1338
|
+
"output": 0.95,
|
|
1339
|
+
"cache_read": 0.13
|
|
1110
1340
|
}
|
|
1111
1341
|
},
|
|
1112
|
-
"
|
|
1113
|
-
"id": "
|
|
1114
|
-
"name": "
|
|
1115
|
-
"description": "
|
|
1116
|
-
"family": "
|
|
1342
|
+
"deepseek-ai/DeepSeek-V4-Flash": {
|
|
1343
|
+
"id": "deepseek-ai/DeepSeek-V4-Flash",
|
|
1344
|
+
"name": "DeepSeek V4 Flash",
|
|
1345
|
+
"description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
|
|
1346
|
+
"family": "deepseek-flash",
|
|
1117
1347
|
"attachment": false,
|
|
1118
1348
|
"reasoning": true,
|
|
1119
1349
|
"reasoning_options": [
|
|
1120
1350
|
{
|
|
1121
1351
|
"type": "toggle"
|
|
1352
|
+
},
|
|
1353
|
+
{
|
|
1354
|
+
"type": "effort",
|
|
1355
|
+
"values": [
|
|
1356
|
+
"low",
|
|
1357
|
+
"medium",
|
|
1358
|
+
"high",
|
|
1359
|
+
"xhigh"
|
|
1360
|
+
]
|
|
1122
1361
|
}
|
|
1123
1362
|
],
|
|
1124
1363
|
"tool_call": true,
|
|
@@ -1127,9 +1366,9 @@
|
|
|
1127
1366
|
},
|
|
1128
1367
|
"structured_output": true,
|
|
1129
1368
|
"temperature": true,
|
|
1130
|
-
"knowledge": "2025-
|
|
1131
|
-
"release_date": "2026-04-
|
|
1132
|
-
"last_updated": "2026-04-
|
|
1369
|
+
"knowledge": "2025-05",
|
|
1370
|
+
"release_date": "2026-04-24",
|
|
1371
|
+
"last_updated": "2026-04-24",
|
|
1133
1372
|
"modalities": {
|
|
1134
1373
|
"input": [
|
|
1135
1374
|
"text"
|
|
@@ -1140,25 +1379,34 @@
|
|
|
1140
1379
|
},
|
|
1141
1380
|
"open_weights": true,
|
|
1142
1381
|
"limit": {
|
|
1143
|
-
"context":
|
|
1382
|
+
"context": 1048576,
|
|
1144
1383
|
"output": 16384
|
|
1145
1384
|
},
|
|
1146
1385
|
"cost": {
|
|
1147
|
-
"input":
|
|
1148
|
-
"output":
|
|
1149
|
-
"cache_read": 0.
|
|
1386
|
+
"input": 0.09,
|
|
1387
|
+
"output": 0.18,
|
|
1388
|
+
"cache_read": 0.018
|
|
1150
1389
|
}
|
|
1151
1390
|
},
|
|
1152
|
-
"
|
|
1153
|
-
"id": "
|
|
1154
|
-
"name": "
|
|
1155
|
-
"description": "
|
|
1156
|
-
"family": "
|
|
1157
|
-
"attachment":
|
|
1391
|
+
"deepseek-ai/DeepSeek-V4-Pro": {
|
|
1392
|
+
"id": "deepseek-ai/DeepSeek-V4-Pro",
|
|
1393
|
+
"name": "DeepSeek V4 Pro",
|
|
1394
|
+
"description": "Open MoE flagship with million-token context for coding and long agent runs",
|
|
1395
|
+
"family": "deepseek-thinking",
|
|
1396
|
+
"attachment": false,
|
|
1158
1397
|
"reasoning": true,
|
|
1159
1398
|
"reasoning_options": [
|
|
1160
1399
|
{
|
|
1161
1400
|
"type": "toggle"
|
|
1401
|
+
},
|
|
1402
|
+
{
|
|
1403
|
+
"type": "effort",
|
|
1404
|
+
"values": [
|
|
1405
|
+
"low",
|
|
1406
|
+
"medium",
|
|
1407
|
+
"high",
|
|
1408
|
+
"xhigh"
|
|
1409
|
+
]
|
|
1162
1410
|
}
|
|
1163
1411
|
],
|
|
1164
1412
|
"tool_call": true,
|
|
@@ -1167,14 +1415,12 @@
|
|
|
1167
1415
|
},
|
|
1168
1416
|
"structured_output": true,
|
|
1169
1417
|
"temperature": true,
|
|
1170
|
-
"knowledge": "
|
|
1171
|
-
"release_date": "2026-04-
|
|
1172
|
-
"last_updated": "2026-04-
|
|
1418
|
+
"knowledge": "2025-05",
|
|
1419
|
+
"release_date": "2026-04-24",
|
|
1420
|
+
"last_updated": "2026-04-24",
|
|
1173
1421
|
"modalities": {
|
|
1174
1422
|
"input": [
|
|
1175
|
-
"text"
|
|
1176
|
-
"image",
|
|
1177
|
-
"video"
|
|
1423
|
+
"text"
|
|
1178
1424
|
],
|
|
1179
1425
|
"output": [
|
|
1180
1426
|
"text"
|
|
@@ -1182,36 +1428,27 @@
|
|
|
1182
1428
|
},
|
|
1183
1429
|
"open_weights": true,
|
|
1184
1430
|
"limit": {
|
|
1185
|
-
"context":
|
|
1431
|
+
"context": 1048576,
|
|
1186
1432
|
"output": 16384
|
|
1187
1433
|
},
|
|
1188
1434
|
"cost": {
|
|
1189
|
-
"input":
|
|
1190
|
-
"output":
|
|
1191
|
-
"cache_read": 0.
|
|
1435
|
+
"input": 1.3,
|
|
1436
|
+
"output": 2.6,
|
|
1437
|
+
"cache_read": 0.1
|
|
1192
1438
|
}
|
|
1193
1439
|
},
|
|
1194
|
-
"
|
|
1195
|
-
"id": "
|
|
1196
|
-
"name": "
|
|
1197
|
-
"description": "
|
|
1198
|
-
"family": "kimi-k2",
|
|
1440
|
+
"stepfun-ai/Step-3.7-Flash": {
|
|
1441
|
+
"id": "stepfun-ai/Step-3.7-Flash",
|
|
1442
|
+
"name": "Step 3.7 Flash",
|
|
1443
|
+
"description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts",
|
|
1199
1444
|
"attachment": true,
|
|
1200
1445
|
"reasoning": true,
|
|
1201
|
-
"reasoning_options": [
|
|
1202
|
-
{
|
|
1203
|
-
"type": "toggle"
|
|
1204
|
-
}
|
|
1205
|
-
],
|
|
1446
|
+
"reasoning_options": [],
|
|
1206
1447
|
"tool_call": true,
|
|
1207
|
-
"interleaved": {
|
|
1208
|
-
"field": "reasoning_content"
|
|
1209
|
-
},
|
|
1210
|
-
"structured_output": true,
|
|
1211
1448
|
"temperature": true,
|
|
1212
|
-
"knowledge": "
|
|
1213
|
-
"release_date": "2026-
|
|
1214
|
-
"last_updated": "2026-
|
|
1449
|
+
"knowledge": "2026-03-01",
|
|
1450
|
+
"release_date": "2026-05-29",
|
|
1451
|
+
"last_updated": "2026-05-29",
|
|
1215
1452
|
"modalities": {
|
|
1216
1453
|
"input": [
|
|
1217
1454
|
"text",
|
|
@@ -1225,18 +1462,18 @@
|
|
|
1225
1462
|
"open_weights": true,
|
|
1226
1463
|
"limit": {
|
|
1227
1464
|
"context": 262144,
|
|
1228
|
-
"output":
|
|
1465
|
+
"output": 256000
|
|
1229
1466
|
},
|
|
1230
1467
|
"cost": {
|
|
1231
|
-
"input": 0.
|
|
1232
|
-
"output":
|
|
1233
|
-
"cache_read": 0.
|
|
1468
|
+
"input": 0.2,
|
|
1469
|
+
"output": 1.15,
|
|
1470
|
+
"cache_read": 0.04
|
|
1234
1471
|
}
|
|
1235
1472
|
},
|
|
1236
|
-
"moonshotai/Kimi-K2.
|
|
1237
|
-
"id": "moonshotai/Kimi-K2.
|
|
1238
|
-
"name": "Kimi K2.
|
|
1239
|
-
"description": "
|
|
1473
|
+
"moonshotai/Kimi-K2.6": {
|
|
1474
|
+
"id": "moonshotai/Kimi-K2.6",
|
|
1475
|
+
"name": "Kimi K2.6",
|
|
1476
|
+
"description": "Kimi multimodal agent model for visual understanding, coding, and planning",
|
|
1240
1477
|
"family": "kimi-k2",
|
|
1241
1478
|
"attachment": true,
|
|
1242
1479
|
"reasoning": true,
|
|
@@ -1250,10 +1487,10 @@
|
|
|
1250
1487
|
"field": "reasoning_content"
|
|
1251
1488
|
},
|
|
1252
1489
|
"structured_output": true,
|
|
1253
|
-
"temperature":
|
|
1254
|
-
"knowledge": "
|
|
1255
|
-
"release_date": "2026-
|
|
1256
|
-
"last_updated": "2026-
|
|
1490
|
+
"temperature": true,
|
|
1491
|
+
"knowledge": "2024-04",
|
|
1492
|
+
"release_date": "2026-04-21",
|
|
1493
|
+
"last_updated": "2026-04-21",
|
|
1257
1494
|
"modalities": {
|
|
1258
1495
|
"input": [
|
|
1259
1496
|
"text",
|
|
@@ -1267,19 +1504,19 @@
|
|
|
1267
1504
|
"open_weights": true,
|
|
1268
1505
|
"limit": {
|
|
1269
1506
|
"context": 262144,
|
|
1270
|
-
"output":
|
|
1507
|
+
"output": 16384
|
|
1271
1508
|
},
|
|
1272
1509
|
"cost": {
|
|
1273
|
-
"input": 0.
|
|
1510
|
+
"input": 0.75,
|
|
1274
1511
|
"output": 3.5,
|
|
1275
1512
|
"cache_read": 0.15
|
|
1276
1513
|
}
|
|
1277
1514
|
},
|
|
1278
|
-
"
|
|
1279
|
-
"id": "
|
|
1280
|
-
"name": "
|
|
1281
|
-
"description": "
|
|
1282
|
-
"family": "
|
|
1515
|
+
"moonshotai/Kimi-K2.5": {
|
|
1516
|
+
"id": "moonshotai/Kimi-K2.5",
|
|
1517
|
+
"name": "Kimi K2.5",
|
|
1518
|
+
"description": "Kimi multimodal agent model for visual understanding, coding, and planning",
|
|
1519
|
+
"family": "kimi-k2",
|
|
1283
1520
|
"attachment": true,
|
|
1284
1521
|
"reasoning": true,
|
|
1285
1522
|
"reasoning_options": [
|
|
@@ -1293,13 +1530,14 @@
|
|
|
1293
1530
|
},
|
|
1294
1531
|
"structured_output": true,
|
|
1295
1532
|
"temperature": true,
|
|
1296
|
-
"knowledge": "
|
|
1297
|
-
"release_date": "2026-
|
|
1298
|
-
"last_updated": "2026-
|
|
1533
|
+
"knowledge": "2025-01",
|
|
1534
|
+
"release_date": "2026-01-27",
|
|
1535
|
+
"last_updated": "2026-01-27",
|
|
1299
1536
|
"modalities": {
|
|
1300
1537
|
"input": [
|
|
1301
1538
|
"text",
|
|
1302
|
-
"
|
|
1539
|
+
"image",
|
|
1540
|
+
"video"
|
|
1303
1541
|
],
|
|
1304
1542
|
"output": [
|
|
1305
1543
|
"text"
|
|
@@ -1307,20 +1545,20 @@
|
|
|
1307
1545
|
},
|
|
1308
1546
|
"open_weights": true,
|
|
1309
1547
|
"limit": {
|
|
1310
|
-
"context":
|
|
1311
|
-
"output":
|
|
1548
|
+
"context": 262144,
|
|
1549
|
+
"output": 32768
|
|
1312
1550
|
},
|
|
1313
1551
|
"cost": {
|
|
1314
|
-
"input":
|
|
1315
|
-
"output":
|
|
1316
|
-
"cache_read": 0.
|
|
1552
|
+
"input": 0.45,
|
|
1553
|
+
"output": 2.25,
|
|
1554
|
+
"cache_read": 0.07
|
|
1317
1555
|
}
|
|
1318
1556
|
},
|
|
1319
|
-
"
|
|
1320
|
-
"id": "
|
|
1321
|
-
"name": "
|
|
1322
|
-
"description": "
|
|
1323
|
-
"family": "
|
|
1557
|
+
"moonshotai/Kimi-K2.7-Code": {
|
|
1558
|
+
"id": "moonshotai/Kimi-K2.7-Code",
|
|
1559
|
+
"name": "Kimi K2.7 Code",
|
|
1560
|
+
"description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
|
|
1561
|
+
"family": "kimi-k2",
|
|
1324
1562
|
"attachment": true,
|
|
1325
1563
|
"reasoning": true,
|
|
1326
1564
|
"reasoning_options": [
|
|
@@ -1333,15 +1571,14 @@
|
|
|
1333
1571
|
"field": "reasoning_content"
|
|
1334
1572
|
},
|
|
1335
1573
|
"structured_output": true,
|
|
1336
|
-
"temperature":
|
|
1337
|
-
"knowledge": "
|
|
1338
|
-
"release_date": "2026-
|
|
1339
|
-
"last_updated": "2026-
|
|
1574
|
+
"temperature": false,
|
|
1575
|
+
"knowledge": "2025-01",
|
|
1576
|
+
"release_date": "2026-06-12",
|
|
1577
|
+
"last_updated": "2026-06-12",
|
|
1340
1578
|
"modalities": {
|
|
1341
1579
|
"input": [
|
|
1342
1580
|
"text",
|
|
1343
1581
|
"image",
|
|
1344
|
-
"audio",
|
|
1345
1582
|
"video"
|
|
1346
1583
|
],
|
|
1347
1584
|
"output": [
|
|
@@ -1351,31 +1588,36 @@
|
|
|
1351
1588
|
"open_weights": true,
|
|
1352
1589
|
"limit": {
|
|
1353
1590
|
"context": 262144,
|
|
1354
|
-
"output":
|
|
1591
|
+
"output": 262144
|
|
1355
1592
|
},
|
|
1356
1593
|
"cost": {
|
|
1357
|
-
"input": 0.
|
|
1358
|
-
"output":
|
|
1359
|
-
"cache_read": 0.
|
|
1594
|
+
"input": 0.74,
|
|
1595
|
+
"output": 3.5,
|
|
1596
|
+
"cache_read": 0.15
|
|
1360
1597
|
}
|
|
1361
1598
|
},
|
|
1362
|
-
"
|
|
1363
|
-
"id": "
|
|
1364
|
-
"name": "
|
|
1365
|
-
"description": "
|
|
1366
|
-
"family": "
|
|
1599
|
+
"moonshotai/Kimi-K3": {
|
|
1600
|
+
"id": "moonshotai/Kimi-K3",
|
|
1601
|
+
"name": "Kimi K3",
|
|
1602
|
+
"description": "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work",
|
|
1603
|
+
"family": "kimi-k3",
|
|
1367
1604
|
"attachment": true,
|
|
1368
1605
|
"reasoning": true,
|
|
1369
1606
|
"reasoning_options": [
|
|
1370
1607
|
{
|
|
1371
|
-
"type": "
|
|
1608
|
+
"type": "effort",
|
|
1609
|
+
"values": [
|
|
1610
|
+
"low",
|
|
1611
|
+
"high",
|
|
1612
|
+
"max"
|
|
1613
|
+
]
|
|
1372
1614
|
}
|
|
1373
1615
|
],
|
|
1374
1616
|
"tool_call": true,
|
|
1375
1617
|
"structured_output": true,
|
|
1376
|
-
"temperature":
|
|
1377
|
-
"release_date": "2026-
|
|
1378
|
-
"last_updated": "2026-
|
|
1618
|
+
"temperature": false,
|
|
1619
|
+
"release_date": "2026-07-16",
|
|
1620
|
+
"last_updated": "2026-07-16",
|
|
1379
1621
|
"modalities": {
|
|
1380
1622
|
"input": [
|
|
1381
1623
|
"text",
|
|
@@ -1387,36 +1629,40 @@
|
|
|
1387
1629
|
},
|
|
1388
1630
|
"open_weights": true,
|
|
1389
1631
|
"limit": {
|
|
1390
|
-
"context":
|
|
1391
|
-
"output":
|
|
1632
|
+
"context": 1048576,
|
|
1633
|
+
"output": 131072
|
|
1392
1634
|
},
|
|
1393
1635
|
"cost": {
|
|
1394
|
-
"input":
|
|
1395
|
-
"output":
|
|
1636
|
+
"input": 2.7,
|
|
1637
|
+
"output": 13.5,
|
|
1638
|
+
"cache_read": 0.27
|
|
1396
1639
|
}
|
|
1397
1640
|
},
|
|
1398
|
-
"
|
|
1399
|
-
"id": "
|
|
1400
|
-
"name": "
|
|
1401
|
-
"description": "
|
|
1402
|
-
"family": "
|
|
1403
|
-
"attachment":
|
|
1641
|
+
"openai/gpt-oss-20b": {
|
|
1642
|
+
"id": "openai/gpt-oss-20b",
|
|
1643
|
+
"name": "GPT OSS 20B",
|
|
1644
|
+
"description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
|
|
1645
|
+
"family": "gpt-oss",
|
|
1646
|
+
"attachment": false,
|
|
1404
1647
|
"reasoning": true,
|
|
1405
1648
|
"reasoning_options": [
|
|
1406
1649
|
{
|
|
1407
|
-
"type": "
|
|
1650
|
+
"type": "effort",
|
|
1651
|
+
"values": [
|
|
1652
|
+
"low",
|
|
1653
|
+
"medium",
|
|
1654
|
+
"high"
|
|
1655
|
+
]
|
|
1408
1656
|
}
|
|
1409
1657
|
],
|
|
1410
1658
|
"tool_call": true,
|
|
1411
1659
|
"structured_output": true,
|
|
1412
1660
|
"temperature": true,
|
|
1413
|
-
"release_date": "
|
|
1414
|
-
"last_updated": "
|
|
1661
|
+
"release_date": "2025-08-05",
|
|
1662
|
+
"last_updated": "2025-08-05",
|
|
1415
1663
|
"modalities": {
|
|
1416
1664
|
"input": [
|
|
1417
|
-
"text"
|
|
1418
|
-
"image",
|
|
1419
|
-
"video"
|
|
1665
|
+
"text"
|
|
1420
1666
|
],
|
|
1421
1667
|
"output": [
|
|
1422
1668
|
"text"
|
|
@@ -1424,12 +1670,12 @@
|
|
|
1424
1670
|
},
|
|
1425
1671
|
"open_weights": true,
|
|
1426
1672
|
"limit": {
|
|
1427
|
-
"context":
|
|
1428
|
-
"output":
|
|
1673
|
+
"context": 131072,
|
|
1674
|
+
"output": 16384
|
|
1429
1675
|
},
|
|
1430
1676
|
"cost": {
|
|
1431
|
-
"input": 0.
|
|
1432
|
-
"output": 0.
|
|
1677
|
+
"input": 0.03,
|
|
1678
|
+
"output": 0.14
|
|
1433
1679
|
}
|
|
1434
1680
|
},
|
|
1435
1681
|
"openai/gpt-oss-120b": {
|
|
@@ -1472,31 +1718,163 @@
|
|
|
1472
1718
|
"output": 0.17
|
|
1473
1719
|
}
|
|
1474
1720
|
},
|
|
1475
|
-
"
|
|
1476
|
-
"id": "
|
|
1477
|
-
"name": "
|
|
1478
|
-
"description": "Open
|
|
1479
|
-
"family": "
|
|
1721
|
+
"meta-llama/Llama-4-Scout-17B-16E-Instruct": {
|
|
1722
|
+
"id": "meta-llama/Llama-4-Scout-17B-16E-Instruct",
|
|
1723
|
+
"name": "Llama 4 Scout 17B",
|
|
1724
|
+
"description": "Open multimodal Llama model for long-context analysis and efficient agents",
|
|
1725
|
+
"family": "llama",
|
|
1726
|
+
"attachment": true,
|
|
1727
|
+
"reasoning": false,
|
|
1728
|
+
"tool_call": true,
|
|
1729
|
+
"structured_output": true,
|
|
1730
|
+
"release_date": "2025-04-05",
|
|
1731
|
+
"last_updated": "2025-04-05",
|
|
1732
|
+
"modalities": {
|
|
1733
|
+
"input": [
|
|
1734
|
+
"text",
|
|
1735
|
+
"image"
|
|
1736
|
+
],
|
|
1737
|
+
"output": [
|
|
1738
|
+
"text"
|
|
1739
|
+
]
|
|
1740
|
+
},
|
|
1741
|
+
"open_weights": true,
|
|
1742
|
+
"limit": {
|
|
1743
|
+
"context": 327680,
|
|
1744
|
+
"output": 16384
|
|
1745
|
+
},
|
|
1746
|
+
"cost": {
|
|
1747
|
+
"input": 0.1,
|
|
1748
|
+
"output": 0.3
|
|
1749
|
+
}
|
|
1750
|
+
},
|
|
1751
|
+
"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8": {
|
|
1752
|
+
"id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
|
|
1753
|
+
"name": "Llama 4 Maverick 17B FP8",
|
|
1754
|
+
"description": "Open multimodal Llama model for strong reasoning and fast responses",
|
|
1755
|
+
"family": "llama",
|
|
1756
|
+
"attachment": true,
|
|
1757
|
+
"reasoning": false,
|
|
1758
|
+
"tool_call": false,
|
|
1759
|
+
"structured_output": true,
|
|
1760
|
+
"release_date": "2025-04-05",
|
|
1761
|
+
"last_updated": "2025-04-05",
|
|
1762
|
+
"modalities": {
|
|
1763
|
+
"input": [
|
|
1764
|
+
"text",
|
|
1765
|
+
"image"
|
|
1766
|
+
],
|
|
1767
|
+
"output": [
|
|
1768
|
+
"text"
|
|
1769
|
+
]
|
|
1770
|
+
},
|
|
1771
|
+
"open_weights": true,
|
|
1772
|
+
"limit": {
|
|
1773
|
+
"context": 1048576,
|
|
1774
|
+
"output": 16384
|
|
1775
|
+
},
|
|
1776
|
+
"cost": {
|
|
1777
|
+
"input": 0.2,
|
|
1778
|
+
"output": 0.8
|
|
1779
|
+
}
|
|
1780
|
+
},
|
|
1781
|
+
"meta-llama/Llama-3.3-70B-Instruct-Turbo": {
|
|
1782
|
+
"id": "meta-llama/Llama-3.3-70B-Instruct-Turbo",
|
|
1783
|
+
"name": "Llama 3.3 70B Turbo",
|
|
1784
|
+
"description": "Compact Llama instruction model for fast chat and local deployment",
|
|
1785
|
+
"family": "llama",
|
|
1480
1786
|
"attachment": false,
|
|
1787
|
+
"reasoning": false,
|
|
1788
|
+
"tool_call": true,
|
|
1789
|
+
"structured_output": true,
|
|
1790
|
+
"release_date": "2024-12-06",
|
|
1791
|
+
"last_updated": "2024-12-06",
|
|
1792
|
+
"modalities": {
|
|
1793
|
+
"input": [
|
|
1794
|
+
"text"
|
|
1795
|
+
],
|
|
1796
|
+
"output": [
|
|
1797
|
+
"text"
|
|
1798
|
+
]
|
|
1799
|
+
},
|
|
1800
|
+
"open_weights": true,
|
|
1801
|
+
"limit": {
|
|
1802
|
+
"context": 131072,
|
|
1803
|
+
"output": 16384
|
|
1804
|
+
},
|
|
1805
|
+
"cost": {
|
|
1806
|
+
"input": 0.1,
|
|
1807
|
+
"output": 0.32
|
|
1808
|
+
}
|
|
1809
|
+
},
|
|
1810
|
+
"XiaomiMiMo/MiMo-V2.5-Pro": {
|
|
1811
|
+
"id": "XiaomiMiMo/MiMo-V2.5-Pro",
|
|
1812
|
+
"name": "MiMo-V2.5-Pro",
|
|
1813
|
+
"description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
|
|
1814
|
+
"family": "mimo",
|
|
1815
|
+
"attachment": true,
|
|
1481
1816
|
"reasoning": true,
|
|
1482
1817
|
"reasoning_options": [
|
|
1483
1818
|
{
|
|
1484
|
-
"type": "
|
|
1485
|
-
"values": [
|
|
1486
|
-
"low",
|
|
1487
|
-
"medium",
|
|
1488
|
-
"high"
|
|
1489
|
-
]
|
|
1819
|
+
"type": "toggle"
|
|
1490
1820
|
}
|
|
1491
1821
|
],
|
|
1492
1822
|
"tool_call": true,
|
|
1823
|
+
"interleaved": {
|
|
1824
|
+
"field": "reasoning_content"
|
|
1825
|
+
},
|
|
1493
1826
|
"structured_output": true,
|
|
1494
1827
|
"temperature": true,
|
|
1495
|
-
"
|
|
1496
|
-
"
|
|
1828
|
+
"knowledge": "2024-12",
|
|
1829
|
+
"release_date": "2026-04-22",
|
|
1830
|
+
"last_updated": "2026-04-22",
|
|
1497
1831
|
"modalities": {
|
|
1498
1832
|
"input": [
|
|
1833
|
+
"text",
|
|
1834
|
+
"audio"
|
|
1835
|
+
],
|
|
1836
|
+
"output": [
|
|
1499
1837
|
"text"
|
|
1838
|
+
]
|
|
1839
|
+
},
|
|
1840
|
+
"open_weights": true,
|
|
1841
|
+
"limit": {
|
|
1842
|
+
"context": 1048576,
|
|
1843
|
+
"output": 16384
|
|
1844
|
+
},
|
|
1845
|
+
"cost": {
|
|
1846
|
+
"input": 1,
|
|
1847
|
+
"output": 3,
|
|
1848
|
+
"cache_read": 0.2
|
|
1849
|
+
}
|
|
1850
|
+
},
|
|
1851
|
+
"XiaomiMiMo/MiMo-V2.5": {
|
|
1852
|
+
"id": "XiaomiMiMo/MiMo-V2.5",
|
|
1853
|
+
"name": "MiMo-V2.5",
|
|
1854
|
+
"description": "Open MiMo model for multimodal coding agents and long-context automation",
|
|
1855
|
+
"family": "mimo",
|
|
1856
|
+
"attachment": true,
|
|
1857
|
+
"reasoning": true,
|
|
1858
|
+
"reasoning_options": [
|
|
1859
|
+
{
|
|
1860
|
+
"type": "toggle"
|
|
1861
|
+
}
|
|
1862
|
+
],
|
|
1863
|
+
"tool_call": true,
|
|
1864
|
+
"interleaved": {
|
|
1865
|
+
"field": "reasoning_content"
|
|
1866
|
+
},
|
|
1867
|
+
"structured_output": true,
|
|
1868
|
+
"temperature": true,
|
|
1869
|
+
"knowledge": "2024-12",
|
|
1870
|
+
"release_date": "2026-04-22",
|
|
1871
|
+
"last_updated": "2026-04-22",
|
|
1872
|
+
"modalities": {
|
|
1873
|
+
"input": [
|
|
1874
|
+
"text",
|
|
1875
|
+
"image",
|
|
1876
|
+
"audio",
|
|
1877
|
+
"video"
|
|
1500
1878
|
],
|
|
1501
1879
|
"output": [
|
|
1502
1880
|
"text"
|
|
@@ -1504,12 +1882,13 @@
|
|
|
1504
1882
|
},
|
|
1505
1883
|
"open_weights": true,
|
|
1506
1884
|
"limit": {
|
|
1507
|
-
"context":
|
|
1885
|
+
"context": 262144,
|
|
1508
1886
|
"output": 16384
|
|
1509
1887
|
},
|
|
1510
1888
|
"cost": {
|
|
1511
|
-
"input": 0.
|
|
1512
|
-
"output":
|
|
1889
|
+
"input": 0.4,
|
|
1890
|
+
"output": 2,
|
|
1891
|
+
"cache_read": 0.08
|
|
1513
1892
|
}
|
|
1514
1893
|
}
|
|
1515
1894
|
}
|