llm.rb 15.1.0 → 15.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +296 -11
  3. data/README.md +68 -43
  4. data/bin/llm.rb +12 -4
  5. data/data/alibaba.json +912 -823
  6. data/data/anthropic.json +234 -187
  7. data/data/bedrock.json +3611 -2217
  8. data/data/deepinfra.json +1164 -989
  9. data/data/deepseek.json +65 -81
  10. data/data/google.json +668 -666
  11. data/data/mistral.json +501 -460
  12. data/data/moonshot.json +43 -248
  13. data/data/openai.json +1008 -914
  14. data/data/openrouter.json +7798 -7661
  15. data/data/xai.json +194 -194
  16. data/data/zai.json +242 -149
  17. data/docs/deepdive/advanced/compaction.md +5 -5
  18. data/docs/deepdive/advanced/guard.md +2 -2
  19. data/docs/deepdive/features/builtin_tools.md +93 -22
  20. data/docs/deepdive/features/{repl.md → console.md} +28 -28
  21. data/docs/deepdive/features/database.md +3 -3
  22. data/docs/deepdive/fundamentals/agents.md +2 -2
  23. data/docs/deepdive/fundamentals/providers.md +91 -6
  24. data/docs/deepdive/fundamentals/skills.md +14 -6
  25. data/docs/deepdive/fundamentals/tools.md +63 -31
  26. data/docs/deepdive/reference/cost.md +2 -2
  27. data/docs/deepdive/reference/model_registry.md +2 -2
  28. data/docs/deepdive/reference/tracer.md +15 -13
  29. data/docs/deepdive.md +2 -2
  30. data/lib/llm/active_record/acts_as_agent.rb +9 -5
  31. data/lib/llm/agent.rb +34 -15
  32. data/lib/llm/{repl → console}/bar.rb +3 -3
  33. data/lib/llm/{repl → console}/buffer.rb +4 -4
  34. data/lib/llm/{repl → console}/color.rb +2 -2
  35. data/lib/llm/{repl → console}/command.rb +12 -12
  36. data/lib/llm/{repl → console}/commands/exit.rb +4 -4
  37. data/lib/llm/{repl → console}/commands/help.rb +1 -1
  38. data/lib/llm/{repl/commands/compact.rb → console/commands/keep.rb} +11 -9
  39. data/lib/llm/{repl → console}/commands/model.rb +2 -2
  40. data/lib/llm/{repl → console}/input/cache.rb +2 -2
  41. data/lib/llm/{repl → console}/input/char.rb +2 -2
  42. data/lib/llm/{repl → console}/input/row.rb +1 -1
  43. data/lib/llm/{repl → console}/input.rb +13 -6
  44. data/lib/llm/console/markdown/parser.rb +78 -0
  45. data/lib/llm/{repl → console}/markdown/table.rb +3 -3
  46. data/lib/llm/{repl → console}/markdown.rb +13 -30
  47. data/lib/llm/{repl → console}/node.rb +3 -3
  48. data/lib/llm/{repl → console}/status.rb +11 -11
  49. data/lib/llm/{repl → console}/stream.rb +9 -9
  50. data/lib/llm/{repl → console}/walker.rb +1 -1
  51. data/lib/llm/{repl → console}/window.rb +10 -10
  52. data/lib/llm/{repl.rb → console.rb} +38 -19
  53. data/lib/llm/context/deserializer.rb +2 -1
  54. data/lib/llm/context.rb +1 -0
  55. data/lib/llm/function/async/reactor.rb +20 -1
  56. data/lib/llm/function.rb +1 -1
  57. data/lib/llm/json_adapter.rb +40 -28
  58. data/lib/llm/message.rb +7 -0
  59. data/lib/llm/provider.rb +2 -2
  60. data/lib/llm/providers/alibaba.rb +1 -1
  61. data/lib/llm/providers/deepseek.rb +1 -1
  62. data/lib/llm/providers/openai.rb +1 -0
  63. data/lib/llm/schema/leaf.rb +34 -2
  64. data/lib/llm/schema.rb +4 -2
  65. data/lib/llm/sequel/agent.rb +9 -5
  66. data/lib/llm/tool/param.rb +5 -1
  67. data/lib/llm/tool.rb +5 -0
  68. data/lib/llm/tools/bundle.rb +53 -0
  69. data/lib/llm/tools/edit-file.rb +7 -2
  70. data/lib/llm/tools/exec.rb +78 -0
  71. data/lib/llm/tools/git.rb +27 -26
  72. data/lib/llm/tools/mkdir.rb +12 -19
  73. data/lib/llm/tools/read_file.rb +69 -9
  74. data/lib/llm/tools/rg.rb +20 -24
  75. data/lib/llm/tools/ruby.rb +17 -25
  76. data/lib/llm/tools/utils.rb +74 -1
  77. data/lib/llm/tools/write_file.rb +4 -1
  78. data/lib/llm/tracer/logger.rb +2 -2
  79. data/lib/llm/tracer/pretty_logger.rb +4 -4
  80. data/lib/llm/tracer/telemetry.rb +2 -2
  81. data/lib/llm/tracer.rb +33 -0
  82. data/lib/llm/transport/utils.rb +1 -1
  83. data/lib/llm/version.rb +1 -1
  84. data/lib/llm.rb +4 -13
  85. data/llm.gemspec +7 -8
  86. metadata +64 -37
  87. data/lib/llm/tools/shell.rb +0 -55
data/data/deepinfra.json CHANGED
@@ -7,137 +7,165 @@
7
7
  "name": "Deep Infra",
8
8
  "doc": "https://deepinfra.com/models",
9
9
  "models": {
10
- "tencent/Hy3": {
11
- "id": "tencent/Hy3",
12
- "name": "Hy3",
13
- "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
14
- "family": "Hy",
15
- "attachment": false,
10
+ "ByteDance/Seed-2.0-mini": {
11
+ "id": "ByteDance/Seed-2.0-mini",
12
+ "name": "Seed 2.0 Mini",
13
+ "description": "Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks",
14
+ "family": "seed",
15
+ "attachment": true,
16
16
  "reasoning": true,
17
17
  "reasoning_options": [],
18
18
  "tool_call": true,
19
19
  "structured_output": true,
20
20
  "temperature": true,
21
- "release_date": "2026-07-06",
22
- "last_updated": "2026-07-06",
21
+ "release_date": "2026-02-14",
22
+ "last_updated": "2026-02-14",
23
23
  "modalities": {
24
24
  "input": [
25
- "text"
25
+ "text",
26
+ "image"
26
27
  ],
27
28
  "output": [
28
29
  "text"
29
30
  ]
30
31
  },
31
- "open_weights": true,
32
+ "open_weights": false,
32
33
  "limit": {
33
- "context": 262144,
34
- "output": 64000
34
+ "context": 256000,
35
+ "output": 32000
35
36
  },
36
37
  "cost": {
37
- "input": 0.14,
38
- "output": 0.58,
39
- "cache_read": 0.035
38
+ "input": 0.1,
39
+ "output": 0.4,
40
+ "cache_read": 0.02,
41
+ "tiers": [
42
+ {
43
+ "input": 0.2,
44
+ "output": 0.8,
45
+ "cache_read": 0.2,
46
+ "tier": {
47
+ "type": "context",
48
+ "size": 128000
49
+ }
50
+ }
51
+ ]
40
52
  }
41
53
  },
42
- "XiaomiMiMo/MiMo-V2.5": {
43
- "id": "XiaomiMiMo/MiMo-V2.5",
44
- "name": "MiMo-V2.5",
45
- "description": "Open MiMo model for multimodal coding agents and long-context automation",
46
- "family": "mimo",
54
+ "ByteDance/Seed-2.0-code": {
55
+ "id": "ByteDance/Seed-2.0-code",
56
+ "name": "Seed 2.0 Code",
57
+ "description": "ByteDance Seed coding model for multimodal software engineering and long-running agents",
58
+ "family": "seed",
47
59
  "attachment": true,
48
60
  "reasoning": true,
49
61
  "reasoning_options": [
50
62
  {
51
- "type": "toggle"
63
+ "type": "effort",
64
+ "values": [
65
+ "none",
66
+ "low",
67
+ "medium",
68
+ "high"
69
+ ]
52
70
  }
53
71
  ],
54
72
  "tool_call": true,
55
- "interleaved": {
56
- "field": "reasoning_content"
57
- },
58
73
  "structured_output": true,
59
74
  "temperature": true,
60
- "knowledge": "2024-12",
61
- "release_date": "2026-04-22",
62
- "last_updated": "2026-04-22",
75
+ "release_date": "2026-02-14",
76
+ "last_updated": "2026-02-14",
63
77
  "modalities": {
64
78
  "input": [
65
79
  "text",
66
- "image",
67
- "audio",
68
- "video"
80
+ "image"
69
81
  ],
70
82
  "output": [
71
83
  "text"
72
84
  ]
73
85
  },
74
- "open_weights": true,
86
+ "open_weights": false,
75
87
  "limit": {
76
- "context": 262144,
77
- "output": 16384
88
+ "context": 256000,
89
+ "output": 131072
78
90
  },
79
91
  "cost": {
80
- "input": 0.4,
81
- "output": 2,
82
- "cache_read": 0.08
92
+ "input": 0.5,
93
+ "output": 3,
94
+ "cache_read": 0.1,
95
+ "tiers": [
96
+ {
97
+ "input": 1,
98
+ "output": 6,
99
+ "cache_read": 0.2,
100
+ "tier": {
101
+ "type": "context",
102
+ "size": 128000
103
+ }
104
+ }
105
+ ]
83
106
  }
84
107
  },
85
- "XiaomiMiMo/MiMo-V2.5-Pro": {
86
- "id": "XiaomiMiMo/MiMo-V2.5-Pro",
87
- "name": "MiMo-V2.5-Pro",
88
- "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
89
- "family": "mimo",
108
+ "ByteDance/Seed-2.0-pro": {
109
+ "id": "ByteDance/Seed-2.0-pro",
110
+ "name": "Seed 2.0 Pro",
111
+ "description": "Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows",
112
+ "family": "seed",
90
113
  "attachment": true,
91
114
  "reasoning": true,
92
- "reasoning_options": [
93
- {
94
- "type": "toggle"
95
- }
96
- ],
115
+ "reasoning_options": [],
97
116
  "tool_call": true,
98
- "interleaved": {
99
- "field": "reasoning_content"
100
- },
101
117
  "structured_output": true,
102
118
  "temperature": true,
103
- "knowledge": "2024-12",
104
- "release_date": "2026-04-22",
105
- "last_updated": "2026-04-22",
119
+ "release_date": "2026-02-14",
120
+ "last_updated": "2026-02-14",
106
121
  "modalities": {
107
122
  "input": [
108
123
  "text",
109
- "audio"
124
+ "image"
110
125
  ],
111
126
  "output": [
112
127
  "text"
113
128
  ]
114
129
  },
115
- "open_weights": true,
130
+ "open_weights": false,
116
131
  "limit": {
117
- "context": 1048576,
118
- "output": 16384
132
+ "context": 256000,
133
+ "output": 128000
119
134
  },
120
135
  "cost": {
121
- "input": 1,
136
+ "input": 0.5,
122
137
  "output": 3,
123
- "cache_read": 0.2
138
+ "cache_read": 0.1,
139
+ "tiers": [
140
+ {
141
+ "input": 1,
142
+ "output": 6,
143
+ "cache_read": 0.2,
144
+ "tier": {
145
+ "type": "context",
146
+ "size": 128000
147
+ }
148
+ }
149
+ ]
124
150
  }
125
151
  },
126
- "MiniMaxAI/MiniMax-M2.7": {
127
- "id": "MiniMaxAI/MiniMax-M2.7",
128
- "name": "MiniMax-M2.7",
129
- "description": "Open MiniMax flagship for coding agents, office automation, and complex environments",
130
- "family": "minimax",
131
- "attachment": false,
152
+ "stepfun-ai/Step-3.7-Flash": {
153
+ "id": "stepfun-ai/Step-3.7-Flash",
154
+ "name": "Step 3.7 Flash",
155
+ "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts",
156
+ "attachment": true,
132
157
  "reasoning": true,
133
158
  "reasoning_options": [],
134
159
  "tool_call": true,
135
160
  "temperature": true,
136
- "release_date": "2026-03-18",
137
- "last_updated": "2026-03-18",
161
+ "knowledge": "2026-03-01",
162
+ "release_date": "2026-05-29",
163
+ "last_updated": "2026-05-29",
138
164
  "modalities": {
139
165
  "input": [
140
- "text"
166
+ "text",
167
+ "image",
168
+ "video"
141
169
  ],
142
170
  "output": [
143
171
  "text"
@@ -145,33 +173,30 @@
145
173
  },
146
174
  "open_weights": true,
147
175
  "limit": {
148
- "context": 196608,
149
- "output": 131072
176
+ "context": 262144,
177
+ "output": 256000
150
178
  },
151
179
  "cost": {
152
- "input": 0.25,
153
- "output": 1,
154
- "cache_read": 0.05
180
+ "input": 0.2,
181
+ "output": 1.15,
182
+ "cache_read": 0.04
155
183
  }
156
184
  },
157
- "MiniMaxAI/MiniMax-M3": {
158
- "id": "MiniMaxAI/MiniMax-M3",
159
- "name": "MiniMax-M3",
160
- "description": "MiniMax multimodal model for long-context coding, perception, and agent planning",
161
- "family": "minimax",
162
- "attachment": true,
163
- "reasoning": true,
164
- "reasoning_options": [],
185
+ "deepseek-ai/DeepSeek-V3": {
186
+ "id": "deepseek-ai/DeepSeek-V3",
187
+ "name": "DeepSeek-V3",
188
+ "description": "Open DeepSeek MoE chat model for coding, math, and general reasoning",
189
+ "family": "deepseek",
190
+ "attachment": false,
191
+ "reasoning": false,
165
192
  "tool_call": true,
166
193
  "structured_output": true,
167
194
  "temperature": true,
168
- "release_date": "2026-06-01",
169
- "last_updated": "2026-06-01",
195
+ "release_date": "2024-12-26",
196
+ "last_updated": "2024-12-26",
170
197
  "modalities": {
171
198
  "input": [
172
- "text",
173
- "image",
174
- "video"
199
+ "text"
175
200
  ],
176
201
  "output": [
177
202
  "text"
@@ -179,31 +204,44 @@
179
204
  },
180
205
  "open_weights": true,
181
206
  "limit": {
182
- "context": 524288,
183
- "output": 128000
207
+ "context": 163840,
208
+ "output": 8192
184
209
  },
185
210
  "cost": {
186
- "input": 0.28,
187
- "output": 1.1,
188
- "cache_read": 0.056
211
+ "input": 0.32,
212
+ "output": 0.89
189
213
  }
190
214
  },
191
- "MiniMaxAI/MiniMax-M2.5": {
192
- "id": "MiniMaxAI/MiniMax-M2.5",
193
- "name": "MiniMax M2.5",
194
- "description": "MiniMax model for chat, coding, office work, and agentic tasks",
195
- "family": "minimax",
215
+ "deepseek-ai/DeepSeek-V4-Flash": {
216
+ "id": "deepseek-ai/DeepSeek-V4-Flash",
217
+ "name": "DeepSeek V4 Flash",
218
+ "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
219
+ "family": "deepseek-flash",
196
220
  "attachment": false,
197
221
  "reasoning": true,
198
- "reasoning_options": [],
222
+ "reasoning_options": [
223
+ {
224
+ "type": "toggle"
225
+ },
226
+ {
227
+ "type": "effort",
228
+ "values": [
229
+ "low",
230
+ "medium",
231
+ "high",
232
+ "xhigh"
233
+ ]
234
+ }
235
+ ],
199
236
  "tool_call": true,
200
237
  "interleaved": {
201
238
  "field": "reasoning_content"
202
239
  },
240
+ "structured_output": true,
203
241
  "temperature": true,
204
- "knowledge": "2025-06",
205
- "release_date": "2026-02-12",
206
- "last_updated": "2026-02-12",
242
+ "knowledge": "2025-05",
243
+ "release_date": "2026-04-24",
244
+ "last_updated": "2026-04-24",
207
245
  "modalities": {
208
246
  "input": [
209
247
  "text"
@@ -214,29 +252,107 @@
214
252
  },
215
253
  "open_weights": true,
216
254
  "limit": {
217
- "context": 196608,
218
- "output": 131072
255
+ "context": 1048576,
256
+ "output": 16384
219
257
  },
220
- "status": "deprecated",
221
258
  "cost": {
222
- "input": 0.15,
223
- "output": 1.15,
224
- "cache_read": 0.03
259
+ "input": 0.09,
260
+ "output": 0.18,
261
+ "cache_read": 0.018
225
262
  }
226
263
  },
227
- "nvidia/Llama-3.3-Nemotron-Super-49B-v1.5": {
228
- "id": "nvidia/Llama-3.3-Nemotron-Super-49B-v1.5",
229
- "name": "Llama 3.3 Nemotron Super 49B v1.5",
230
- "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
231
- "family": "nemotron",
264
+ "deepseek-ai/DeepSeek-V3-0324": {
265
+ "id": "deepseek-ai/DeepSeek-V3-0324",
266
+ "name": "DeepSeek V3 0324",
267
+ "description": "March 2025 checkpoint of DeepSeek-V3 with improved reasoning and coding",
268
+ "family": "deepseek",
232
269
  "attachment": false,
270
+ "reasoning": false,
271
+ "tool_call": true,
272
+ "structured_output": true,
273
+ "temperature": true,
274
+ "release_date": "2025-03-24",
275
+ "last_updated": "2025-03-24",
276
+ "modalities": {
277
+ "input": [
278
+ "text"
279
+ ],
280
+ "output": [
281
+ "text"
282
+ ]
283
+ },
284
+ "open_weights": true,
285
+ "limit": {
286
+ "context": 163840,
287
+ "output": 163840
288
+ },
289
+ "cost": {
290
+ "input": 0.24,
291
+ "output": 0.9,
292
+ "cache_read": 0.135
293
+ }
294
+ },
295
+ "deepseek-ai/DeepSeek-V4-Flash-Vision-Exp": {
296
+ "id": "deepseek-ai/DeepSeek-V4-Flash-Vision-Exp",
297
+ "name": "DeepSeek V4 Flash Vision Exp",
298
+ "description": "Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work",
299
+ "family": "deepseek-flash",
300
+ "attachment": true,
233
301
  "reasoning": true,
234
- "reasoning_options": [],
302
+ "reasoning_options": [
303
+ {
304
+ "type": "effort",
305
+ "values": [
306
+ "none",
307
+ "low",
308
+ "high",
309
+ "max"
310
+ ]
311
+ }
312
+ ],
313
+ "tool_call": true,
314
+ "structured_output": true,
315
+ "temperature": true,
316
+ "release_date": "2026-08-21",
317
+ "last_updated": "2026-08-21",
318
+ "modalities": {
319
+ "input": [
320
+ "text",
321
+ "image"
322
+ ],
323
+ "output": [
324
+ "text"
325
+ ]
326
+ },
327
+ "open_weights": false,
328
+ "limit": {
329
+ "context": 1048576,
330
+ "output": 384000
331
+ },
332
+ "cost": {
333
+ "input": 0.44,
334
+ "output": 1.32,
335
+ "cache_read": 0.014
336
+ }
337
+ },
338
+ "deepseek-ai/DeepSeek-V4-Flash-0731": {
339
+ "id": "deepseek-ai/DeepSeek-V4-Flash-0731",
340
+ "name": "DeepSeek V4 Flash 0731",
341
+ "description": "Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding",
342
+ "family": "deepseek-flash",
343
+ "attachment": false,
344
+ "reasoning": true,
345
+ "reasoning_options": [
346
+ {
347
+ "type": "toggle"
348
+ }
349
+ ],
235
350
  "tool_call": true,
236
351
  "structured_output": true,
237
352
  "temperature": true,
238
- "release_date": "2025-07-25",
239
- "last_updated": "2025-07-25",
353
+ "knowledge": "2025-05",
354
+ "release_date": "2026-07-31",
355
+ "last_updated": "2026-07-31",
240
356
  "modalities": {
241
357
  "input": [
242
358
  "text"
@@ -247,20 +363,20 @@
247
363
  },
248
364
  "open_weights": true,
249
365
  "limit": {
250
- "context": 131072,
251
- "output": 131072
366
+ "context": 1048576,
367
+ "output": 384000
252
368
  },
253
- "status": "deprecated",
254
369
  "cost": {
255
- "input": 0.4,
256
- "output": 0.4
370
+ "input": 0.06,
371
+ "output": 0.18,
372
+ "cache_read": 0.015
257
373
  }
258
374
  },
259
- "nvidia/Nemotron-3-Nano-30B-A3B": {
260
- "id": "nvidia/Nemotron-3-Nano-30B-A3B",
261
- "name": "Nemotron 3 Nano 30B A3B",
262
- "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents",
263
- "family": "nemotron",
375
+ "deepseek-ai/DeepSeek-V3.1": {
376
+ "id": "deepseek-ai/DeepSeek-V3.1",
377
+ "name": "DeepSeek-V3.1",
378
+ "description": "Hybrid-reasoning DeepSeek model with thinking and non-thinking modes",
379
+ "family": "deepseek",
264
380
  "attachment": false,
265
381
  "reasoning": true,
266
382
  "reasoning_options": [
@@ -269,9 +385,10 @@
269
385
  }
270
386
  ],
271
387
  "tool_call": true,
388
+ "structured_output": true,
272
389
  "temperature": true,
273
- "release_date": "2025-12-15",
274
- "last_updated": "2025-12-15",
390
+ "release_date": "2025-08-21",
391
+ "last_updated": "2025-08-21",
275
392
  "modalities": {
276
393
  "input": [
277
394
  "text"
@@ -282,67 +399,75 @@
282
399
  },
283
400
  "open_weights": true,
284
401
  "limit": {
285
- "context": 262144,
286
- "output": 262144
402
+ "context": 163840,
403
+ "output": 8192
287
404
  },
288
405
  "cost": {
289
- "input": 0.05,
290
- "output": 0.2,
291
- "cache_read": 0.025
406
+ "input": 0.25,
407
+ "output": 0.95,
408
+ "cache_read": 0.13
292
409
  }
293
410
  },
294
- "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning": {
295
- "id": "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning",
296
- "name": "Nemotron 3 Nano Omni 30B A3B Reasoning",
297
- "description": "Open Nemotron omni model combining reasoning with text, vision, and audio",
298
- "family": "nemotron",
299
- "attachment": true,
411
+ "deepseek-ai/DeepSeek-R1-0528": {
412
+ "id": "deepseek-ai/DeepSeek-R1-0528",
413
+ "name": "DeepSeek-R1-0528",
414
+ "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
415
+ "attachment": false,
300
416
  "reasoning": true,
301
417
  "reasoning_options": [],
302
418
  "tool_call": true,
419
+ "interleaved": {
420
+ "field": "reasoning_content"
421
+ },
303
422
  "structured_output": true,
304
423
  "temperature": true,
305
- "release_date": "2026-04-28",
306
- "last_updated": "2026-04-28",
424
+ "knowledge": "2024-07",
425
+ "release_date": "2025-05-28",
426
+ "last_updated": "2025-05-28",
307
427
  "modalities": {
308
428
  "input": [
309
- "text",
310
- "image",
311
- "video",
312
- "audio"
429
+ "text"
313
430
  ],
314
431
  "output": [
315
432
  "text"
316
433
  ]
317
434
  },
318
- "open_weights": true,
435
+ "open_weights": false,
319
436
  "limit": {
320
- "context": 262144,
321
- "output": 65536
437
+ "context": 163840,
438
+ "output": 64000
322
439
  },
323
- "status": "deprecated",
324
440
  "cost": {
325
- "input": 0.2,
326
- "output": 0.8
441
+ "input": 0.5,
442
+ "output": 2.15,
443
+ "cache_read": 0.35
327
444
  }
328
445
  },
329
- "google/gemma-4-26B-A4B-it": {
330
- "id": "google/gemma-4-26B-A4B-it",
331
- "name": "Gemma 4 26B A4B IT",
332
- "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
333
- "family": "gemma",
446
+ "deepseek-ai/DeepSeek-V4.1-Flash": {
447
+ "id": "deepseek-ai/DeepSeek-V4.1-Flash",
448
+ "name": "DeepSeek V4.1 Flash",
449
+ "description": "DeepSeek V4.1 Flash model for reasoning and agentic coding",
450
+ "family": "deepseek-flash",
334
451
  "attachment": true,
335
452
  "reasoning": true,
336
453
  "reasoning_options": [
337
454
  {
338
- "type": "toggle"
455
+ "type": "effort",
456
+ "values": [
457
+ "none",
458
+ "low",
459
+ "high",
460
+ "xhigh",
461
+ "max"
462
+ ]
339
463
  }
340
464
  ],
341
465
  "tool_call": true,
342
466
  "structured_output": true,
343
467
  "temperature": true,
344
- "release_date": "2026-04-02",
345
- "last_updated": "2026-04-02",
468
+ "knowledge": "2025-05",
469
+ "release_date": "2026-09-10",
470
+ "last_updated": "2026-09-10",
346
471
  "modalities": {
347
472
  "input": [
348
473
  "text",
@@ -354,36 +479,42 @@
354
479
  },
355
480
  "open_weights": true,
356
481
  "limit": {
357
- "context": 262144,
358
- "output": 32768
482
+ "context": 1048576,
483
+ "output": 384000
359
484
  },
360
485
  "cost": {
361
- "input": 0.07,
362
- "output": 0.34
486
+ "input": 0.3,
487
+ "output": 1.2,
488
+ "cache_read": 0.006
363
489
  }
364
490
  },
365
- "google/gemma-4-E4B-it": {
366
- "id": "google/gemma-4-E4B-it",
367
- "name": "Gemma 4 E4B IT",
368
- "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
369
- "family": "gemma",
370
- "attachment": true,
491
+ "deepseek-ai/DeepSeek-V4-Pro-0813": {
492
+ "id": "deepseek-ai/DeepSeek-V4-Pro-0813",
493
+ "name": "DeepSeek V4 Pro 0813",
494
+ "description": "DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes",
495
+ "family": "deepseek-thinking",
496
+ "attachment": false,
371
497
  "reasoning": true,
372
498
  "reasoning_options": [
373
499
  {
374
500
  "type": "toggle"
501
+ },
502
+ {
503
+ "type": "effort",
504
+ "values": [
505
+ "high",
506
+ "max"
507
+ ]
375
508
  }
376
509
  ],
377
510
  "tool_call": true,
378
511
  "structured_output": true,
379
512
  "temperature": true,
380
- "release_date": "2026-04-02",
381
- "last_updated": "2026-04-02",
513
+ "release_date": "2026-08-12",
514
+ "last_updated": "2026-08-22",
382
515
  "modalities": {
383
516
  "input": [
384
- "text",
385
- "image",
386
- "audio"
517
+ "text"
387
518
  ],
388
519
  "output": [
389
520
  "text"
@@ -391,20 +522,20 @@
391
522
  },
392
523
  "open_weights": true,
393
524
  "limit": {
394
- "context": 131072,
395
- "output": 8192
525
+ "context": 1048576,
526
+ "output": 384000
396
527
  },
397
528
  "cost": {
398
- "input": 0.02,
399
- "output": 0.1
529
+ "input": 1.3,
530
+ "output": 2.6,
531
+ "cache_read": 0.1
400
532
  }
401
533
  },
402
- "google/gemma-4-31B-it": {
403
- "id": "google/gemma-4-31B-it",
404
- "name": "Gemma 4 31B IT",
405
- "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
406
- "family": "gemma",
407
- "attachment": true,
534
+ "deepseek-ai/DeepSeek-V3.2": {
535
+ "id": "deepseek-ai/DeepSeek-V3.2",
536
+ "name": "DeepSeek-V3.2",
537
+ "description": "DeepSeek chat model for instruction following, coding, and analysis",
538
+ "attachment": false,
408
539
  "reasoning": true,
409
540
  "reasoning_options": [
410
541
  {
@@ -412,177 +543,189 @@
412
543
  }
413
544
  ],
414
545
  "tool_call": true,
546
+ "interleaved": {
547
+ "field": "reasoning_content"
548
+ },
415
549
  "structured_output": true,
416
550
  "temperature": true,
417
- "release_date": "2026-04-02",
418
- "last_updated": "2026-04-02",
551
+ "knowledge": "2024-12",
552
+ "release_date": "2025-12-02",
553
+ "last_updated": "2025-12-02",
419
554
  "modalities": {
420
555
  "input": [
421
- "text",
422
- "image",
423
- "video"
556
+ "text"
424
557
  ],
425
558
  "output": [
426
559
  "text"
427
560
  ]
428
561
  },
429
- "open_weights": true,
562
+ "open_weights": false,
430
563
  "limit": {
431
- "context": 262144,
432
- "output": 32768
564
+ "context": 163840,
565
+ "output": 64000
433
566
  },
434
567
  "cost": {
435
- "input": 0.13,
436
- "output": 0.38
568
+ "input": 0.26,
569
+ "output": 0.38,
570
+ "cache_read": 0.13
437
571
  }
438
572
  },
439
- "ByteDance/Seed-2.0-code": {
440
- "id": "ByteDance/Seed-2.0-code",
441
- "name": "Seed 2.0 Code",
442
- "description": "ByteDance Seed coding model for multimodal software engineering and long-running agents",
443
- "family": "seed",
444
- "attachment": true,
573
+ "deepseek-ai/DeepSeek-V4-Pro": {
574
+ "id": "deepseek-ai/DeepSeek-V4-Pro",
575
+ "name": "DeepSeek V4 Pro",
576
+ "description": "Open MoE flagship with million-token context for coding and long agent runs",
577
+ "family": "deepseek-thinking",
578
+ "attachment": false,
445
579
  "reasoning": true,
446
580
  "reasoning_options": [
581
+ {
582
+ "type": "toggle"
583
+ },
447
584
  {
448
585
  "type": "effort",
449
586
  "values": [
450
- "none",
451
587
  "low",
452
588
  "medium",
453
- "high"
589
+ "high",
590
+ "xhigh"
454
591
  ]
455
592
  }
456
593
  ],
457
594
  "tool_call": true,
595
+ "interleaved": {
596
+ "field": "reasoning_content"
597
+ },
598
+ "structured_output": true,
599
+ "temperature": true,
600
+ "knowledge": "2025-05",
601
+ "release_date": "2026-04-24",
602
+ "last_updated": "2026-04-24",
603
+ "modalities": {
604
+ "input": [
605
+ "text"
606
+ ],
607
+ "output": [
608
+ "text"
609
+ ]
610
+ },
611
+ "open_weights": true,
612
+ "limit": {
613
+ "context": 1048576,
614
+ "output": 16384
615
+ },
616
+ "cost": {
617
+ "input": 1.3,
618
+ "output": 2.6,
619
+ "cache_read": 0.1
620
+ }
621
+ },
622
+ "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning": {
623
+ "id": "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning",
624
+ "name": "Nemotron 3 Nano Omni 30B A3B Reasoning",
625
+ "description": "Open Nemotron omni model combining reasoning with text, vision, and audio",
626
+ "family": "nemotron",
627
+ "attachment": true,
628
+ "reasoning": true,
629
+ "reasoning_options": [],
630
+ "tool_call": true,
458
631
  "structured_output": true,
459
632
  "temperature": true,
460
- "release_date": "2026-02-14",
461
- "last_updated": "2026-02-14",
633
+ "release_date": "2026-04-28",
634
+ "last_updated": "2026-04-28",
462
635
  "modalities": {
463
636
  "input": [
464
637
  "text",
465
- "image"
638
+ "image",
639
+ "video",
640
+ "audio"
466
641
  ],
467
642
  "output": [
468
643
  "text"
469
644
  ]
470
645
  },
471
- "open_weights": false,
646
+ "open_weights": true,
472
647
  "limit": {
473
- "context": 256000,
474
- "output": 131072
648
+ "context": 262144,
649
+ "output": 65536
475
650
  },
651
+ "status": "deprecated",
476
652
  "cost": {
477
- "input": 0.5,
478
- "output": 3,
479
- "cache_read": 0.1,
480
- "tiers": [
481
- {
482
- "input": 1,
483
- "output": 6,
484
- "cache_read": 0.2,
485
- "tier": {
486
- "type": "context",
487
- "size": 128000
488
- }
489
- }
490
- ]
653
+ "input": 0.2,
654
+ "output": 0.8
491
655
  }
492
656
  },
493
- "ByteDance/Seed-2.0-pro": {
494
- "id": "ByteDance/Seed-2.0-pro",
495
- "name": "Seed 2.0 Pro",
496
- "description": "Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows",
497
- "family": "seed",
498
- "attachment": true,
657
+ "nvidia/Llama-3.3-Nemotron-Super-49B-v1.5": {
658
+ "id": "nvidia/Llama-3.3-Nemotron-Super-49B-v1.5",
659
+ "name": "Llama 3.3 Nemotron Super 49B v1.5",
660
+ "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
661
+ "family": "nemotron",
662
+ "attachment": false,
499
663
  "reasoning": true,
500
664
  "reasoning_options": [],
501
665
  "tool_call": true,
502
666
  "structured_output": true,
503
667
  "temperature": true,
504
- "release_date": "2026-02-14",
505
- "last_updated": "2026-02-14",
668
+ "release_date": "2025-07-25",
669
+ "last_updated": "2025-07-25",
506
670
  "modalities": {
507
671
  "input": [
508
- "text",
509
- "image"
672
+ "text"
510
673
  ],
511
674
  "output": [
512
675
  "text"
513
676
  ]
514
677
  },
515
- "open_weights": false,
678
+ "open_weights": true,
516
679
  "limit": {
517
- "context": 256000,
518
- "output": 128000
680
+ "context": 131072,
681
+ "output": 131072
519
682
  },
683
+ "status": "deprecated",
520
684
  "cost": {
521
- "input": 0.5,
522
- "output": 3,
523
- "cache_read": 0.1,
524
- "tiers": [
525
- {
526
- "input": 1,
527
- "output": 6,
528
- "cache_read": 0.2,
529
- "tier": {
530
- "type": "context",
531
- "size": 128000
532
- }
533
- }
534
- ]
685
+ "input": 0.4,
686
+ "output": 0.4
535
687
  }
536
688
  },
537
- "ByteDance/Seed-2.0-mini": {
538
- "id": "ByteDance/Seed-2.0-mini",
539
- "name": "Seed 2.0 Mini",
540
- "description": "Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks",
541
- "family": "seed",
542
- "attachment": true,
689
+ "nvidia/Nemotron-3-Nano-30B-A3B": {
690
+ "id": "nvidia/Nemotron-3-Nano-30B-A3B",
691
+ "name": "Nemotron 3 Nano 30B A3B",
692
+ "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents",
693
+ "family": "nemotron",
694
+ "attachment": false,
543
695
  "reasoning": true,
544
- "reasoning_options": [],
696
+ "reasoning_options": [
697
+ {
698
+ "type": "toggle"
699
+ }
700
+ ],
545
701
  "tool_call": true,
546
- "structured_output": true,
547
702
  "temperature": true,
548
- "release_date": "2026-02-14",
549
- "last_updated": "2026-02-14",
703
+ "release_date": "2025-12-15",
704
+ "last_updated": "2025-12-15",
550
705
  "modalities": {
551
706
  "input": [
552
- "text",
553
- "image"
707
+ "text"
554
708
  ],
555
709
  "output": [
556
710
  "text"
557
711
  ]
558
712
  },
559
- "open_weights": false,
713
+ "open_weights": true,
560
714
  "limit": {
561
- "context": 256000,
562
- "output": 32000
715
+ "context": 262144,
716
+ "output": 262144
563
717
  },
564
718
  "cost": {
565
- "input": 0.1,
566
- "output": 0.4,
567
- "cache_read": 0.02,
568
- "tiers": [
569
- {
570
- "input": 0.2,
571
- "output": 0.8,
572
- "cache_read": 0.2,
573
- "tier": {
574
- "type": "context",
575
- "size": 128000
576
- }
577
- }
578
- ]
719
+ "input": 0.05,
720
+ "output": 0.2,
721
+ "cache_read": 0.025
579
722
  }
580
723
  },
581
- "moonshotai/Kimi-K2.5": {
582
- "id": "moonshotai/Kimi-K2.5",
583
- "name": "Kimi K2.5",
584
- "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
585
- "family": "kimi-k2",
724
+ "google/gemma-4-31B-it": {
725
+ "id": "google/gemma-4-31B-it",
726
+ "name": "Gemma 4 31B IT",
727
+ "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
728
+ "family": "gemma",
586
729
  "attachment": true,
587
730
  "reasoning": true,
588
731
  "reasoning_options": [
@@ -591,14 +734,10 @@
591
734
  }
592
735
  ],
593
736
  "tool_call": true,
594
- "interleaved": {
595
- "field": "reasoning_content"
596
- },
597
737
  "structured_output": true,
598
738
  "temperature": true,
599
- "knowledge": "2025-01",
600
- "release_date": "2026-01-27",
601
- "last_updated": "2026-01-27",
739
+ "release_date": "2026-04-02",
740
+ "last_updated": "2026-04-02",
602
741
  "modalities": {
603
742
  "input": [
604
743
  "text",
@@ -615,16 +754,15 @@
615
754
  "output": 32768
616
755
  },
617
756
  "cost": {
618
- "input": 0.45,
619
- "output": 2.25,
620
- "cache_read": 0.07
757
+ "input": 0.13,
758
+ "output": 0.38
621
759
  }
622
760
  },
623
- "moonshotai/Kimi-K2.7-Code": {
624
- "id": "moonshotai/Kimi-K2.7-Code",
625
- "name": "Kimi K2.7 Code",
626
- "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
627
- "family": "kimi-k2",
761
+ "google/gemma-4-E4B-it": {
762
+ "id": "google/gemma-4-E4B-it",
763
+ "name": "Gemma 4 E4B IT",
764
+ "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
765
+ "family": "gemma",
628
766
  "attachment": true,
629
767
  "reasoning": true,
630
768
  "reasoning_options": [
@@ -633,19 +771,15 @@
633
771
  }
634
772
  ],
635
773
  "tool_call": true,
636
- "interleaved": {
637
- "field": "reasoning_content"
638
- },
639
774
  "structured_output": true,
640
- "temperature": false,
641
- "knowledge": "2025-01",
642
- "release_date": "2026-06-12",
643
- "last_updated": "2026-06-12",
775
+ "temperature": true,
776
+ "release_date": "2026-04-02",
777
+ "last_updated": "2026-04-02",
644
778
  "modalities": {
645
779
  "input": [
646
780
  "text",
647
781
  "image",
648
- "video"
782
+ "audio"
649
783
  ],
650
784
  "output": [
651
785
  "text"
@@ -653,37 +787,31 @@
653
787
  },
654
788
  "open_weights": true,
655
789
  "limit": {
656
- "context": 262144,
657
- "output": 262144
790
+ "context": 131072,
791
+ "output": 8192
658
792
  },
659
793
  "cost": {
660
- "input": 0.68,
661
- "output": 3.4,
662
- "cache_read": 0.136
794
+ "input": 0.02,
795
+ "output": 0.1
663
796
  }
664
797
  },
665
- "moonshotai/Kimi-K3": {
666
- "id": "moonshotai/Kimi-K3",
667
- "name": "Kimi K3",
668
- "description": "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work",
669
- "family": "kimi-k3",
798
+ "google/gemma-4-26B-A4B-it": {
799
+ "id": "google/gemma-4-26B-A4B-it",
800
+ "name": "Gemma 4 26B A4B IT",
801
+ "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
802
+ "family": "gemma",
670
803
  "attachment": true,
671
804
  "reasoning": true,
672
805
  "reasoning_options": [
673
806
  {
674
- "type": "effort",
675
- "values": [
676
- "low",
677
- "high",
678
- "max"
679
- ]
807
+ "type": "toggle"
680
808
  }
681
809
  ],
682
810
  "tool_call": true,
683
811
  "structured_output": true,
684
- "temperature": false,
685
- "release_date": "2026-07-16",
686
- "last_updated": "2026-07-16",
812
+ "temperature": true,
813
+ "release_date": "2026-04-02",
814
+ "last_updated": "2026-04-02",
687
815
  "modalities": {
688
816
  "input": [
689
817
  "text",
@@ -695,21 +823,20 @@
695
823
  },
696
824
  "open_weights": true,
697
825
  "limit": {
698
- "context": 1048576,
699
- "output": 131072
826
+ "context": 262144,
827
+ "output": 32768
700
828
  },
701
829
  "cost": {
702
- "input": 2.85,
703
- "output": 14.25,
704
- "cache_read": 0.285
830
+ "input": 0.07,
831
+ "output": 0.34
705
832
  }
706
833
  },
707
- "moonshotai/Kimi-K2.6": {
708
- "id": "moonshotai/Kimi-K2.6",
709
- "name": "Kimi K2.6",
710
- "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
711
- "family": "kimi-k2",
712
- "attachment": true,
834
+ "zai-org/GLM-5.1": {
835
+ "id": "zai-org/GLM-5.1",
836
+ "name": "GLM-5.1",
837
+ "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
838
+ "family": "glm",
839
+ "attachment": false,
713
840
  "reasoning": true,
714
841
  "reasoning_options": [
715
842
  {
@@ -722,14 +849,12 @@
722
849
  },
723
850
  "structured_output": true,
724
851
  "temperature": true,
725
- "knowledge": "2024-04",
726
- "release_date": "2026-04-21",
727
- "last_updated": "2026-04-21",
852
+ "knowledge": "2025-04",
853
+ "release_date": "2026-04-07",
854
+ "last_updated": "2026-04-07",
728
855
  "modalities": {
729
856
  "input": [
730
- "text",
731
- "image",
732
- "video"
857
+ "text"
733
858
  ],
734
859
  "output": [
735
860
  "text"
@@ -737,20 +862,20 @@
737
862
  },
738
863
  "open_weights": true,
739
864
  "limit": {
740
- "context": 262144,
865
+ "context": 202752,
741
866
  "output": 16384
742
867
  },
743
868
  "cost": {
744
- "input": 0.75,
869
+ "input": 1.05,
745
870
  "output": 3.5,
746
- "cache_read": 0.15
871
+ "cache_read": 0.205
747
872
  }
748
873
  },
749
- "openai/gpt-oss-120b": {
750
- "id": "openai/gpt-oss-120b",
751
- "name": "GPT OSS 120B",
752
- "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
753
- "family": "gpt-oss",
874
+ "zai-org/GLM-5.3": {
875
+ "id": "zai-org/GLM-5.3",
876
+ "name": "GLM-5.3",
877
+ "description": "Flagship GLM model for long-horizon coding, agents, and complex project delivery",
878
+ "family": "glm",
754
879
  "attachment": false,
755
880
  "reasoning": true,
756
881
  "reasoning_options": [
@@ -758,16 +883,16 @@
758
883
  "type": "effort",
759
884
  "values": [
760
885
  "low",
761
- "medium",
762
- "high"
886
+ "high",
887
+ "max"
763
888
  ]
764
889
  }
765
890
  ],
766
891
  "tool_call": true,
767
892
  "structured_output": true,
768
893
  "temperature": true,
769
- "release_date": "2025-08-05",
770
- "last_updated": "2025-08-05",
894
+ "release_date": "2026-08-14",
895
+ "last_updated": "2026-08-14",
771
896
  "modalities": {
772
897
  "input": [
773
898
  "text"
@@ -778,36 +903,44 @@
778
903
  },
779
904
  "open_weights": true,
780
905
  "limit": {
781
- "context": 131072,
782
- "output": 16384
906
+ "context": 1048576,
907
+ "output": 131072
783
908
  },
784
909
  "cost": {
785
- "input": 0.037,
786
- "output": 0.17
910
+ "input": 1.2,
911
+ "output": 4,
912
+ "cache_read": 0.12
787
913
  }
788
914
  },
789
- "openai/gpt-oss-20b": {
790
- "id": "openai/gpt-oss-20b",
791
- "name": "GPT OSS 20B",
792
- "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
793
- "family": "gpt-oss",
915
+ "zai-org/GLM-5.2": {
916
+ "id": "zai-org/GLM-5.2",
917
+ "name": "GLM-5.2",
918
+ "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
919
+ "family": "glm",
794
920
  "attachment": false,
795
921
  "reasoning": true,
796
922
  "reasoning_options": [
923
+ {
924
+ "type": "toggle"
925
+ },
797
926
  {
798
927
  "type": "effort",
799
928
  "values": [
800
929
  "low",
801
930
  "medium",
802
- "high"
931
+ "high",
932
+ "xhigh"
803
933
  ]
804
934
  }
805
935
  ],
806
936
  "tool_call": true,
937
+ "interleaved": {
938
+ "field": "reasoning_content"
939
+ },
807
940
  "structured_output": true,
808
941
  "temperature": true,
809
- "release_date": "2025-08-05",
810
- "last_updated": "2025-08-05",
942
+ "release_date": "2026-06-13",
943
+ "last_updated": "2026-06-13",
811
944
  "modalities": {
812
945
  "input": [
813
946
  "text"
@@ -818,44 +951,32 @@
818
951
  },
819
952
  "open_weights": true,
820
953
  "limit": {
821
- "context": 131072,
822
- "output": 16384
954
+ "context": 1048576,
955
+ "output": 32768
823
956
  },
824
957
  "cost": {
825
- "input": 0.03,
826
- "output": 0.14
958
+ "input": 0.75,
959
+ "output": 2.4,
960
+ "cache_read": 0.14
827
961
  }
828
962
  },
829
- "deepseek-ai/DeepSeek-V4-Pro": {
830
- "id": "deepseek-ai/DeepSeek-V4-Pro",
831
- "name": "DeepSeek V4 Pro",
832
- "description": "Open MoE flagship with million-token context for coding and long agent runs",
833
- "family": "deepseek-thinking",
963
+ "zai-org/GLM-4.7-Flash": {
964
+ "id": "zai-org/GLM-4.7-Flash",
965
+ "name": "GLM-4.7-Flash",
966
+ "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
967
+ "family": "glm-flash",
834
968
  "attachment": false,
835
969
  "reasoning": true,
836
- "reasoning_options": [
837
- {
838
- "type": "toggle"
839
- },
840
- {
841
- "type": "effort",
842
- "values": [
843
- "low",
844
- "medium",
845
- "high",
846
- "xhigh"
847
- ]
848
- }
849
- ],
970
+ "reasoning_options": [],
850
971
  "tool_call": true,
851
972
  "interleaved": {
852
973
  "field": "reasoning_content"
853
974
  },
854
975
  "structured_output": true,
855
976
  "temperature": true,
856
- "knowledge": "2025-05",
857
- "release_date": "2026-04-24",
858
- "last_updated": "2026-04-24",
977
+ "knowledge": "2025-04",
978
+ "release_date": "2026-01-19",
979
+ "last_updated": "2026-01-19",
859
980
  "modalities": {
860
981
  "input": [
861
982
  "text"
@@ -866,20 +987,20 @@
866
987
  },
867
988
  "open_weights": true,
868
989
  "limit": {
869
- "context": 1048576,
990
+ "context": 202752,
870
991
  "output": 16384
871
992
  },
872
993
  "cost": {
873
- "input": 1.3,
874
- "output": 2.6,
875
- "cache_read": 0.1
994
+ "input": 0.06,
995
+ "output": 0.4,
996
+ "cache_read": 0.01
876
997
  }
877
998
  },
878
- "deepseek-ai/DeepSeek-V4-Flash-0731": {
879
- "id": "deepseek-ai/DeepSeek-V4-Flash-0731",
880
- "name": "DeepSeek V4 Flash 0731",
881
- "description": "Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding",
882
- "family": "deepseek-flash",
999
+ "zai-org/GLM-4.7": {
1000
+ "id": "zai-org/GLM-4.7",
1001
+ "name": "GLM-4.7",
1002
+ "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
1003
+ "family": "glm",
883
1004
  "attachment": false,
884
1005
  "reasoning": true,
885
1006
  "reasoning_options": [
@@ -888,11 +1009,14 @@
888
1009
  }
889
1010
  ],
890
1011
  "tool_call": true,
1012
+ "interleaved": {
1013
+ "field": "reasoning_content"
1014
+ },
891
1015
  "structured_output": true,
892
1016
  "temperature": true,
893
- "knowledge": "2025-05",
894
- "release_date": "2026-07-31",
895
- "last_updated": "2026-07-31",
1017
+ "knowledge": "2025-04",
1018
+ "release_date": "2025-12-22",
1019
+ "last_updated": "2025-12-22",
896
1020
  "modalities": {
897
1021
  "input": [
898
1022
  "text"
@@ -903,19 +1027,20 @@
903
1027
  },
904
1028
  "open_weights": true,
905
1029
  "limit": {
906
- "context": 1048576,
907
- "output": 384000
1030
+ "context": 202752,
1031
+ "output": 16384
908
1032
  },
909
1033
  "cost": {
910
- "input": 0.08,
911
- "output": 0.18,
912
- "cache_read": 0.016
1034
+ "input": 0.4,
1035
+ "output": 1.75,
1036
+ "cache_read": 0.08
913
1037
  }
914
1038
  },
915
- "deepseek-ai/DeepSeek-V3.2": {
916
- "id": "deepseek-ai/DeepSeek-V3.2",
917
- "name": "DeepSeek-V3.2",
918
- "description": "DeepSeek chat model for instruction following, coding, and analysis",
1039
+ "zai-org/GLM-5": {
1040
+ "id": "zai-org/GLM-5",
1041
+ "name": "GLM-5",
1042
+ "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
1043
+ "family": "glm",
919
1044
  "attachment": false,
920
1045
  "reasoning": true,
921
1046
  "reasoning_options": [
@@ -929,9 +1054,9 @@
929
1054
  },
930
1055
  "structured_output": true,
931
1056
  "temperature": true,
932
- "knowledge": "2024-12",
933
- "release_date": "2025-12-02",
934
- "last_updated": "2025-12-02",
1057
+ "knowledge": "2025-12",
1058
+ "release_date": "2026-02-12",
1059
+ "last_updated": "2026-02-12",
935
1060
  "modalities": {
936
1061
  "input": [
937
1062
  "text"
@@ -940,33 +1065,38 @@
940
1065
  "text"
941
1066
  ]
942
1067
  },
943
- "open_weights": false,
1068
+ "open_weights": true,
944
1069
  "limit": {
945
- "context": 163840,
946
- "output": 64000
1070
+ "context": 202752,
1071
+ "output": 16384
947
1072
  },
948
1073
  "cost": {
949
- "input": 0.26,
950
- "output": 0.38,
951
- "cache_read": 0.13
1074
+ "input": 0.6,
1075
+ "output": 2.08,
1076
+ "cache_read": 0.12
952
1077
  }
953
1078
  },
954
- "deepseek-ai/DeepSeek-R1-0528": {
955
- "id": "deepseek-ai/DeepSeek-R1-0528",
956
- "name": "DeepSeek-R1-0528",
957
- "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
1079
+ "zai-org/GLM-4.6": {
1080
+ "id": "zai-org/GLM-4.6",
1081
+ "name": "GLM-4.6",
1082
+ "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
1083
+ "family": "glm",
958
1084
  "attachment": false,
959
1085
  "reasoning": true,
960
- "reasoning_options": [],
1086
+ "reasoning_options": [
1087
+ {
1088
+ "type": "toggle"
1089
+ }
1090
+ ],
961
1091
  "tool_call": true,
962
1092
  "interleaved": {
963
1093
  "field": "reasoning_content"
964
1094
  },
965
1095
  "structured_output": true,
966
1096
  "temperature": true,
967
- "knowledge": "2024-07",
968
- "release_date": "2025-05-28",
969
- "last_updated": "2025-05-28",
1097
+ "knowledge": "2025-04",
1098
+ "release_date": "2025-09-30",
1099
+ "last_updated": "2025-09-30",
970
1100
  "modalities": {
971
1101
  "input": [
972
1102
  "text"
@@ -975,37 +1105,45 @@
975
1105
  "text"
976
1106
  ]
977
1107
  },
978
- "open_weights": false,
1108
+ "open_weights": true,
979
1109
  "limit": {
980
- "context": 163840,
981
- "output": 64000
1110
+ "context": 202752,
1111
+ "output": 131072
982
1112
  },
983
1113
  "cost": {
984
1114
  "input": 0.5,
985
- "output": 2.15,
986
- "cache_read": 0.35
1115
+ "output": 2,
1116
+ "cache_read": 0.1
987
1117
  }
988
1118
  },
989
- "deepseek-ai/DeepSeek-V3.1": {
990
- "id": "deepseek-ai/DeepSeek-V3.1",
991
- "name": "DeepSeek-V3.1",
992
- "description": "Hybrid-reasoning DeepSeek model with thinking and non-thinking modes",
993
- "family": "deepseek",
994
- "attachment": false,
1119
+ "zai-org/GLM-5.3-Flash": {
1120
+ "id": "zai-org/GLM-5.3-Flash",
1121
+ "name": "GLM-5.3-Flash",
1122
+ "description": "Native multimodal GLM model for efficient coding and long-horizon agent tasks",
1123
+ "family": "glm",
1124
+ "attachment": true,
995
1125
  "reasoning": true,
996
1126
  "reasoning_options": [
997
1127
  {
998
- "type": "toggle"
1128
+ "type": "effort",
1129
+ "values": [
1130
+ "low",
1131
+ "high",
1132
+ "max"
1133
+ ]
999
1134
  }
1000
1135
  ],
1001
1136
  "tool_call": true,
1002
1137
  "structured_output": true,
1003
1138
  "temperature": true,
1004
- "release_date": "2025-08-21",
1005
- "last_updated": "2025-08-21",
1139
+ "release_date": "2026-08-26",
1140
+ "last_updated": "2026-08-26",
1006
1141
  "modalities": {
1007
1142
  "input": [
1008
- "text"
1143
+ "text",
1144
+ "image",
1145
+ "video",
1146
+ "pdf"
1009
1147
  ],
1010
1148
  "output": [
1011
1149
  "text"
@@ -1013,30 +1151,33 @@
1013
1151
  },
1014
1152
  "open_weights": true,
1015
1153
  "limit": {
1016
- "context": 163840,
1017
- "output": 8192
1154
+ "context": 1048576,
1155
+ "output": 131072
1018
1156
  },
1019
1157
  "cost": {
1020
- "input": 0.25,
1021
- "output": 0.95,
1022
- "cache_read": 0.13
1158
+ "input": 0.15,
1159
+ "output": 0.5,
1160
+ "cache_read": 0.03
1023
1161
  }
1024
1162
  },
1025
- "deepseek-ai/DeepSeek-V3-0324": {
1026
- "id": "deepseek-ai/DeepSeek-V3-0324",
1027
- "name": "DeepSeek V3 0324",
1028
- "description": "March 2025 checkpoint of DeepSeek-V3 with improved reasoning and coding",
1029
- "family": "deepseek",
1030
- "attachment": false,
1031
- "reasoning": false,
1163
+ "thinkingmachines/Inkling-Small": {
1164
+ "id": "thinkingmachines/Inkling-Small",
1165
+ "name": "Inkling Small",
1166
+ "description": "Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio",
1167
+ "family": "ling",
1168
+ "attachment": true,
1169
+ "reasoning": true,
1170
+ "reasoning_options": [],
1032
1171
  "tool_call": true,
1033
1172
  "structured_output": true,
1034
1173
  "temperature": true,
1035
- "release_date": "2025-03-24",
1036
- "last_updated": "2025-03-24",
1174
+ "release_date": "2026-07-30",
1175
+ "last_updated": "2026-07-30",
1037
1176
  "modalities": {
1038
1177
  "input": [
1039
- "text"
1178
+ "text",
1179
+ "image",
1180
+ "audio"
1040
1181
  ],
1041
1182
  "output": [
1042
1183
  "text"
@@ -1044,70 +1185,60 @@
1044
1185
  },
1045
1186
  "open_weights": true,
1046
1187
  "limit": {
1047
- "context": 163840,
1048
- "output": 163840
1188
+ "context": 524288,
1189
+ "output": 1048576
1049
1190
  },
1050
1191
  "cost": {
1051
- "input": 0.24,
1052
- "output": 0.9,
1053
- "cache_read": 0.135
1192
+ "input": 0.45,
1193
+ "output": 1.2,
1194
+ "cache_read": 0.1
1054
1195
  }
1055
1196
  },
1056
- "deepseek-ai/DeepSeek-V4-Pro-0813": {
1057
- "id": "deepseek-ai/DeepSeek-V4-Pro-0813",
1058
- "name": "DeepSeek V4 Pro 0813",
1059
- "description": "DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes",
1060
- "family": "deepseek-thinking",
1061
- "attachment": false,
1197
+ "thinkingmachines/Inkling": {
1198
+ "id": "thinkingmachines/Inkling",
1199
+ "name": "Inkling",
1200
+ "description": "Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio",
1201
+ "family": "ling",
1202
+ "attachment": true,
1062
1203
  "reasoning": true,
1063
- "reasoning_options": [
1064
- {
1065
- "type": "toggle"
1066
- },
1067
- {
1068
- "type": "effort",
1069
- "values": [
1070
- "high",
1071
- "max"
1072
- ]
1073
- }
1074
- ],
1204
+ "reasoning_options": [],
1075
1205
  "tool_call": true,
1076
- "structured_output": true,
1077
1206
  "temperature": true,
1078
- "release_date": "2026-08-12",
1079
- "last_updated": "2026-08-12",
1207
+ "release_date": "2026-07-15",
1208
+ "last_updated": "2026-07-15",
1080
1209
  "modalities": {
1081
1210
  "input": [
1082
- "text"
1211
+ "text",
1212
+ "image",
1213
+ "audio"
1083
1214
  ],
1084
1215
  "output": [
1085
1216
  "text"
1086
1217
  ]
1087
1218
  },
1088
- "open_weights": false,
1219
+ "open_weights": true,
1089
1220
  "limit": {
1090
- "context": 1048576,
1091
- "output": 384000
1221
+ "context": 524288,
1222
+ "output": 1048576
1092
1223
  },
1093
1224
  "cost": {
1094
- "input": 1.3,
1095
- "output": 2.6,
1096
- "cache_read": 0.1
1225
+ "input": 0.95,
1226
+ "output": 4.05,
1227
+ "cache_read": 0.16
1097
1228
  }
1098
1229
  },
1099
- "deepseek-ai/DeepSeek-V3": {
1100
- "id": "deepseek-ai/DeepSeek-V3",
1101
- "name": "DeepSeek-V3",
1102
- "description": "Open DeepSeek MoE chat model for coding, math, and general reasoning",
1103
- "family": "deepseek",
1230
+ "Qwen/Qwen3.7-Max": {
1231
+ "id": "Qwen/Qwen3.7-Max",
1232
+ "name": "Qwen3.7 Max",
1233
+ "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
1234
+ "family": "qwen",
1104
1235
  "attachment": false,
1105
1236
  "reasoning": false,
1106
1237
  "tool_call": true,
1107
1238
  "structured_output": true,
1108
1239
  "temperature": true,
1109
- "release_date": "2024-12-26",
1110
- "last_updated": "2024-12-26",
1240
+ "release_date": "2026-05-21",
1241
+ "last_updated": "2026-05-21",
1111
1242
  "modalities": {
1112
1243
  "input": [
1113
1244
  "text"
@@ -1116,22 +1247,43 @@
1116
1247
  "text"
1117
1248
  ]
1118
1249
  },
1119
- "open_weights": true,
1250
+ "open_weights": false,
1120
1251
  "limit": {
1121
- "context": 163840,
1122
- "output": 8192
1252
+ "context": 256000,
1253
+ "output": 65536
1123
1254
  },
1124
1255
  "cost": {
1125
- "input": 0.32,
1126
- "output": 0.89
1256
+ "input": 2.5,
1257
+ "output": 7.5,
1258
+ "cache_read": 0.5,
1259
+ "tiers": [
1260
+ {
1261
+ "input": 5,
1262
+ "output": 15,
1263
+ "cache_read": 1,
1264
+ "tier": {
1265
+ "type": "context",
1266
+ "size": 32000
1267
+ }
1268
+ },
1269
+ {
1270
+ "input": 6.25,
1271
+ "output": 18.5,
1272
+ "cache_read": 1.25,
1273
+ "tier": {
1274
+ "type": "context",
1275
+ "size": 128000
1276
+ }
1277
+ }
1278
+ ]
1127
1279
  }
1128
1280
  },
1129
- "deepseek-ai/DeepSeek-V4-Flash": {
1130
- "id": "deepseek-ai/DeepSeek-V4-Flash",
1131
- "name": "DeepSeek V4 Flash",
1132
- "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
1133
- "family": "deepseek-flash",
1134
- "attachment": false,
1281
+ "Qwen/Qwen3.8-27B": {
1282
+ "id": "Qwen/Qwen3.8-27B",
1283
+ "name": "Qwen3.8 27B",
1284
+ "description": "Dense 27B vision-language model for coding, agent tasks, and image and video understanding",
1285
+ "family": "qwen",
1286
+ "attachment": true,
1135
1287
  "reasoning": true,
1136
1288
  "reasoning_options": [
1137
1289
  {
@@ -1142,23 +1294,19 @@
1142
1294
  "values": [
1143
1295
  "low",
1144
1296
  "medium",
1145
- "high",
1146
1297
  "xhigh"
1147
1298
  ]
1148
1299
  }
1149
1300
  ],
1150
1301
  "tool_call": true,
1151
- "interleaved": {
1152
- "field": "reasoning_content"
1153
- },
1154
1302
  "structured_output": true,
1155
1303
  "temperature": true,
1156
- "knowledge": "2025-05",
1157
- "release_date": "2026-04-24",
1158
- "last_updated": "2026-04-24",
1304
+ "release_date": "2026-08-14",
1305
+ "last_updated": "2026-08-14",
1159
1306
  "modalities": {
1160
1307
  "input": [
1161
- "text"
1308
+ "text",
1309
+ "image"
1162
1310
  ],
1163
1311
  "output": [
1164
1312
  "text"
@@ -1166,32 +1314,34 @@
1166
1314
  },
1167
1315
  "open_weights": true,
1168
1316
  "limit": {
1169
- "context": 1048576,
1170
- "output": 16384
1317
+ "context": 262144,
1318
+ "output": 32768
1171
1319
  },
1172
1320
  "cost": {
1173
- "input": 0.09,
1174
- "output": 0.18,
1175
- "cache_read": 0.018
1321
+ "input": 0.4,
1322
+ "output": 3,
1323
+ "cache_read": 0.04
1176
1324
  }
1177
1325
  },
1178
- "stepfun-ai/Step-3.7-Flash": {
1179
- "id": "stepfun-ai/Step-3.7-Flash",
1180
- "name": "Step 3.7 Flash",
1181
- "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts",
1326
+ "Qwen/Qwen3.5-27B": {
1327
+ "id": "Qwen/Qwen3.5-27B",
1328
+ "name": "Qwen3.5 27B",
1329
+ "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
1330
+ "family": "qwen",
1182
1331
  "attachment": true,
1183
1332
  "reasoning": true,
1184
1333
  "reasoning_options": [],
1185
1334
  "tool_call": true,
1335
+ "structured_output": true,
1186
1336
  "temperature": true,
1187
- "knowledge": "2026-03-01",
1188
- "release_date": "2026-05-29",
1189
- "last_updated": "2026-05-29",
1337
+ "release_date": "2026-02-23",
1338
+ "last_updated": "2026-02-23",
1190
1339
  "modalities": {
1191
1340
  "input": [
1192
1341
  "text",
1193
1342
  "image",
1194
- "video"
1343
+ "video",
1344
+ "audio"
1195
1345
  ],
1196
1346
  "output": [
1197
1347
  "text"
@@ -1200,26 +1350,26 @@
1200
1350
  "open_weights": true,
1201
1351
  "limit": {
1202
1352
  "context": 262144,
1203
- "output": 256000
1353
+ "output": 65536
1204
1354
  },
1205
1355
  "cost": {
1206
- "input": 0.2,
1207
- "output": 1.15,
1208
- "cache_read": 0.04
1356
+ "input": 0.26,
1357
+ "output": 2.6
1209
1358
  }
1210
1359
  },
1211
- "Qwen/Qwen3-235B-A22B-Instruct-2507": {
1212
- "id": "Qwen/Qwen3-235B-A22B-Instruct-2507",
1213
- "name": "Qwen3 235B-A22B Instruct 2507",
1214
- "description": "Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use",
1360
+ "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo": {
1361
+ "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo",
1362
+ "name": "Qwen3 Coder 480B A35B Instruct Turbo",
1363
+ "description": "Qwen coding model for software agents, repository edits, and code reasoning",
1215
1364
  "family": "qwen",
1216
1365
  "attachment": false,
1217
1366
  "reasoning": false,
1218
1367
  "tool_call": true,
1219
1368
  "structured_output": true,
1220
1369
  "temperature": true,
1221
- "release_date": "2025-07-21",
1222
- "last_updated": "2025-07-21",
1370
+ "knowledge": "2025-04",
1371
+ "release_date": "2025-07-23",
1372
+ "last_updated": "2025-07-23",
1223
1373
  "modalities": {
1224
1374
  "input": [
1225
1375
  "text"
@@ -1231,78 +1381,46 @@
1231
1381
  "open_weights": true,
1232
1382
  "limit": {
1233
1383
  "context": 262144,
1234
- "output": 16384
1384
+ "output": 66536
1235
1385
  },
1236
1386
  "cost": {
1237
- "input": 0.09,
1238
- "output": 0.55
1387
+ "input": 0.3,
1388
+ "output": 1,
1389
+ "cache_read": 0.1
1239
1390
  }
1240
1391
  },
1241
- "Qwen/Qwen3-VL-235B-A22B-Instruct": {
1242
- "id": "Qwen/Qwen3-VL-235B-A22B-Instruct",
1243
- "name": "Qwen3 VL 235B A22B Instruct",
1244
- "description": "Qwen vision-language instruct model for visual reasoning, documents, and agent tasks",
1392
+ "Qwen/Qwen3.8-Max": {
1393
+ "id": "Qwen/Qwen3.8-Max",
1394
+ "name": "Qwen3.8 Max",
1395
+ "description": "2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows",
1245
1396
  "family": "qwen",
1246
1397
  "attachment": true,
1247
1398
  "reasoning": false,
1248
1399
  "tool_call": true,
1249
1400
  "structured_output": true,
1250
1401
  "temperature": true,
1251
- "knowledge": "2025-03-31",
1252
- "release_date": "2025-09-23",
1253
- "last_updated": "2025-09-23",
1254
- "modalities": {
1255
- "input": [
1256
- "text",
1257
- "image"
1258
- ],
1259
- "output": [
1260
- "text"
1261
- ]
1262
- },
1263
- "open_weights": true,
1264
- "limit": {
1265
- "context": 262144,
1266
- "output": 32768
1267
- },
1268
- "cost": {
1269
- "input": 0.2,
1270
- "output": 0.88,
1271
- "cache_read": 0.11
1272
- }
1273
- },
1274
- "Qwen/Qwen3.6-27B": {
1275
- "id": "Qwen/Qwen3.6-27B",
1276
- "name": "Qwen3.6 27B",
1277
- "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
1278
- "family": "qwen",
1279
- "attachment": true,
1280
- "reasoning": true,
1281
- "reasoning_options": [],
1282
- "tool_call": true,
1283
- "structured_output": true,
1284
- "temperature": true,
1285
- "release_date": "2026-04-22",
1286
- "last_updated": "2026-04-22",
1402
+ "release_date": "2026-08-03",
1403
+ "last_updated": "2026-08-03",
1287
1404
  "modalities": {
1288
1405
  "input": [
1289
1406
  "text",
1290
1407
  "image",
1291
1408
  "video",
1292
- "audio"
1409
+ "pdf"
1293
1410
  ],
1294
1411
  "output": [
1295
1412
  "text"
1296
1413
  ]
1297
1414
  },
1298
- "open_weights": true,
1415
+ "open_weights": false,
1299
1416
  "limit": {
1300
- "context": 262144,
1301
- "output": 65536
1417
+ "context": 256000,
1418
+ "output": 131072
1302
1419
  },
1303
1420
  "cost": {
1304
- "input": 0.32,
1305
- "output": 3.2
1421
+ "input": 1.65,
1422
+ "output": 4.951,
1423
+ "cache_read": 0.206
1306
1424
  }
1307
1425
  },
1308
1426
  "Qwen/Qwen3.5-9B": {
@@ -1338,19 +1456,19 @@
1338
1456
  "output": 0.15
1339
1457
  }
1340
1458
  },
1341
- "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo": {
1342
- "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo",
1343
- "name": "Qwen3 Coder 480B A35B Instruct Turbo",
1344
- "description": "Qwen coding model for software agents, repository edits, and code reasoning",
1459
+ "Qwen/Qwen3-30B-A3B": {
1460
+ "id": "Qwen/Qwen3-30B-A3B",
1461
+ "name": "Qwen3 30B A3B",
1462
+ "description": "Sparse MoE Qwen model with 3B active parameters for efficient chat and reasoning",
1345
1463
  "family": "qwen",
1346
1464
  "attachment": false,
1347
- "reasoning": false,
1465
+ "reasoning": true,
1466
+ "reasoning_options": [],
1348
1467
  "tool_call": true,
1349
1468
  "structured_output": true,
1350
1469
  "temperature": true,
1351
- "knowledge": "2025-04",
1352
- "release_date": "2025-07-23",
1353
- "last_updated": "2025-07-23",
1470
+ "release_date": "2025-04-28",
1471
+ "last_updated": "2025-04-28",
1354
1472
  "modalities": {
1355
1473
  "input": [
1356
1474
  "text"
@@ -1361,13 +1479,12 @@
1361
1479
  },
1362
1480
  "open_weights": true,
1363
1481
  "limit": {
1364
- "context": 262144,
1365
- "output": 66536
1482
+ "context": 40960,
1483
+ "output": 16384
1366
1484
  },
1367
1485
  "cost": {
1368
- "input": 0.3,
1369
- "output": 1,
1370
- "cache_read": 0.1
1486
+ "input": 0.12,
1487
+ "output": 0.5
1371
1488
  }
1372
1489
  },
1373
1490
  "Qwen/Qwen3.5-122B-A10B": {
@@ -1404,20 +1521,28 @@
1404
1521
  "output": 2.4
1405
1522
  }
1406
1523
  },
1407
- "Qwen/Qwen3-32B": {
1408
- "id": "Qwen/Qwen3-32B",
1409
- "name": "Qwen3 32B",
1410
- "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding",
1524
+ "Qwen/Qwen3.8-2.4T-A95B": {
1525
+ "id": "Qwen/Qwen3.8-2.4T-A95B",
1526
+ "name": "Qwen3.8 2.4T A95B",
1527
+ "description": "Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows",
1411
1528
  "family": "qwen",
1412
1529
  "attachment": false,
1413
1530
  "reasoning": true,
1414
- "reasoning_options": [],
1531
+ "reasoning_options": [
1532
+ {
1533
+ "type": "effort",
1534
+ "values": [
1535
+ "low",
1536
+ "medium",
1537
+ "xhigh"
1538
+ ]
1539
+ }
1540
+ ],
1415
1541
  "tool_call": true,
1416
1542
  "structured_output": true,
1417
1543
  "temperature": true,
1418
- "knowledge": "2025-04",
1419
- "release_date": "2025-04",
1420
- "last_updated": "2025-04",
1544
+ "release_date": "2026-08-12",
1545
+ "last_updated": "2026-08-12",
1421
1546
  "modalities": {
1422
1547
  "input": [
1423
1548
  "text"
@@ -1428,60 +1553,91 @@
1428
1553
  },
1429
1554
  "open_weights": true,
1430
1555
  "limit": {
1431
- "context": 40960,
1556
+ "context": 262144,
1557
+ "output": 131072
1558
+ },
1559
+ "cost": {
1560
+ "input": 2,
1561
+ "output": 6,
1562
+ "cache_read": 0.2
1563
+ }
1564
+ },
1565
+ "Qwen/Qwen3-235B-A22B-Instruct-2507": {
1566
+ "id": "Qwen/Qwen3-235B-A22B-Instruct-2507",
1567
+ "name": "Qwen3 235B-A22B Instruct 2507",
1568
+ "description": "Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use",
1569
+ "family": "qwen",
1570
+ "attachment": false,
1571
+ "reasoning": false,
1572
+ "tool_call": true,
1573
+ "structured_output": true,
1574
+ "temperature": true,
1575
+ "release_date": "2025-07-21",
1576
+ "last_updated": "2025-07-21",
1577
+ "modalities": {
1578
+ "input": [
1579
+ "text"
1580
+ ],
1581
+ "output": [
1582
+ "text"
1583
+ ]
1584
+ },
1585
+ "open_weights": true,
1586
+ "limit": {
1587
+ "context": 262144,
1432
1588
  "output": 16384
1433
1589
  },
1434
1590
  "cost": {
1435
- "input": 0.08,
1436
- "output": 0.28
1591
+ "input": 0.09,
1592
+ "output": 0.55
1437
1593
  }
1438
1594
  },
1439
- "Qwen/Qwen3.8-Max": {
1440
- "id": "Qwen/Qwen3.8-Max",
1441
- "name": "Qwen3.8 Max",
1442
- "description": "2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows",
1595
+ "Qwen/Qwen3-VL-235B-A22B-Instruct": {
1596
+ "id": "Qwen/Qwen3-VL-235B-A22B-Instruct",
1597
+ "name": "Qwen3 VL 235B A22B Instruct",
1598
+ "description": "Qwen vision-language instruct model for visual reasoning, documents, and agent tasks",
1443
1599
  "family": "qwen",
1444
1600
  "attachment": true,
1445
1601
  "reasoning": false,
1446
1602
  "tool_call": true,
1447
1603
  "structured_output": true,
1448
1604
  "temperature": true,
1449
- "release_date": "2026-08-03",
1450
- "last_updated": "2026-08-03",
1605
+ "knowledge": "2025-03-31",
1606
+ "release_date": "2025-09-23",
1607
+ "last_updated": "2025-09-23",
1451
1608
  "modalities": {
1452
1609
  "input": [
1453
1610
  "text",
1454
- "image",
1455
- "video",
1456
- "pdf"
1611
+ "image"
1457
1612
  ],
1458
1613
  "output": [
1459
1614
  "text"
1460
1615
  ]
1461
1616
  },
1462
- "open_weights": false,
1617
+ "open_weights": true,
1463
1618
  "limit": {
1464
- "context": 256000,
1465
- "output": 131072
1619
+ "context": 262144,
1620
+ "output": 32768
1466
1621
  },
1467
1622
  "cost": {
1468
- "input": 1.65,
1469
- "output": 4.951,
1470
- "cache_read": 0.206
1623
+ "input": 0.2,
1624
+ "output": 0.88,
1625
+ "cache_read": 0.11
1471
1626
  }
1472
1627
  },
1473
- "Qwen/Qwen3.7-Max": {
1474
- "id": "Qwen/Qwen3.7-Max",
1475
- "name": "Qwen3.7 Max",
1476
- "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
1628
+ "Qwen/Qwen3-Next-80B-A3B-Instruct": {
1629
+ "id": "Qwen/Qwen3-Next-80B-A3B-Instruct",
1630
+ "name": "Qwen3-Next 80B-A3B Instruct",
1631
+ "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
1477
1632
  "family": "qwen",
1478
1633
  "attachment": false,
1479
1634
  "reasoning": false,
1480
1635
  "tool_call": true,
1481
1636
  "structured_output": true,
1482
1637
  "temperature": true,
1483
- "release_date": "2026-05-21",
1484
- "last_updated": "2026-05-21",
1638
+ "knowledge": "2025-04",
1639
+ "release_date": "2025-09",
1640
+ "last_updated": "2025-09",
1485
1641
  "modalities": {
1486
1642
  "input": [
1487
1643
  "text"
@@ -1490,40 +1646,19 @@
1490
1646
  "text"
1491
1647
  ]
1492
1648
  },
1493
- "open_weights": false,
1649
+ "open_weights": true,
1494
1650
  "limit": {
1495
- "context": 256000,
1496
- "output": 65536
1651
+ "context": 262144,
1652
+ "output": 32768
1497
1653
  },
1498
1654
  "cost": {
1499
- "input": 2.5,
1500
- "output": 7.5,
1501
- "cache_read": 0.5,
1502
- "tiers": [
1503
- {
1504
- "input": 5,
1505
- "output": 15,
1506
- "cache_read": 1,
1507
- "tier": {
1508
- "type": "context",
1509
- "size": 32000
1510
- }
1511
- },
1512
- {
1513
- "input": 6.25,
1514
- "output": 18.5,
1515
- "cache_read": 1.25,
1516
- "tier": {
1517
- "type": "context",
1518
- "size": 128000
1519
- }
1520
- }
1521
- ]
1655
+ "input": 0.09,
1656
+ "output": 1.1
1522
1657
  }
1523
1658
  },
1524
- "Qwen/Qwen3.5-27B": {
1525
- "id": "Qwen/Qwen3.5-27B",
1526
- "name": "Qwen3.5 27B",
1659
+ "Qwen/Qwen3.5-397B-A17B": {
1660
+ "id": "Qwen/Qwen3.5-397B-A17B",
1661
+ "name": "Qwen 3.5 397B A17B",
1527
1662
  "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
1528
1663
  "family": "qwen",
1529
1664
  "attachment": true,
@@ -1532,14 +1667,14 @@
1532
1667
  "tool_call": true,
1533
1668
  "structured_output": true,
1534
1669
  "temperature": true,
1535
- "release_date": "2026-02-23",
1536
- "last_updated": "2026-02-23",
1670
+ "knowledge": "2025-01",
1671
+ "release_date": "2026-02-01",
1672
+ "last_updated": "2026-04-20",
1537
1673
  "modalities": {
1538
1674
  "input": [
1539
1675
  "text",
1540
1676
  "image",
1541
- "video",
1542
- "audio"
1677
+ "video"
1543
1678
  ],
1544
1679
  "output": [
1545
1680
  "text"
@@ -1548,11 +1683,12 @@
1548
1683
  "open_weights": true,
1549
1684
  "limit": {
1550
1685
  "context": 262144,
1551
- "output": 65536
1686
+ "output": 81920
1552
1687
  },
1553
1688
  "cost": {
1554
- "input": 0.26,
1555
- "output": 2.6
1689
+ "input": 0.45,
1690
+ "output": 3,
1691
+ "cache_read": 0.22
1556
1692
  }
1557
1693
  },
1558
1694
  "Qwen/Qwen3.5-35B-A3B": {
@@ -1590,62 +1726,10 @@
1590
1726
  "cache_read": 0.05
1591
1727
  }
1592
1728
  },
1593
- "Qwen/Qwen3-Max": {
1594
- "id": "Qwen/Qwen3-Max",
1595
- "name": "Qwen3 Max",
1596
- "description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use",
1597
- "family": "qwen",
1598
- "attachment": false,
1599
- "reasoning": false,
1600
- "tool_call": true,
1601
- "structured_output": true,
1602
- "temperature": true,
1603
- "knowledge": "2025-04",
1604
- "release_date": "2025-09-23",
1605
- "last_updated": "2025-09-23",
1606
- "modalities": {
1607
- "input": [
1608
- "text"
1609
- ],
1610
- "output": [
1611
- "text"
1612
- ]
1613
- },
1614
- "open_weights": false,
1615
- "limit": {
1616
- "context": 256000,
1617
- "output": 65536
1618
- },
1619
- "cost": {
1620
- "input": 1.2,
1621
- "output": 6,
1622
- "cache_read": 0.24,
1623
- "tiers": [
1624
- {
1625
- "input": 2.4,
1626
- "output": 12,
1627
- "cache_read": 0.48,
1628
- "tier": {
1629
- "type": "context",
1630
- "size": 32000
1631
- }
1632
- },
1633
- {
1634
- "input": 3,
1635
- "output": 15,
1636
- "cache_read": 0.6,
1637
- "tier": {
1638
- "type": "context",
1639
- "size": 128000
1640
- }
1641
- }
1642
- ]
1643
- }
1644
- },
1645
- "Qwen/Qwen3-30B-A3B": {
1646
- "id": "Qwen/Qwen3-30B-A3B",
1647
- "name": "Qwen3 30B A3B",
1648
- "description": "Sparse MoE Qwen model with 3B active parameters for efficient chat and reasoning",
1729
+ "Qwen/Qwen3-32B": {
1730
+ "id": "Qwen/Qwen3-32B",
1731
+ "name": "Qwen3 32B",
1732
+ "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding",
1649
1733
  "family": "qwen",
1650
1734
  "attachment": false,
1651
1735
  "reasoning": true,
@@ -1653,8 +1737,9 @@
1653
1737
  "tool_call": true,
1654
1738
  "structured_output": true,
1655
1739
  "temperature": true,
1656
- "release_date": "2025-04-28",
1657
- "last_updated": "2025-04-28",
1740
+ "knowledge": "2025-04",
1741
+ "release_date": "2025-04",
1742
+ "last_updated": "2025-04",
1658
1743
  "modalities": {
1659
1744
  "input": [
1660
1745
  "text"
@@ -1669,13 +1754,13 @@
1669
1754
  "output": 16384
1670
1755
  },
1671
1756
  "cost": {
1672
- "input": 0.12,
1673
- "output": 0.5
1757
+ "input": 0.08,
1758
+ "output": 0.28
1674
1759
  }
1675
1760
  },
1676
- "Qwen/Qwen3.5-397B-A17B": {
1677
- "id": "Qwen/Qwen3.5-397B-A17B",
1678
- "name": "Qwen 3.5 397B A17B",
1761
+ "Qwen/Qwen3.6-27B": {
1762
+ "id": "Qwen/Qwen3.6-27B",
1763
+ "name": "Qwen3.6 27B",
1679
1764
  "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
1680
1765
  "family": "qwen",
1681
1766
  "attachment": true,
@@ -1684,14 +1769,14 @@
1684
1769
  "tool_call": true,
1685
1770
  "structured_output": true,
1686
1771
  "temperature": true,
1687
- "knowledge": "2025-01",
1688
- "release_date": "2026-02-01",
1689
- "last_updated": "2026-04-20",
1772
+ "release_date": "2026-04-22",
1773
+ "last_updated": "2026-04-22",
1690
1774
  "modalities": {
1691
1775
  "input": [
1692
1776
  "text",
1693
1777
  "image",
1694
- "video"
1778
+ "video",
1779
+ "audio"
1695
1780
  ],
1696
1781
  "output": [
1697
1782
  "text"
@@ -1700,12 +1785,11 @@
1700
1785
  "open_weights": true,
1701
1786
  "limit": {
1702
1787
  "context": 262144,
1703
- "output": 81920
1788
+ "output": 65536
1704
1789
  },
1705
1790
  "cost": {
1706
- "input": 0.45,
1707
- "output": 3,
1708
- "cache_read": 0.22
1791
+ "input": 0.32,
1792
+ "output": 3.2
1709
1793
  }
1710
1794
  },
1711
1795
  "Qwen/Qwen3.6-35B-A3B": {
@@ -1741,28 +1825,74 @@
1741
1825
  "output": 0.95
1742
1826
  }
1743
1827
  },
1744
- "Qwen/Qwen3.8-2.4T-A95B": {
1745
- "id": "Qwen/Qwen3.8-2.4T-A95B",
1746
- "name": "Qwen3.8 2.4T A95B",
1747
- "description": "Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows",
1828
+ "Qwen/Qwen3-Max": {
1829
+ "id": "Qwen/Qwen3-Max",
1830
+ "name": "Qwen3 Max",
1831
+ "description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use",
1748
1832
  "family": "qwen",
1749
1833
  "attachment": false,
1750
- "reasoning": true,
1751
- "reasoning_options": [
1752
- {
1753
- "type": "effort",
1754
- "values": [
1755
- "low",
1756
- "medium",
1757
- "xhigh"
1758
- ]
1759
- }
1760
- ],
1834
+ "reasoning": false,
1761
1835
  "tool_call": true,
1762
1836
  "structured_output": true,
1763
1837
  "temperature": true,
1764
- "release_date": "2026-08-12",
1765
- "last_updated": "2026-08-12",
1838
+ "knowledge": "2025-04",
1839
+ "release_date": "2025-09-23",
1840
+ "last_updated": "2025-09-23",
1841
+ "modalities": {
1842
+ "input": [
1843
+ "text"
1844
+ ],
1845
+ "output": [
1846
+ "text"
1847
+ ]
1848
+ },
1849
+ "open_weights": false,
1850
+ "limit": {
1851
+ "context": 256000,
1852
+ "output": 65536
1853
+ },
1854
+ "cost": {
1855
+ "input": 1.2,
1856
+ "output": 6,
1857
+ "cache_read": 0.24,
1858
+ "tiers": [
1859
+ {
1860
+ "input": 2.4,
1861
+ "output": 12,
1862
+ "cache_read": 0.48,
1863
+ "tier": {
1864
+ "type": "context",
1865
+ "size": 32000
1866
+ }
1867
+ },
1868
+ {
1869
+ "input": 3,
1870
+ "output": 15,
1871
+ "cache_read": 0.6,
1872
+ "tier": {
1873
+ "type": "context",
1874
+ "size": 128000
1875
+ }
1876
+ }
1877
+ ]
1878
+ }
1879
+ },
1880
+ "MiniMaxAI/MiniMax-M2.5": {
1881
+ "id": "MiniMaxAI/MiniMax-M2.5",
1882
+ "name": "MiniMax M2.5",
1883
+ "description": "MiniMax model for chat, coding, office work, and agentic tasks",
1884
+ "family": "minimax",
1885
+ "attachment": false,
1886
+ "reasoning": true,
1887
+ "reasoning_options": [],
1888
+ "tool_call": true,
1889
+ "interleaved": {
1890
+ "field": "reasoning_content"
1891
+ },
1892
+ "temperature": true,
1893
+ "knowledge": "2025-06",
1894
+ "release_date": "2026-02-12",
1895
+ "last_updated": "2026-02-12",
1766
1896
  "modalities": {
1767
1897
  "input": [
1768
1898
  "text"
@@ -1773,44 +1903,34 @@
1773
1903
  },
1774
1904
  "open_weights": true,
1775
1905
  "limit": {
1776
- "context": 262144,
1906
+ "context": 196608,
1777
1907
  "output": 131072
1778
1908
  },
1909
+ "status": "deprecated",
1779
1910
  "cost": {
1780
- "input": 2,
1781
- "output": 6,
1782
- "cache_read": 0.2
1911
+ "input": 0.15,
1912
+ "output": 1.15,
1913
+ "cache_read": 0.03
1783
1914
  }
1784
1915
  },
1785
- "Qwen/Qwen3.8-27B": {
1786
- "id": "Qwen/Qwen3.8-27B",
1787
- "name": "Qwen3.8 27B",
1788
- "description": "Dense 27B vision-language model for coding, agent tasks, and image and video understanding",
1789
- "family": "qwen",
1916
+ "MiniMaxAI/MiniMax-M3": {
1917
+ "id": "MiniMaxAI/MiniMax-M3",
1918
+ "name": "MiniMax-M3",
1919
+ "description": "MiniMax multimodal model for long-context coding, perception, and agent planning",
1920
+ "family": "minimax",
1790
1921
  "attachment": true,
1791
1922
  "reasoning": true,
1792
- "reasoning_options": [
1793
- {
1794
- "type": "toggle"
1795
- },
1796
- {
1797
- "type": "effort",
1798
- "values": [
1799
- "low",
1800
- "medium",
1801
- "xhigh"
1802
- ]
1803
- }
1804
- ],
1923
+ "reasoning_options": [],
1805
1924
  "tool_call": true,
1806
1925
  "structured_output": true,
1807
1926
  "temperature": true,
1808
- "release_date": "2026-08-14",
1809
- "last_updated": "2026-08-14",
1927
+ "release_date": "2026-06-01",
1928
+ "last_updated": "2026-06-01",
1810
1929
  "modalities": {
1811
1930
  "input": [
1812
1931
  "text",
1813
- "image"
1932
+ "image",
1933
+ "video"
1814
1934
  ],
1815
1935
  "output": [
1816
1936
  "text"
@@ -1818,28 +1938,27 @@
1818
1938
  },
1819
1939
  "open_weights": true,
1820
1940
  "limit": {
1821
- "context": 262144,
1822
- "output": 32768
1941
+ "context": 524288,
1942
+ "output": 512000
1823
1943
  },
1824
1944
  "cost": {
1825
- "input": 0.4,
1826
- "output": 3,
1827
- "cache_read": 0.04
1945
+ "input": 0.28,
1946
+ "output": 1.1,
1947
+ "cache_read": 0.056
1828
1948
  }
1829
1949
  },
1830
- "Qwen/Qwen3-Next-80B-A3B-Instruct": {
1831
- "id": "Qwen/Qwen3-Next-80B-A3B-Instruct",
1832
- "name": "Qwen3-Next 80B-A3B Instruct",
1833
- "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
1834
- "family": "qwen",
1950
+ "MiniMaxAI/MiniMax-M2.7": {
1951
+ "id": "MiniMaxAI/MiniMax-M2.7",
1952
+ "name": "MiniMax-M2.7",
1953
+ "description": "Open MiniMax flagship for coding agents, office automation, and complex environments",
1954
+ "family": "minimax",
1835
1955
  "attachment": false,
1836
- "reasoning": false,
1956
+ "reasoning": true,
1957
+ "reasoning_options": [],
1837
1958
  "tool_call": true,
1838
- "structured_output": true,
1839
1959
  "temperature": true,
1840
- "knowledge": "2025-04",
1841
- "release_date": "2025-09",
1842
- "last_updated": "2025-09",
1960
+ "release_date": "2026-03-18",
1961
+ "last_updated": "2026-03-18",
1843
1962
  "modalities": {
1844
1963
  "input": [
1845
1964
  "text"
@@ -1850,38 +1969,30 @@
1850
1969
  },
1851
1970
  "open_weights": true,
1852
1971
  "limit": {
1853
- "context": 262144,
1854
- "output": 32768
1972
+ "context": 196608,
1973
+ "output": 131072
1855
1974
  },
1856
1975
  "cost": {
1857
- "input": 0.09,
1858
- "output": 1.1
1976
+ "input": 0.25,
1977
+ "output": 1,
1978
+ "cache_read": 0.05
1859
1979
  }
1860
1980
  },
1861
- "zai-org/GLM-5.1": {
1862
- "id": "zai-org/GLM-5.1",
1863
- "name": "GLM-5.1",
1864
- "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
1865
- "family": "glm",
1866
- "attachment": false,
1867
- "reasoning": true,
1868
- "reasoning_options": [
1869
- {
1870
- "type": "toggle"
1871
- }
1872
- ],
1981
+ "meta-llama/Llama-4-Scout-17B-16E-Instruct": {
1982
+ "id": "meta-llama/Llama-4-Scout-17B-16E-Instruct",
1983
+ "name": "Llama 4 Scout 17B",
1984
+ "description": "Open multimodal Llama model for long-context analysis and efficient agents",
1985
+ "family": "llama",
1986
+ "attachment": true,
1987
+ "reasoning": false,
1873
1988
  "tool_call": true,
1874
- "interleaved": {
1875
- "field": "reasoning_content"
1876
- },
1877
1989
  "structured_output": true,
1878
- "temperature": true,
1879
- "knowledge": "2025-04",
1880
- "release_date": "2026-04-07",
1881
- "last_updated": "2026-04-07",
1990
+ "release_date": "2025-04-05",
1991
+ "last_updated": "2025-04-05",
1882
1992
  "modalities": {
1883
1993
  "input": [
1884
- "text"
1994
+ "text",
1995
+ "image"
1885
1996
  ],
1886
1997
  "output": [
1887
1998
  "text"
@@ -1889,36 +2000,25 @@
1889
2000
  },
1890
2001
  "open_weights": true,
1891
2002
  "limit": {
1892
- "context": 202752,
2003
+ "context": 327680,
1893
2004
  "output": 16384
1894
2005
  },
1895
2006
  "cost": {
1896
- "input": 1.05,
1897
- "output": 3.5,
1898
- "cache_read": 0.205
2007
+ "input": 0.1,
2008
+ "output": 0.3
1899
2009
  }
1900
2010
  },
1901
- "zai-org/GLM-4.6": {
1902
- "id": "zai-org/GLM-4.6",
1903
- "name": "GLM-4.6",
1904
- "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
1905
- "family": "glm",
2011
+ "meta-llama/Llama-3.3-70B-Instruct-Turbo": {
2012
+ "id": "meta-llama/Llama-3.3-70B-Instruct-Turbo",
2013
+ "name": "Llama 3.3 70B Turbo",
2014
+ "description": "Compact Llama instruction model for fast chat and local deployment",
2015
+ "family": "llama",
1906
2016
  "attachment": false,
1907
- "reasoning": true,
1908
- "reasoning_options": [
1909
- {
1910
- "type": "toggle"
1911
- }
1912
- ],
2017
+ "reasoning": false,
1913
2018
  "tool_call": true,
1914
- "interleaved": {
1915
- "field": "reasoning_content"
1916
- },
1917
2019
  "structured_output": true,
1918
- "temperature": true,
1919
- "knowledge": "2025-04",
1920
- "release_date": "2025-09-30",
1921
- "last_updated": "2025-09-30",
2020
+ "release_date": "2024-12-06",
2021
+ "last_updated": "2024-12-06",
1922
2022
  "modalities": {
1923
2023
  "input": [
1924
2024
  "text"
@@ -1929,36 +2029,66 @@
1929
2029
  },
1930
2030
  "open_weights": true,
1931
2031
  "limit": {
1932
- "context": 202752,
1933
- "output": 131072
2032
+ "context": 131072,
2033
+ "output": 16384
1934
2034
  },
1935
2035
  "cost": {
1936
- "input": 0.5,
1937
- "output": 2,
1938
- "cache_read": 0.1
2036
+ "input": 0.1,
2037
+ "output": 0.32
1939
2038
  }
1940
2039
  },
1941
- "zai-org/GLM-5": {
1942
- "id": "zai-org/GLM-5",
1943
- "name": "GLM-5",
1944
- "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
1945
- "family": "glm",
2040
+ "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8": {
2041
+ "id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
2042
+ "name": "Llama 4 Maverick 17B FP8",
2043
+ "description": "Open multimodal Llama model for strong reasoning and fast responses",
2044
+ "family": "llama",
2045
+ "attachment": true,
2046
+ "reasoning": false,
2047
+ "tool_call": false,
2048
+ "structured_output": true,
2049
+ "release_date": "2025-04-05",
2050
+ "last_updated": "2025-04-05",
2051
+ "modalities": {
2052
+ "input": [
2053
+ "text",
2054
+ "image"
2055
+ ],
2056
+ "output": [
2057
+ "text"
2058
+ ]
2059
+ },
2060
+ "open_weights": true,
2061
+ "limit": {
2062
+ "context": 1048576,
2063
+ "output": 16384
2064
+ },
2065
+ "cost": {
2066
+ "input": 0.2,
2067
+ "output": 0.8
2068
+ }
2069
+ },
2070
+ "openai/gpt-oss-20b": {
2071
+ "id": "openai/gpt-oss-20b",
2072
+ "name": "GPT OSS 20B",
2073
+ "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
2074
+ "family": "gpt-oss",
1946
2075
  "attachment": false,
1947
2076
  "reasoning": true,
1948
2077
  "reasoning_options": [
1949
2078
  {
1950
- "type": "toggle"
2079
+ "type": "effort",
2080
+ "values": [
2081
+ "low",
2082
+ "medium",
2083
+ "high"
2084
+ ]
1951
2085
  }
1952
2086
  ],
1953
2087
  "tool_call": true,
1954
- "interleaved": {
1955
- "field": "reasoning_content"
1956
- },
1957
2088
  "structured_output": true,
1958
2089
  "temperature": true,
1959
- "knowledge": "2025-12",
1960
- "release_date": "2026-02-12",
1961
- "last_updated": "2026-02-12",
2090
+ "release_date": "2025-08-05",
2091
+ "last_updated": "2025-08-05",
1962
2092
  "modalities": {
1963
2093
  "input": [
1964
2094
  "text"
@@ -1969,36 +2099,36 @@
1969
2099
  },
1970
2100
  "open_weights": true,
1971
2101
  "limit": {
1972
- "context": 202752,
2102
+ "context": 131072,
1973
2103
  "output": 16384
1974
2104
  },
1975
2105
  "cost": {
1976
- "input": 0.6,
1977
- "output": 2.08,
1978
- "cache_read": 0.12
2106
+ "input": 0.03,
2107
+ "output": 0.14
1979
2108
  }
1980
2109
  },
1981
- "zai-org/GLM-4.7": {
1982
- "id": "zai-org/GLM-4.7",
1983
- "name": "GLM-4.7",
1984
- "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
1985
- "family": "glm",
2110
+ "openai/gpt-oss-120b": {
2111
+ "id": "openai/gpt-oss-120b",
2112
+ "name": "GPT OSS 120B",
2113
+ "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
2114
+ "family": "gpt-oss",
1986
2115
  "attachment": false,
1987
2116
  "reasoning": true,
1988
2117
  "reasoning_options": [
1989
2118
  {
1990
- "type": "toggle"
2119
+ "type": "effort",
2120
+ "values": [
2121
+ "low",
2122
+ "medium",
2123
+ "high"
2124
+ ]
1991
2125
  }
1992
2126
  ],
1993
2127
  "tool_call": true,
1994
- "interleaved": {
1995
- "field": "reasoning_content"
1996
- },
1997
2128
  "structured_output": true,
1998
2129
  "temperature": true,
1999
- "knowledge": "2025-04",
2000
- "release_date": "2025-12-22",
2001
- "last_updated": "2025-12-22",
2130
+ "release_date": "2025-08-05",
2131
+ "last_updated": "2025-08-05",
2002
2132
  "modalities": {
2003
2133
  "input": [
2004
2134
  "text"
@@ -2009,34 +2139,24 @@
2009
2139
  },
2010
2140
  "open_weights": true,
2011
2141
  "limit": {
2012
- "context": 202752,
2142
+ "context": 131072,
2013
2143
  "output": 16384
2014
2144
  },
2015
2145
  "cost": {
2016
- "input": 0.4,
2017
- "output": 1.75,
2018
- "cache_read": 0.08
2146
+ "input": 0.037,
2147
+ "output": 0.17
2019
2148
  }
2020
2149
  },
2021
- "zai-org/GLM-5.2": {
2022
- "id": "zai-org/GLM-5.2",
2023
- "name": "GLM-5.2",
2024
- "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
2025
- "family": "glm",
2026
- "attachment": false,
2150
+ "moonshotai/Kimi-K2.5": {
2151
+ "id": "moonshotai/Kimi-K2.5",
2152
+ "name": "Kimi K2.5",
2153
+ "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
2154
+ "family": "kimi-k2",
2155
+ "attachment": true,
2027
2156
  "reasoning": true,
2028
2157
  "reasoning_options": [
2029
2158
  {
2030
2159
  "type": "toggle"
2031
- },
2032
- {
2033
- "type": "effort",
2034
- "values": [
2035
- "low",
2036
- "medium",
2037
- "high",
2038
- "xhigh"
2039
- ]
2040
2160
  }
2041
2161
  ],
2042
2162
  "tool_call": true,
@@ -2045,11 +2165,14 @@
2045
2165
  },
2046
2166
  "structured_output": true,
2047
2167
  "temperature": true,
2048
- "release_date": "2026-06-13",
2049
- "last_updated": "2026-06-13",
2168
+ "knowledge": "2025-01",
2169
+ "release_date": "2026-01-27",
2170
+ "last_updated": "2026-01-27",
2050
2171
  "modalities": {
2051
2172
  "input": [
2052
- "text"
2173
+ "text",
2174
+ "image",
2175
+ "video"
2053
2176
  ],
2054
2177
  "output": [
2055
2178
  "text"
@@ -2057,35 +2180,42 @@
2057
2180
  },
2058
2181
  "open_weights": true,
2059
2182
  "limit": {
2060
- "context": 1048576,
2183
+ "context": 262144,
2061
2184
  "output": 32768
2062
2185
  },
2186
+ "status": "deprecated",
2063
2187
  "cost": {
2064
- "input": 0.75,
2065
- "output": 2.4,
2066
- "cache_read": 0.14
2188
+ "input": 0.45,
2189
+ "output": 2.25,
2190
+ "cache_read": 0.07
2067
2191
  }
2068
2192
  },
2069
- "zai-org/GLM-4.7-Flash": {
2070
- "id": "zai-org/GLM-4.7-Flash",
2071
- "name": "GLM-4.7-Flash",
2072
- "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
2073
- "family": "glm-flash",
2074
- "attachment": false,
2193
+ "moonshotai/Kimi-K2.7-Code": {
2194
+ "id": "moonshotai/Kimi-K2.7-Code",
2195
+ "name": "Kimi K2.7 Code",
2196
+ "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
2197
+ "family": "kimi-k2",
2198
+ "attachment": true,
2075
2199
  "reasoning": true,
2076
- "reasoning_options": [],
2200
+ "reasoning_options": [
2201
+ {
2202
+ "type": "toggle"
2203
+ }
2204
+ ],
2077
2205
  "tool_call": true,
2078
2206
  "interleaved": {
2079
2207
  "field": "reasoning_content"
2080
2208
  },
2081
2209
  "structured_output": true,
2082
- "temperature": true,
2083
- "knowledge": "2025-04",
2084
- "release_date": "2026-01-19",
2085
- "last_updated": "2026-01-19",
2210
+ "temperature": false,
2211
+ "knowledge": "2025-01",
2212
+ "release_date": "2026-06-12",
2213
+ "last_updated": "2026-06-12",
2086
2214
  "modalities": {
2087
2215
  "input": [
2088
- "text"
2216
+ "text",
2217
+ "image",
2218
+ "video"
2089
2219
  ],
2090
2220
  "output": [
2091
2221
  "text"
@@ -2093,32 +2223,41 @@
2093
2223
  },
2094
2224
  "open_weights": true,
2095
2225
  "limit": {
2096
- "context": 202752,
2097
- "output": 16384
2226
+ "context": 262144,
2227
+ "output": 262144
2098
2228
  },
2099
2229
  "cost": {
2100
- "input": 0.06,
2101
- "output": 0.4,
2102
- "cache_read": 0.01
2230
+ "input": 0.68,
2231
+ "output": 3.4,
2232
+ "cache_read": 0.136
2103
2233
  }
2104
2234
  },
2105
- "thinkingmachines/Inkling": {
2106
- "id": "thinkingmachines/Inkling",
2107
- "name": "Inkling",
2108
- "description": "Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio",
2109
- "family": "ling",
2235
+ "moonshotai/Kimi-K2.6": {
2236
+ "id": "moonshotai/Kimi-K2.6",
2237
+ "name": "Kimi K2.6",
2238
+ "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
2239
+ "family": "kimi-k2",
2110
2240
  "attachment": true,
2111
2241
  "reasoning": true,
2112
- "reasoning_options": [],
2242
+ "reasoning_options": [
2243
+ {
2244
+ "type": "toggle"
2245
+ }
2246
+ ],
2113
2247
  "tool_call": true,
2248
+ "interleaved": {
2249
+ "field": "reasoning_content"
2250
+ },
2251
+ "structured_output": true,
2114
2252
  "temperature": true,
2115
- "release_date": "2026-07-15",
2116
- "last_updated": "2026-07-15",
2253
+ "knowledge": "2024-04",
2254
+ "release_date": "2026-04-21",
2255
+ "last_updated": "2026-04-21",
2117
2256
  "modalities": {
2118
2257
  "input": [
2119
2258
  "text",
2120
2259
  "image",
2121
- "audio"
2260
+ "video"
2122
2261
  ],
2123
2262
  "output": [
2124
2263
  "text"
@@ -2126,33 +2265,41 @@
2126
2265
  },
2127
2266
  "open_weights": true,
2128
2267
  "limit": {
2129
- "context": 524288,
2130
- "output": 1048576
2268
+ "context": 262144,
2269
+ "output": 16384
2131
2270
  },
2132
2271
  "cost": {
2133
- "input": 0.95,
2134
- "output": 4.05,
2135
- "cache_read": 0.16
2272
+ "input": 0.75,
2273
+ "output": 3.5,
2274
+ "cache_read": 0.15
2136
2275
  }
2137
2276
  },
2138
- "thinkingmachines/Inkling-Small": {
2139
- "id": "thinkingmachines/Inkling-Small",
2140
- "name": "Inkling Small",
2141
- "description": "Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio",
2142
- "family": "ling",
2277
+ "moonshotai/Kimi-K3": {
2278
+ "id": "moonshotai/Kimi-K3",
2279
+ "name": "Kimi K3",
2280
+ "description": "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work",
2281
+ "family": "kimi-k3",
2143
2282
  "attachment": true,
2144
2283
  "reasoning": true,
2145
- "reasoning_options": [],
2284
+ "reasoning_options": [
2285
+ {
2286
+ "type": "effort",
2287
+ "values": [
2288
+ "low",
2289
+ "high",
2290
+ "max"
2291
+ ]
2292
+ }
2293
+ ],
2146
2294
  "tool_call": true,
2147
2295
  "structured_output": true,
2148
- "temperature": true,
2149
- "release_date": "2026-07-30",
2150
- "last_updated": "2026-07-30",
2296
+ "temperature": false,
2297
+ "release_date": "2026-07-16",
2298
+ "last_updated": "2026-07-16",
2151
2299
  "modalities": {
2152
2300
  "input": [
2153
2301
  "text",
2154
- "image",
2155
- "audio"
2302
+ "image"
2156
2303
  ],
2157
2304
  "output": [
2158
2305
  "text"
@@ -2160,30 +2307,31 @@
2160
2307
  },
2161
2308
  "open_weights": true,
2162
2309
  "limit": {
2163
- "context": 524288,
2164
- "output": 1048576
2310
+ "context": 1048576,
2311
+ "output": 131072
2165
2312
  },
2166
2313
  "cost": {
2167
- "input": 0.45,
2168
- "output": 1.2,
2169
- "cache_read": 0.1
2314
+ "input": 2.85,
2315
+ "output": 14.25,
2316
+ "cache_read": 0.285
2170
2317
  }
2171
2318
  },
2172
- "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8": {
2173
- "id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
2174
- "name": "Llama 4 Maverick 17B FP8",
2175
- "description": "Open multimodal Llama model for strong reasoning and fast responses",
2176
- "family": "llama",
2177
- "attachment": true,
2178
- "reasoning": false,
2179
- "tool_call": false,
2319
+ "tencent/Hy3": {
2320
+ "id": "tencent/Hy3",
2321
+ "name": "Hy3",
2322
+ "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
2323
+ "family": "Hy",
2324
+ "attachment": false,
2325
+ "reasoning": true,
2326
+ "reasoning_options": [],
2327
+ "tool_call": true,
2180
2328
  "structured_output": true,
2181
- "release_date": "2025-04-05",
2182
- "last_updated": "2025-04-05",
2329
+ "temperature": true,
2330
+ "release_date": "2026-07-06",
2331
+ "last_updated": "2026-07-06",
2183
2332
  "modalities": {
2184
2333
  "input": [
2185
- "text",
2186
- "image"
2334
+ "text"
2187
2335
  ],
2188
2336
  "output": [
2189
2337
  "text"
@@ -2191,28 +2339,41 @@
2191
2339
  },
2192
2340
  "open_weights": true,
2193
2341
  "limit": {
2194
- "context": 1048576,
2195
- "output": 16384
2342
+ "context": 262144,
2343
+ "input": 192000,
2344
+ "output": 128000
2196
2345
  },
2197
2346
  "cost": {
2198
- "input": 0.2,
2199
- "output": 0.8
2347
+ "input": 0.14,
2348
+ "output": 0.58,
2349
+ "cache_read": 0.035
2200
2350
  }
2201
2351
  },
2202
- "meta-llama/Llama-3.3-70B-Instruct-Turbo": {
2203
- "id": "meta-llama/Llama-3.3-70B-Instruct-Turbo",
2204
- "name": "Llama 3.3 70B Turbo",
2205
- "description": "Compact Llama instruction model for fast chat and local deployment",
2206
- "family": "llama",
2207
- "attachment": false,
2208
- "reasoning": false,
2352
+ "XiaomiMiMo/MiMo-V2.5-Pro": {
2353
+ "id": "XiaomiMiMo/MiMo-V2.5-Pro",
2354
+ "name": "MiMo-V2.5-Pro",
2355
+ "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
2356
+ "family": "mimo",
2357
+ "attachment": true,
2358
+ "reasoning": true,
2359
+ "reasoning_options": [
2360
+ {
2361
+ "type": "toggle"
2362
+ }
2363
+ ],
2209
2364
  "tool_call": true,
2365
+ "interleaved": {
2366
+ "field": "reasoning_content"
2367
+ },
2210
2368
  "structured_output": true,
2211
- "release_date": "2024-12-06",
2212
- "last_updated": "2024-12-06",
2369
+ "temperature": true,
2370
+ "knowledge": "2024-12",
2371
+ "release_date": "2026-04-22",
2372
+ "last_updated": "2026-04-22",
2213
2373
  "modalities": {
2214
2374
  "input": [
2215
- "text"
2375
+ "text",
2376
+ "audio"
2216
2377
  ],
2217
2378
  "output": [
2218
2379
  "text"
@@ -2220,29 +2381,42 @@
2220
2381
  },
2221
2382
  "open_weights": true,
2222
2383
  "limit": {
2223
- "context": 131072,
2384
+ "context": 1048576,
2224
2385
  "output": 16384
2225
2386
  },
2226
2387
  "cost": {
2227
- "input": 0.1,
2228
- "output": 0.32
2388
+ "input": 1,
2389
+ "output": 3,
2390
+ "cache_read": 0.2
2229
2391
  }
2230
2392
  },
2231
- "meta-llama/Llama-4-Scout-17B-16E-Instruct": {
2232
- "id": "meta-llama/Llama-4-Scout-17B-16E-Instruct",
2233
- "name": "Llama 4 Scout 17B",
2234
- "description": "Open multimodal Llama model for long-context analysis and efficient agents",
2235
- "family": "llama",
2393
+ "XiaomiMiMo/MiMo-V2.5": {
2394
+ "id": "XiaomiMiMo/MiMo-V2.5",
2395
+ "name": "MiMo-V2.5",
2396
+ "description": "Open MiMo model for multimodal coding agents and long-context automation",
2397
+ "family": "mimo",
2236
2398
  "attachment": true,
2237
- "reasoning": false,
2399
+ "reasoning": true,
2400
+ "reasoning_options": [
2401
+ {
2402
+ "type": "toggle"
2403
+ }
2404
+ ],
2238
2405
  "tool_call": true,
2406
+ "interleaved": {
2407
+ "field": "reasoning_content"
2408
+ },
2239
2409
  "structured_output": true,
2240
- "release_date": "2025-04-05",
2241
- "last_updated": "2025-04-05",
2410
+ "temperature": true,
2411
+ "knowledge": "2024-12",
2412
+ "release_date": "2026-04-22",
2413
+ "last_updated": "2026-04-22",
2242
2414
  "modalities": {
2243
2415
  "input": [
2244
2416
  "text",
2245
- "image"
2417
+ "image",
2418
+ "audio",
2419
+ "video"
2246
2420
  ],
2247
2421
  "output": [
2248
2422
  "text"
@@ -2250,12 +2424,13 @@
2250
2424
  },
2251
2425
  "open_weights": true,
2252
2426
  "limit": {
2253
- "context": 327680,
2427
+ "context": 262144,
2254
2428
  "output": 16384
2255
2429
  },
2256
2430
  "cost": {
2257
- "input": 0.1,
2258
- "output": 0.3
2431
+ "input": 0.4,
2432
+ "output": 2,
2433
+ "cache_read": 0.08
2259
2434
  }
2260
2435
  }
2261
2436
  }