llm.rb 13.0.0 → 14.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +505 -14
  3. data/README.md +484 -50
  4. data/bin/llm.rb +148 -0
  5. data/data/anthropic.json +206 -263
  6. data/data/bedrock.json +2138 -1860
  7. data/data/deepinfra.json +1003 -624
  8. data/data/deepseek.json +38 -34
  9. data/data/google.json +1079 -371
  10. data/data/mistral.json +448 -368
  11. data/data/moonshot.json +384 -0
  12. data/data/openai.json +974 -1343
  13. data/data/xai.json +154 -126
  14. data/data/zai.json +191 -191
  15. data/lib/llm/agent.rb +123 -20
  16. data/lib/llm/context.rb +71 -88
  17. data/lib/llm/cost.rb +23 -17
  18. data/lib/llm/error.rb +0 -8
  19. data/lib/llm/function/array.rb +3 -3
  20. data/lib/llm/function/async/task.rb +2 -0
  21. data/lib/llm/function/fiber/task.rb +2 -0
  22. data/lib/llm/function/fork/task.rb +2 -0
  23. data/lib/llm/function/ractor/task.rb +2 -0
  24. data/lib/llm/function/sequential/group.rb +4 -1
  25. data/lib/llm/function/sequential/task.rb +1 -1
  26. data/lib/llm/function/task.rb +4 -0
  27. data/lib/llm/function/thread/task.rb +2 -0
  28. data/lib/llm/function.rb +33 -6
  29. data/lib/llm/guard/loop.rb +89 -0
  30. data/lib/llm/guard/null.rb +19 -0
  31. data/lib/llm/guard.rb +61 -0
  32. data/lib/llm/provider.rb +36 -0
  33. data/lib/llm/providers/anthropic/stream_parser.rb +1 -0
  34. data/lib/llm/providers/anthropic.rb +2 -9
  35. data/lib/llm/providers/bedrock/request_adapter.rb +1 -1
  36. data/lib/llm/providers/bedrock/stream_parser.rb +1 -0
  37. data/lib/llm/providers/bedrock.rb +1 -8
  38. data/lib/llm/providers/google/stream_parser.rb +1 -0
  39. data/lib/llm/providers/google.rb +1 -8
  40. data/lib/llm/providers/mistral.rb +1 -1
  41. data/lib/llm/providers/moonshot.rb +76 -0
  42. data/lib/llm/providers/ollama.rb +2 -9
  43. data/lib/llm/providers/openai/responses/stream_parser.rb +1 -0
  44. data/lib/llm/providers/openai/responses.rb +7 -9
  45. data/lib/llm/providers/openai/stream_parser.rb +1 -0
  46. data/lib/llm/providers/openai.rb +4 -11
  47. data/lib/llm/repl/bar.rb +4 -3
  48. data/lib/llm/repl/{transcript.rb → buffer.rb} +69 -29
  49. data/lib/llm/repl/color.rb +78 -0
  50. data/lib/llm/repl/command.rb +12 -5
  51. data/lib/llm/repl/commands/compact.rb +2 -2
  52. data/lib/llm/repl/commands/help.rb +3 -5
  53. data/lib/llm/repl/input/char.rb +46 -0
  54. data/lib/llm/repl/input/row.rb +39 -0
  55. data/lib/llm/repl/input.rb +251 -66
  56. data/lib/llm/repl/markdown/table.rb +11 -3
  57. data/lib/llm/repl/markdown.rb +34 -8
  58. data/lib/llm/repl/node.rb +37 -0
  59. data/lib/llm/repl/status.rb +42 -7
  60. data/lib/llm/repl/stream.rb +18 -6
  61. data/lib/llm/repl/walker.rb +3 -2
  62. data/lib/llm/repl/window.rb +54 -35
  63. data/lib/llm/repl.rb +74 -32
  64. data/lib/llm/skill.rb +20 -4
  65. data/lib/llm/stream.rb +8 -7
  66. data/lib/llm/tool.rb +29 -0
  67. data/lib/llm/tools/{swap_text.rb → edit-file.rb} +3 -3
  68. data/lib/llm/tools/git.rb +3 -0
  69. data/lib/llm/tools/mkdir.rb +3 -0
  70. data/lib/llm/tools/rg.rb +3 -0
  71. data/lib/llm/tools/ruby.rb +46 -0
  72. data/lib/llm/tools/shell.rb +3 -0
  73. data/lib/llm/tracer/pretty_logger.rb +127 -0
  74. data/lib/llm/tracer.rb +1 -0
  75. data/lib/llm/transformer/null.rb +21 -0
  76. data/lib/llm/transformer.rb +55 -0
  77. data/lib/llm/version.rb +1 -1
  78. data/lib/llm.rb +12 -2
  79. data/llm.gemspec +9 -2
  80. data/resources/deepdive/advanced/cancellation.md +74 -0
  81. data/resources/deepdive/advanced/compaction.md +83 -0
  82. data/resources/deepdive/advanced/context.md +267 -0
  83. data/resources/deepdive/advanced/guard.md +371 -0
  84. data/resources/deepdive/advanced/tracer.md +180 -0
  85. data/resources/deepdive/advanced/transformer.md +67 -0
  86. data/resources/deepdive/advanced/transports.md +45 -0
  87. data/resources/deepdive/everything_else/audio.md +122 -0
  88. data/resources/deepdive/everything_else/cost.md +99 -0
  89. data/resources/deepdive/everything_else/images.md +89 -0
  90. data/resources/deepdive/everything_else/object.md +108 -0
  91. data/resources/deepdive/everything_else/ocr.md +48 -0
  92. data/resources/deepdive/fundamentals/agents.md +202 -0
  93. data/resources/deepdive/fundamentals/builtin_tools.md +191 -0
  94. data/resources/deepdive/fundamentals/concurrency.md +104 -0
  95. data/resources/deepdive/fundamentals/database.md +449 -0
  96. data/resources/deepdive/fundamentals/embeddings.md +157 -0
  97. data/resources/deepdive/fundamentals/repl.md +87 -0
  98. data/resources/deepdive/fundamentals/schema.md +61 -0
  99. data/resources/deepdive/fundamentals/skills.md +106 -0
  100. data/resources/deepdive/fundamentals/stream.md +110 -0
  101. data/resources/deepdive/fundamentals/tools.md +265 -0
  102. data/resources/deepdive/protocols/a2a.md +106 -0
  103. data/resources/deepdive/protocols/mcp.md +111 -0
  104. data/resources/deepdive.md +58 -1792
  105. metadata +51 -7
  106. data/lib/llm/loop_guard.rb +0 -107
data/data/deepinfra.json CHANGED
@@ -7,24 +7,22 @@
7
7
  "name": "Deep Infra",
8
8
  "doc": "https://deepinfra.com/models",
9
9
  "models": {
10
- "Qwen/Qwen3.5-9B": {
11
- "id": "Qwen/Qwen3.5-9B",
12
- "name": "Qwen3.5 9B",
13
- "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
14
- "family": "qwen",
15
- "attachment": true,
10
+ "nvidia/Llama-3.3-Nemotron-Super-49B-v1.5": {
11
+ "id": "nvidia/Llama-3.3-Nemotron-Super-49B-v1.5",
12
+ "name": "Llama 3.3 Nemotron Super 49B v1.5",
13
+ "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
14
+ "family": "nemotron",
15
+ "attachment": false,
16
16
  "reasoning": true,
17
17
  "reasoning_options": [],
18
18
  "tool_call": true,
19
19
  "structured_output": true,
20
20
  "temperature": true,
21
- "release_date": "2026-02-23",
22
- "last_updated": "2026-02-23",
21
+ "release_date": "2025-07-25",
22
+ "last_updated": "2025-07-25",
23
23
  "modalities": {
24
24
  "input": [
25
- "text",
26
- "image",
27
- "video"
25
+ "text"
28
26
  ],
29
27
  "output": [
30
28
  "text"
@@ -32,33 +30,34 @@
32
30
  },
33
31
  "open_weights": true,
34
32
  "limit": {
35
- "context": 262144,
36
- "output": 65536
33
+ "context": 131072,
34
+ "output": 131072
37
35
  },
36
+ "status": "deprecated",
38
37
  "cost": {
39
- "input": 0.1,
40
- "output": 0.15
38
+ "input": 0.4,
39
+ "output": 0.4
41
40
  }
42
41
  },
43
- "Qwen/Qwen3.5-35B-A3B": {
44
- "id": "Qwen/Qwen3.5-35B-A3B",
45
- "name": "Qwen 3.5 35B A3B",
46
- "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
47
- "family": "qwen",
42
+ "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning": {
43
+ "id": "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning",
44
+ "name": "Nemotron 3 Nano Omni 30B A3B Reasoning",
45
+ "description": "Open Nemotron omni model combining reasoning with text, vision, and audio",
46
+ "family": "nemotron",
48
47
  "attachment": true,
49
48
  "reasoning": true,
50
49
  "reasoning_options": [],
51
50
  "tool_call": true,
52
51
  "structured_output": true,
53
52
  "temperature": true,
54
- "knowledge": "2025-01",
55
- "release_date": "2026-02-01",
56
- "last_updated": "2026-04-20",
53
+ "release_date": "2026-04-28",
54
+ "last_updated": "2026-04-28",
57
55
  "modalities": {
58
56
  "input": [
59
57
  "text",
60
58
  "image",
61
- "video"
59
+ "video",
60
+ "audio"
62
61
  ],
63
62
  "output": [
64
63
  "text"
@@ -67,27 +66,30 @@
67
66
  "open_weights": true,
68
67
  "limit": {
69
68
  "context": 262144,
70
- "output": 81920
69
+ "output": 65536
71
70
  },
71
+ "status": "deprecated",
72
72
  "cost": {
73
- "input": 0.14,
74
- "output": 1,
75
- "cache_read": 0.05
73
+ "input": 0.2,
74
+ "output": 0.8
76
75
  }
77
76
  },
78
- "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo": {
79
- "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo",
80
- "name": "Qwen3 Coder 480B A35B Instruct Turbo",
81
- "description": "Qwen coding model for software agents, repository edits, and code reasoning",
82
- "family": "qwen",
77
+ "nvidia/Nemotron-3-Nano-30B-A3B": {
78
+ "id": "nvidia/Nemotron-3-Nano-30B-A3B",
79
+ "name": "Nemotron 3 Nano 30B A3B",
80
+ "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents",
81
+ "family": "nemotron",
83
82
  "attachment": false,
84
- "reasoning": false,
83
+ "reasoning": true,
84
+ "reasoning_options": [
85
+ {
86
+ "type": "toggle"
87
+ }
88
+ ],
85
89
  "tool_call": true,
86
- "structured_output": true,
87
90
  "temperature": true,
88
- "knowledge": "2025-04",
89
- "release_date": "2025-07-23",
90
- "last_updated": "2025-07-23",
91
+ "release_date": "2025-12-15",
92
+ "last_updated": "2025-12-15",
91
93
  "modalities": {
92
94
  "input": [
93
95
  "text"
@@ -99,31 +101,35 @@
99
101
  "open_weights": true,
100
102
  "limit": {
101
103
  "context": 262144,
102
- "output": 66536
104
+ "output": 262144
103
105
  },
104
106
  "cost": {
105
- "input": 0.3,
106
- "output": 1,
107
- "cache_read": 0.1
107
+ "input": 0.05,
108
+ "output": 0.2,
109
+ "cache_read": 0.025
108
110
  }
109
111
  },
110
- "Qwen/Qwen3-32B": {
111
- "id": "Qwen/Qwen3-32B",
112
- "name": "Qwen3 32B",
113
- "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding",
114
- "family": "qwen",
115
- "attachment": false,
112
+ "google/gemma-4-26B-A4B-it": {
113
+ "id": "google/gemma-4-26B-A4B-it",
114
+ "name": "Gemma 4 26B A4B IT",
115
+ "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
116
+ "family": "gemma",
117
+ "attachment": true,
116
118
  "reasoning": true,
117
- "reasoning_options": [],
119
+ "reasoning_options": [
120
+ {
121
+ "type": "toggle"
122
+ }
123
+ ],
118
124
  "tool_call": true,
119
125
  "structured_output": true,
120
126
  "temperature": true,
121
- "knowledge": "2025-04",
122
- "release_date": "2025-04",
123
- "last_updated": "2025-04",
127
+ "release_date": "2026-04-02",
128
+ "last_updated": "2026-04-02",
124
129
  "modalities": {
125
130
  "input": [
126
- "text"
131
+ "text",
132
+ "image"
127
133
  ],
128
134
  "output": [
129
135
  "text"
@@ -131,33 +137,36 @@
131
137
  },
132
138
  "open_weights": true,
133
139
  "limit": {
134
- "context": 40960,
135
- "output": 16384
140
+ "context": 262144,
141
+ "output": 32768
136
142
  },
137
143
  "cost": {
138
- "input": 0.08,
139
- "output": 0.28
144
+ "input": 0.07,
145
+ "output": 0.34
140
146
  }
141
147
  },
142
- "Qwen/Qwen3.5-397B-A17B": {
143
- "id": "Qwen/Qwen3.5-397B-A17B",
144
- "name": "Qwen 3.5 397B A17B",
145
- "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
146
- "family": "qwen",
148
+ "google/gemma-4-E4B-it": {
149
+ "id": "google/gemma-4-E4B-it",
150
+ "name": "Gemma 4 E4B IT",
151
+ "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
152
+ "family": "gemma",
147
153
  "attachment": true,
148
154
  "reasoning": true,
149
- "reasoning_options": [],
155
+ "reasoning_options": [
156
+ {
157
+ "type": "toggle"
158
+ }
159
+ ],
150
160
  "tool_call": true,
151
161
  "structured_output": true,
152
162
  "temperature": true,
153
- "knowledge": "2025-01",
154
- "release_date": "2026-02-01",
155
- "last_updated": "2026-04-20",
163
+ "release_date": "2026-04-02",
164
+ "last_updated": "2026-04-02",
156
165
  "modalities": {
157
166
  "input": [
158
167
  "text",
159
168
  "image",
160
- "video"
169
+ "audio"
161
170
  ],
162
171
  "output": [
163
172
  "text"
@@ -165,34 +174,36 @@
165
174
  },
166
175
  "open_weights": true,
167
176
  "limit": {
168
- "context": 262144,
169
- "output": 81920
177
+ "context": 131072,
178
+ "output": 8192
170
179
  },
171
180
  "cost": {
172
- "input": 0.45,
173
- "output": 3,
174
- "cache_read": 0.22
181
+ "input": 0.02,
182
+ "output": 0.1
175
183
  }
176
184
  },
177
- "Qwen/Qwen3.5-27B": {
178
- "id": "Qwen/Qwen3.5-27B",
179
- "name": "Qwen3.5 27B",
180
- "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
181
- "family": "qwen",
185
+ "google/gemma-4-31B-it": {
186
+ "id": "google/gemma-4-31B-it",
187
+ "name": "Gemma 4 31B IT",
188
+ "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
189
+ "family": "gemma",
182
190
  "attachment": true,
183
191
  "reasoning": true,
184
- "reasoning_options": [],
192
+ "reasoning_options": [
193
+ {
194
+ "type": "toggle"
195
+ }
196
+ ],
185
197
  "tool_call": true,
186
198
  "structured_output": true,
187
199
  "temperature": true,
188
- "release_date": "2026-02-23",
189
- "last_updated": "2026-02-23",
200
+ "release_date": "2026-04-02",
201
+ "last_updated": "2026-04-02",
190
202
  "modalities": {
191
203
  "input": [
192
204
  "text",
193
205
  "image",
194
- "video",
195
- "audio"
206
+ "video"
196
207
  ],
197
208
  "output": [
198
209
  "text"
@@ -201,31 +212,29 @@
201
212
  "open_weights": true,
202
213
  "limit": {
203
214
  "context": 262144,
204
- "output": 65536
215
+ "output": 32768
205
216
  },
206
217
  "cost": {
207
- "input": 0.26,
208
- "output": 2.6
218
+ "input": 0.13,
219
+ "output": 0.38
209
220
  }
210
221
  },
211
- "Qwen/Qwen3.6-27B": {
212
- "id": "Qwen/Qwen3.6-27B",
213
- "name": "Qwen3.6 27B",
214
- "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
215
- "family": "qwen",
222
+ "thinkingmachines/Inkling-Small": {
223
+ "id": "thinkingmachines/Inkling-Small",
224
+ "name": "Inkling Small",
225
+ "description": "Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio",
226
+ "family": "ling",
216
227
  "attachment": true,
217
228
  "reasoning": true,
218
229
  "reasoning_options": [],
219
230
  "tool_call": true,
220
- "structured_output": true,
221
231
  "temperature": true,
222
- "release_date": "2026-04-22",
223
- "last_updated": "2026-04-22",
232
+ "release_date": "2026-07-30",
233
+ "last_updated": "2026-07-30",
224
234
  "modalities": {
225
235
  "input": [
226
236
  "text",
227
237
  "image",
228
- "video",
229
238
  "audio"
230
239
  ],
231
240
  "output": [
@@ -234,32 +243,31 @@
234
243
  },
235
244
  "open_weights": true,
236
245
  "limit": {
237
- "context": 262144,
238
- "output": 65536
246
+ "context": 524288,
247
+ "output": 1048576
239
248
  },
240
249
  "cost": {
241
- "input": 0.32,
242
- "output": 3.2
250
+ "input": 0.45,
251
+ "output": 1.2,
252
+ "cache_read": 0.1
243
253
  }
244
254
  },
245
- "Qwen/Qwen3.5-122B-A10B": {
246
- "id": "Qwen/Qwen3.5-122B-A10B",
247
- "name": "Qwen3.5 122B-A10B",
248
- "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
249
- "family": "qwen",
255
+ "thinkingmachines/Inkling": {
256
+ "id": "thinkingmachines/Inkling",
257
+ "name": "Inkling",
258
+ "description": "Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio",
259
+ "family": "ling",
250
260
  "attachment": true,
251
261
  "reasoning": true,
252
262
  "reasoning_options": [],
253
263
  "tool_call": true,
254
- "structured_output": true,
255
264
  "temperature": true,
256
- "release_date": "2026-02-23",
257
- "last_updated": "2026-02-23",
265
+ "release_date": "2026-07-15",
266
+ "last_updated": "2026-07-15",
258
267
  "modalities": {
259
268
  "input": [
260
269
  "text",
261
270
  "image",
262
- "video",
263
271
  "audio"
264
272
  ],
265
273
  "output": [
@@ -268,27 +276,36 @@
268
276
  },
269
277
  "open_weights": true,
270
278
  "limit": {
271
- "context": 262144,
272
- "output": 65536
279
+ "context": 524288,
280
+ "output": 1048576
273
281
  },
274
282
  "cost": {
275
- "input": 0.29,
276
- "output": 2.4
283
+ "input": 0.95,
284
+ "output": 4.05,
285
+ "cache_read": 0.16
277
286
  }
278
287
  },
279
- "Qwen/Qwen3-Next-80B-A3B-Instruct": {
280
- "id": "Qwen/Qwen3-Next-80B-A3B-Instruct",
281
- "name": "Qwen3-Next 80B-A3B Instruct",
282
- "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
283
- "family": "qwen",
288
+ "zai-org/GLM-5": {
289
+ "id": "zai-org/GLM-5",
290
+ "name": "GLM-5",
291
+ "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
292
+ "family": "glm",
284
293
  "attachment": false,
285
- "reasoning": false,
294
+ "reasoning": true,
295
+ "reasoning_options": [
296
+ {
297
+ "type": "toggle"
298
+ }
299
+ ],
286
300
  "tool_call": true,
301
+ "interleaved": {
302
+ "field": "reasoning_content"
303
+ },
287
304
  "structured_output": true,
288
305
  "temperature": true,
289
- "knowledge": "2025-04",
290
- "release_date": "2025-09",
291
- "last_updated": "2025-09",
306
+ "knowledge": "2025-12",
307
+ "release_date": "2026-02-12",
308
+ "last_updated": "2026-02-12",
292
309
  "modalities": {
293
310
  "input": [
294
311
  "text"
@@ -299,26 +316,32 @@
299
316
  },
300
317
  "open_weights": true,
301
318
  "limit": {
302
- "context": 262144,
303
- "output": 32768
319
+ "context": 202752,
320
+ "output": 16384
304
321
  },
305
322
  "cost": {
306
- "input": 0.09,
307
- "output": 1.1
323
+ "input": 0.6,
324
+ "output": 2.08,
325
+ "cache_read": 0.12
308
326
  }
309
327
  },
310
- "Qwen/Qwen3.7-Max": {
311
- "id": "Qwen/Qwen3.7-Max",
312
- "name": "Qwen3.7 Max",
313
- "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
314
- "family": "qwen",
328
+ "zai-org/GLM-4.7-Flash": {
329
+ "id": "zai-org/GLM-4.7-Flash",
330
+ "name": "GLM-4.7-Flash",
331
+ "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
332
+ "family": "glm-flash",
315
333
  "attachment": false,
316
- "reasoning": false,
334
+ "reasoning": true,
335
+ "reasoning_options": [],
317
336
  "tool_call": true,
337
+ "interleaved": {
338
+ "field": "reasoning_content"
339
+ },
318
340
  "structured_output": true,
319
341
  "temperature": true,
320
- "release_date": "2026-05-21",
321
- "last_updated": "2026-05-21",
342
+ "knowledge": "2025-04",
343
+ "release_date": "2026-01-19",
344
+ "last_updated": "2026-01-19",
322
345
  "modalities": {
323
346
  "input": [
324
347
  "text"
@@ -327,50 +350,86 @@
327
350
  "text"
328
351
  ]
329
352
  },
330
- "open_weights": false,
353
+ "open_weights": true,
331
354
  "limit": {
332
- "context": 256000,
333
- "output": 65536
355
+ "context": 202752,
356
+ "output": 16384
334
357
  },
335
358
  "cost": {
336
- "input": 2.5,
337
- "output": 7.5,
338
- "cache_read": 0.5,
339
- "tiers": [
340
- {
341
- "input": 5,
342
- "output": 15,
343
- "cache_read": 1,
344
- "tier": {
345
- "type": "context",
346
- "size": 32000
347
- }
348
- },
349
- {
350
- "input": 6.25,
351
- "output": 18.5,
352
- "cache_read": 1.25,
353
- "tier": {
354
- "type": "context",
355
- "size": 128000
356
- }
357
- }
359
+ "input": 0.06,
360
+ "output": 0.4,
361
+ "cache_read": 0.01
362
+ }
363
+ },
364
+ "zai-org/GLM-5.2": {
365
+ "id": "zai-org/GLM-5.2",
366
+ "name": "GLM-5.2",
367
+ "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
368
+ "family": "glm",
369
+ "attachment": false,
370
+ "reasoning": true,
371
+ "reasoning_options": [
372
+ {
373
+ "type": "toggle"
374
+ },
375
+ {
376
+ "type": "effort",
377
+ "values": [
378
+ "low",
379
+ "medium",
380
+ "high",
381
+ "xhigh"
382
+ ]
383
+ }
384
+ ],
385
+ "tool_call": true,
386
+ "interleaved": {
387
+ "field": "reasoning_content"
388
+ },
389
+ "structured_output": true,
390
+ "temperature": true,
391
+ "release_date": "2026-06-13",
392
+ "last_updated": "2026-06-13",
393
+ "modalities": {
394
+ "input": [
395
+ "text"
396
+ ],
397
+ "output": [
398
+ "text"
358
399
  ]
400
+ },
401
+ "open_weights": true,
402
+ "limit": {
403
+ "context": 1048576,
404
+ "output": 32768
405
+ },
406
+ "cost": {
407
+ "input": 0.75,
408
+ "output": 2.4,
409
+ "cache_read": 0.14
359
410
  }
360
411
  },
361
- "Qwen/Qwen3-Max": {
362
- "id": "Qwen/Qwen3-Max",
363
- "name": "Qwen3 Max",
364
- "description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use",
365
- "family": "qwen",
412
+ "zai-org/GLM-5.1": {
413
+ "id": "zai-org/GLM-5.1",
414
+ "name": "GLM-5.1",
415
+ "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
416
+ "family": "glm",
366
417
  "attachment": false,
367
- "reasoning": false,
418
+ "reasoning": true,
419
+ "reasoning_options": [
420
+ {
421
+ "type": "toggle"
422
+ }
423
+ ],
368
424
  "tool_call": true,
425
+ "interleaved": {
426
+ "field": "reasoning_content"
427
+ },
369
428
  "structured_output": true,
370
429
  "temperature": true,
371
430
  "knowledge": "2025-04",
372
- "release_date": "2025-09-23",
373
- "last_updated": "2025-09-23",
431
+ "release_date": "2026-04-07",
432
+ "last_updated": "2026-04-07",
374
433
  "modalities": {
375
434
  "input": [
376
435
  "text"
@@ -379,40 +438,132 @@
379
438
  "text"
380
439
  ]
381
440
  },
382
- "open_weights": false,
441
+ "open_weights": true,
383
442
  "limit": {
384
- "context": 256000,
385
- "output": 65536
443
+ "context": 202752,
444
+ "output": 16384
386
445
  },
387
446
  "cost": {
388
- "input": 1.2,
389
- "output": 6,
390
- "cache_read": 0.24,
391
- "tiers": [
392
- {
393
- "input": 2.4,
394
- "output": 12,
395
- "cache_read": 0.48,
396
- "tier": {
397
- "type": "context",
398
- "size": 32000
399
- }
400
- },
401
- {
402
- "input": 3,
403
- "output": 15,
404
- "cache_read": 0.6,
405
- "tier": {
406
- "type": "context",
407
- "size": 128000
408
- }
409
- }
447
+ "input": 1.05,
448
+ "output": 3.5,
449
+ "cache_read": 0.205
450
+ }
451
+ },
452
+ "zai-org/GLM-4.6": {
453
+ "id": "zai-org/GLM-4.6",
454
+ "name": "GLM-4.6",
455
+ "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
456
+ "family": "glm",
457
+ "attachment": false,
458
+ "reasoning": true,
459
+ "reasoning_options": [
460
+ {
461
+ "type": "toggle"
462
+ }
463
+ ],
464
+ "tool_call": true,
465
+ "interleaved": {
466
+ "field": "reasoning_content"
467
+ },
468
+ "structured_output": true,
469
+ "temperature": true,
470
+ "knowledge": "2025-04",
471
+ "release_date": "2025-09-30",
472
+ "last_updated": "2025-09-30",
473
+ "modalities": {
474
+ "input": [
475
+ "text"
476
+ ],
477
+ "output": [
478
+ "text"
410
479
  ]
480
+ },
481
+ "open_weights": true,
482
+ "limit": {
483
+ "context": 202752,
484
+ "output": 131072
485
+ },
486
+ "cost": {
487
+ "input": 0.5,
488
+ "output": 2,
489
+ "cache_read": 0.1
411
490
  }
412
491
  },
413
- "Qwen/Qwen3.6-35B-A3B": {
414
- "id": "Qwen/Qwen3.6-35B-A3B",
415
- "name": "Qwen3.6 35B A3B",
492
+ "zai-org/GLM-4.7": {
493
+ "id": "zai-org/GLM-4.7",
494
+ "name": "GLM-4.7",
495
+ "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
496
+ "family": "glm",
497
+ "attachment": false,
498
+ "reasoning": true,
499
+ "reasoning_options": [
500
+ {
501
+ "type": "toggle"
502
+ }
503
+ ],
504
+ "tool_call": true,
505
+ "interleaved": {
506
+ "field": "reasoning_content"
507
+ },
508
+ "structured_output": true,
509
+ "temperature": true,
510
+ "knowledge": "2025-04",
511
+ "release_date": "2025-12-22",
512
+ "last_updated": "2025-12-22",
513
+ "modalities": {
514
+ "input": [
515
+ "text"
516
+ ],
517
+ "output": [
518
+ "text"
519
+ ]
520
+ },
521
+ "open_weights": true,
522
+ "limit": {
523
+ "context": 202752,
524
+ "output": 16384
525
+ },
526
+ "cost": {
527
+ "input": 0.4,
528
+ "output": 1.75,
529
+ "cache_read": 0.08
530
+ }
531
+ },
532
+ "tencent/Hy3": {
533
+ "id": "tencent/Hy3",
534
+ "name": "Hy3",
535
+ "description": "Tencent Hy reasoning model for coding, instruction following, and agent tasks",
536
+ "family": "Hy",
537
+ "attachment": false,
538
+ "reasoning": true,
539
+ "reasoning_options": [],
540
+ "tool_call": true,
541
+ "structured_output": true,
542
+ "temperature": true,
543
+ "release_date": "2026-07-06",
544
+ "last_updated": "2026-07-06",
545
+ "modalities": {
546
+ "input": [
547
+ "text"
548
+ ],
549
+ "output": [
550
+ "text"
551
+ ]
552
+ },
553
+ "open_weights": true,
554
+ "limit": {
555
+ "context": 262144,
556
+ "output": 64000
557
+ },
558
+ "cost": {
559
+ "input": 0.14,
560
+ "output": 0.58,
561
+ "cache_read": 0.035
562
+ }
563
+ },
564
+ "Qwen/Qwen3.5-27B": {
565
+ "id": "Qwen/Qwen3.5-27B",
566
+ "name": "Qwen3.5 27B",
416
567
  "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
417
568
  "family": "qwen",
418
569
  "attachment": true,
@@ -421,8 +572,42 @@
421
572
  "tool_call": true,
422
573
  "structured_output": true,
423
574
  "temperature": true,
424
- "release_date": "2026-04-01",
425
- "last_updated": "2026-04-01",
575
+ "release_date": "2026-02-23",
576
+ "last_updated": "2026-02-23",
577
+ "modalities": {
578
+ "input": [
579
+ "text",
580
+ "image",
581
+ "video",
582
+ "audio"
583
+ ],
584
+ "output": [
585
+ "text"
586
+ ]
587
+ },
588
+ "open_weights": true,
589
+ "limit": {
590
+ "context": 262144,
591
+ "output": 65536
592
+ },
593
+ "cost": {
594
+ "input": 0.26,
595
+ "output": 2.6
596
+ }
597
+ },
598
+ "Qwen/Qwen3.5-9B": {
599
+ "id": "Qwen/Qwen3.5-9B",
600
+ "name": "Qwen3.5 9B",
601
+ "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
602
+ "family": "qwen",
603
+ "attachment": true,
604
+ "reasoning": true,
605
+ "reasoning_options": [],
606
+ "tool_call": true,
607
+ "structured_output": true,
608
+ "temperature": true,
609
+ "release_date": "2026-02-23",
610
+ "last_updated": "2026-02-23",
426
611
  "modalities": {
427
612
  "input": [
428
613
  "text",
@@ -436,29 +621,25 @@
436
621
  "open_weights": true,
437
622
  "limit": {
438
623
  "context": 262144,
439
- "output": 81920
624
+ "output": 65536
440
625
  },
441
626
  "cost": {
442
- "input": 0.15,
443
- "output": 0.95
627
+ "input": 0.1,
628
+ "output": 0.15
444
629
  }
445
630
  },
446
- "nvidia/Nemotron-3-Nano-30B-A3B": {
447
- "id": "nvidia/Nemotron-3-Nano-30B-A3B",
448
- "name": "Nemotron 3 Nano 30B A3B",
449
- "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents",
450
- "family": "nemotron",
631
+ "Qwen/Qwen3-235B-A22B-Instruct-2507": {
632
+ "id": "Qwen/Qwen3-235B-A22B-Instruct-2507",
633
+ "name": "Qwen3 235B-A22B Instruct 2507",
634
+ "description": "Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use",
635
+ "family": "qwen",
451
636
  "attachment": false,
452
- "reasoning": true,
453
- "reasoning_options": [
454
- {
455
- "type": "toggle"
456
- }
457
- ],
637
+ "reasoning": false,
458
638
  "tool_call": true,
639
+ "structured_output": true,
459
640
  "temperature": true,
460
- "release_date": "2025-12-15",
461
- "last_updated": "2025-12-15",
641
+ "release_date": "2025-07-21",
642
+ "last_updated": "2025-07-21",
462
643
  "modalities": {
463
644
  "input": [
464
645
  "text"
@@ -470,26 +651,59 @@
470
651
  "open_weights": true,
471
652
  "limit": {
472
653
  "context": 262144,
473
- "output": 262144
654
+ "output": 16384
474
655
  },
475
656
  "cost": {
476
- "input": 0.05,
477
- "output": 0.2
657
+ "input": 0.09,
658
+ "output": 0.55
478
659
  }
479
660
  },
480
- "nvidia/Llama-3.3-Nemotron-Super-49B-v1.5": {
481
- "id": "nvidia/Llama-3.3-Nemotron-Super-49B-v1.5",
482
- "name": "Llama 3.3 Nemotron Super 49B v1.5",
483
- "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
484
- "family": "nemotron",
485
- "attachment": false,
661
+ "Qwen/Qwen3.5-122B-A10B": {
662
+ "id": "Qwen/Qwen3.5-122B-A10B",
663
+ "name": "Qwen3.5 122B-A10B",
664
+ "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
665
+ "family": "qwen",
666
+ "attachment": true,
486
667
  "reasoning": true,
487
668
  "reasoning_options": [],
488
669
  "tool_call": true,
489
670
  "structured_output": true,
490
671
  "temperature": true,
491
- "release_date": "2025-07-25",
492
- "last_updated": "2025-07-25",
672
+ "release_date": "2026-02-23",
673
+ "last_updated": "2026-02-23",
674
+ "modalities": {
675
+ "input": [
676
+ "text",
677
+ "image",
678
+ "video",
679
+ "audio"
680
+ ],
681
+ "output": [
682
+ "text"
683
+ ]
684
+ },
685
+ "open_weights": true,
686
+ "limit": {
687
+ "context": 262144,
688
+ "output": 65536
689
+ },
690
+ "cost": {
691
+ "input": 0.29,
692
+ "output": 2.4
693
+ }
694
+ },
695
+ "Qwen/Qwen3.7-Max": {
696
+ "id": "Qwen/Qwen3.7-Max",
697
+ "name": "Qwen3.7 Max",
698
+ "description": "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks",
699
+ "family": "qwen",
700
+ "attachment": false,
701
+ "reasoning": false,
702
+ "tool_call": true,
703
+ "structured_output": true,
704
+ "temperature": true,
705
+ "release_date": "2026-05-21",
706
+ "last_updated": "2026-05-21",
493
707
  "modalities": {
494
708
  "input": [
495
709
  "text"
@@ -498,30 +712,85 @@
498
712
  "text"
499
713
  ]
500
714
  },
715
+ "open_weights": false,
716
+ "limit": {
717
+ "context": 256000,
718
+ "output": 65536
719
+ },
720
+ "cost": {
721
+ "input": 2.5,
722
+ "output": 7.5,
723
+ "cache_read": 0.5,
724
+ "tiers": [
725
+ {
726
+ "input": 5,
727
+ "output": 15,
728
+ "cache_read": 1,
729
+ "tier": {
730
+ "type": "context",
731
+ "size": 32000
732
+ }
733
+ },
734
+ {
735
+ "input": 6.25,
736
+ "output": 18.5,
737
+ "cache_read": 1.25,
738
+ "tier": {
739
+ "type": "context",
740
+ "size": 128000
741
+ }
742
+ }
743
+ ]
744
+ }
745
+ },
746
+ "Qwen/Qwen3.5-35B-A3B": {
747
+ "id": "Qwen/Qwen3.5-35B-A3B",
748
+ "name": "Qwen 3.5 35B A3B",
749
+ "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
750
+ "family": "qwen",
751
+ "attachment": true,
752
+ "reasoning": true,
753
+ "reasoning_options": [],
754
+ "tool_call": true,
755
+ "structured_output": true,
756
+ "temperature": true,
757
+ "knowledge": "2025-01",
758
+ "release_date": "2026-02-01",
759
+ "last_updated": "2026-04-20",
760
+ "modalities": {
761
+ "input": [
762
+ "text",
763
+ "image",
764
+ "video"
765
+ ],
766
+ "output": [
767
+ "text"
768
+ ]
769
+ },
501
770
  "open_weights": true,
502
771
  "limit": {
503
- "context": 131072,
504
- "output": 131072
772
+ "context": 262144,
773
+ "output": 81920
505
774
  },
506
- "status": "deprecated",
507
775
  "cost": {
508
- "input": 0.4,
509
- "output": 0.4
776
+ "input": 0.14,
777
+ "output": 1,
778
+ "cache_read": 0.05
510
779
  }
511
780
  },
512
- "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning": {
513
- "id": "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning",
514
- "name": "Nemotron 3 Nano Omni 30B A3B Reasoning",
515
- "description": "Open Nemotron omni model combining reasoning with text, vision, and audio",
516
- "family": "nemotron",
781
+ "Qwen/Qwen3.6-27B": {
782
+ "id": "Qwen/Qwen3.6-27B",
783
+ "name": "Qwen3.6 27B",
784
+ "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
785
+ "family": "qwen",
517
786
  "attachment": true,
518
787
  "reasoning": true,
519
788
  "reasoning_options": [],
520
789
  "tool_call": true,
521
790
  "structured_output": true,
522
791
  "temperature": true,
523
- "release_date": "2026-04-28",
524
- "last_updated": "2026-04-28",
792
+ "release_date": "2026-04-22",
793
+ "last_updated": "2026-04-22",
525
794
  "modalities": {
526
795
  "input": [
527
796
  "text",
@@ -538,27 +807,30 @@
538
807
  "context": 262144,
539
808
  "output": 65536
540
809
  },
541
- "status": "deprecated",
542
810
  "cost": {
543
- "input": 0.2,
544
- "output": 0.8
811
+ "input": 0.32,
812
+ "output": 3.2
545
813
  }
546
814
  },
547
- "meta-llama/Llama-4-Scout-17B-16E-Instruct": {
548
- "id": "meta-llama/Llama-4-Scout-17B-16E-Instruct",
549
- "name": "Llama 4 Scout 17B",
550
- "description": "Open multimodal Llama model for long-context analysis and efficient agents",
551
- "family": "llama",
815
+ "Qwen/Qwen3.5-397B-A17B": {
816
+ "id": "Qwen/Qwen3.5-397B-A17B",
817
+ "name": "Qwen 3.5 397B A17B",
818
+ "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
819
+ "family": "qwen",
552
820
  "attachment": true,
553
- "reasoning": false,
821
+ "reasoning": true,
822
+ "reasoning_options": [],
554
823
  "tool_call": true,
555
824
  "structured_output": true,
556
- "release_date": "2025-04-05",
557
- "last_updated": "2025-04-05",
825
+ "temperature": true,
826
+ "knowledge": "2025-01",
827
+ "release_date": "2026-02-01",
828
+ "last_updated": "2026-04-20",
558
829
  "modalities": {
559
830
  "input": [
560
831
  "text",
561
- "image"
832
+ "image",
833
+ "video"
562
834
  ],
563
835
  "output": [
564
836
  "text"
@@ -566,29 +838,33 @@
566
838
  },
567
839
  "open_weights": true,
568
840
  "limit": {
569
- "context": 327680,
570
- "output": 16384
841
+ "context": 262144,
842
+ "output": 81920
571
843
  },
572
844
  "cost": {
573
- "input": 0.1,
574
- "output": 0.3
845
+ "input": 0.45,
846
+ "output": 3,
847
+ "cache_read": 0.22
575
848
  }
576
849
  },
577
- "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8": {
578
- "id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
579
- "name": "Llama 4 Maverick 17B FP8",
580
- "description": "Open multimodal Llama model for strong reasoning and fast responses",
581
- "family": "llama",
850
+ "Qwen/Qwen3.6-35B-A3B": {
851
+ "id": "Qwen/Qwen3.6-35B-A3B",
852
+ "name": "Qwen3.6 35B A3B",
853
+ "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
854
+ "family": "qwen",
582
855
  "attachment": true,
583
- "reasoning": false,
584
- "tool_call": false,
856
+ "reasoning": true,
857
+ "reasoning_options": [],
858
+ "tool_call": true,
585
859
  "structured_output": true,
586
- "release_date": "2025-04-05",
587
- "last_updated": "2025-04-05",
860
+ "temperature": true,
861
+ "release_date": "2026-04-01",
862
+ "last_updated": "2026-04-01",
588
863
  "modalities": {
589
864
  "input": [
590
865
  "text",
591
- "image"
866
+ "image",
867
+ "video"
592
868
  ],
593
869
  "output": [
594
870
  "text"
@@ -596,59 +872,62 @@
596
872
  },
597
873
  "open_weights": true,
598
874
  "limit": {
599
- "context": 1048576,
600
- "output": 16384
875
+ "context": 262144,
876
+ "output": 81920
601
877
  },
602
878
  "cost": {
603
- "input": 0.2,
604
- "output": 0.8
879
+ "input": 0.1,
880
+ "output": 0.95
605
881
  }
606
882
  },
607
- "meta-llama/Llama-3.3-70B-Instruct-Turbo": {
608
- "id": "meta-llama/Llama-3.3-70B-Instruct-Turbo",
609
- "name": "Llama 3.3 70B Turbo",
610
- "description": "Compact Llama instruction model for fast chat and local deployment",
611
- "family": "llama",
612
- "attachment": false,
883
+ "Qwen/Qwen3.8-Max": {
884
+ "id": "Qwen/Qwen3.8-Max",
885
+ "name": "Qwen3.8 Max",
886
+ "description": "2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows",
887
+ "family": "qwen",
888
+ "attachment": true,
613
889
  "reasoning": false,
614
890
  "tool_call": true,
615
891
  "structured_output": true,
616
- "release_date": "2024-12-06",
617
- "last_updated": "2024-12-06",
892
+ "temperature": true,
893
+ "release_date": "2026-08-03",
894
+ "last_updated": "2026-08-03",
618
895
  "modalities": {
619
896
  "input": [
620
- "text"
897
+ "text",
898
+ "image",
899
+ "video",
900
+ "pdf"
621
901
  ],
622
902
  "output": [
623
903
  "text"
624
904
  ]
625
905
  },
626
- "open_weights": true,
906
+ "open_weights": false,
627
907
  "limit": {
628
- "context": 131072,
629
- "output": 16384
908
+ "context": 256000,
909
+ "output": 131072
630
910
  },
631
911
  "cost": {
632
- "input": 0.1,
633
- "output": 0.32
912
+ "input": 1.65,
913
+ "output": 4.951,
914
+ "cache_read": 0.206
634
915
  }
635
916
  },
636
- "deepseek-ai/DeepSeek-R1-0528": {
637
- "id": "deepseek-ai/DeepSeek-R1-0528",
638
- "name": "DeepSeek-R1-0528",
639
- "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
917
+ "Qwen/Qwen3-32B": {
918
+ "id": "Qwen/Qwen3-32B",
919
+ "name": "Qwen3 32B",
920
+ "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding",
921
+ "family": "qwen",
640
922
  "attachment": false,
641
923
  "reasoning": true,
642
924
  "reasoning_options": [],
643
925
  "tool_call": true,
644
- "interleaved": {
645
- "field": "reasoning_content"
646
- },
647
926
  "structured_output": true,
648
927
  "temperature": true,
649
- "knowledge": "2024-07",
650
- "release_date": "2025-05-28",
651
- "last_updated": "2025-05-28",
928
+ "knowledge": "2025-04",
929
+ "release_date": "2025-04",
930
+ "last_updated": "2025-04",
652
931
  "modalities": {
653
932
  "input": [
654
933
  "text"
@@ -657,47 +936,29 @@
657
936
  "text"
658
937
  ]
659
938
  },
660
- "open_weights": false,
939
+ "open_weights": true,
661
940
  "limit": {
662
- "context": 163840,
663
- "output": 64000
941
+ "context": 40960,
942
+ "output": 16384
664
943
  },
665
944
  "cost": {
666
- "input": 0.5,
667
- "output": 2.15,
668
- "cache_read": 0.35
945
+ "input": 0.08,
946
+ "output": 0.28
669
947
  }
670
948
  },
671
- "deepseek-ai/DeepSeek-V4-Pro": {
672
- "id": "deepseek-ai/DeepSeek-V4-Pro",
673
- "name": "DeepSeek V4 Pro",
674
- "description": "Open MoE flagship with million-token context for coding and long agent runs",
675
- "family": "deepseek-thinking",
949
+ "Qwen/Qwen3-Next-80B-A3B-Instruct": {
950
+ "id": "Qwen/Qwen3-Next-80B-A3B-Instruct",
951
+ "name": "Qwen3-Next 80B-A3B Instruct",
952
+ "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
953
+ "family": "qwen",
676
954
  "attachment": false,
677
- "reasoning": true,
678
- "reasoning_options": [
679
- {
680
- "type": "toggle"
681
- },
682
- {
683
- "type": "effort",
684
- "values": [
685
- "low",
686
- "medium",
687
- "high",
688
- "xhigh"
689
- ]
690
- }
691
- ],
955
+ "reasoning": false,
692
956
  "tool_call": true,
693
- "interleaved": {
694
- "field": "reasoning_content"
695
- },
696
957
  "structured_output": true,
697
958
  "temperature": true,
698
- "knowledge": "2025-05",
699
- "release_date": "2026-04-24",
700
- "last_updated": "2026-04-24",
959
+ "knowledge": "2025-04",
960
+ "release_date": "2025-09",
961
+ "last_updated": "2025-09",
701
962
  "modalities": {
702
963
  "input": [
703
964
  "text"
@@ -708,45 +969,27 @@
708
969
  },
709
970
  "open_weights": true,
710
971
  "limit": {
711
- "context": 1048576,
712
- "output": 16384
972
+ "context": 262144,
973
+ "output": 32768
713
974
  },
714
975
  "cost": {
715
- "input": 1.3,
716
- "output": 2.6,
717
- "cache_read": 0.1
976
+ "input": 0.09,
977
+ "output": 1.1
718
978
  }
719
979
  },
720
- "deepseek-ai/DeepSeek-V4-Flash": {
721
- "id": "deepseek-ai/DeepSeek-V4-Flash",
722
- "name": "DeepSeek V4 Flash",
723
- "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
724
- "family": "deepseek-flash",
980
+ "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo": {
981
+ "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo",
982
+ "name": "Qwen3 Coder 480B A35B Instruct Turbo",
983
+ "description": "Qwen coding model for software agents, repository edits, and code reasoning",
984
+ "family": "qwen",
725
985
  "attachment": false,
726
- "reasoning": true,
727
- "reasoning_options": [
728
- {
729
- "type": "toggle"
730
- },
731
- {
732
- "type": "effort",
733
- "values": [
734
- "low",
735
- "medium",
736
- "high",
737
- "xhigh"
738
- ]
739
- }
740
- ],
986
+ "reasoning": false,
741
987
  "tool_call": true,
742
- "interleaved": {
743
- "field": "reasoning_content"
744
- },
745
988
  "structured_output": true,
746
989
  "temperature": true,
747
- "knowledge": "2025-05",
748
- "release_date": "2026-04-24",
749
- "last_updated": "2026-04-24",
990
+ "knowledge": "2025-04",
991
+ "release_date": "2025-07-23",
992
+ "last_updated": "2025-07-23",
750
993
  "modalities": {
751
994
  "input": [
752
995
  "text"
@@ -757,35 +1000,28 @@
757
1000
  },
758
1001
  "open_weights": true,
759
1002
  "limit": {
760
- "context": 1048576,
761
- "output": 16384
1003
+ "context": 262144,
1004
+ "output": 66536
762
1005
  },
763
1006
  "cost": {
764
- "input": 0.09,
765
- "output": 0.18,
766
- "cache_read": 0.018
1007
+ "input": 0.3,
1008
+ "output": 1,
1009
+ "cache_read": 0.1
767
1010
  }
768
1011
  },
769
- "deepseek-ai/DeepSeek-V3.2": {
770
- "id": "deepseek-ai/DeepSeek-V3.2",
771
- "name": "DeepSeek-V3.2",
772
- "description": "DeepSeek chat model for instruction following, coding, and analysis",
1012
+ "Qwen/Qwen3-Max": {
1013
+ "id": "Qwen/Qwen3-Max",
1014
+ "name": "Qwen3 Max",
1015
+ "description": "Flagship Qwen3 model for coding agents, complex reasoning, and tool use",
1016
+ "family": "qwen",
773
1017
  "attachment": false,
774
- "reasoning": true,
775
- "reasoning_options": [
776
- {
777
- "type": "toggle"
778
- }
779
- ],
1018
+ "reasoning": false,
780
1019
  "tool_call": true,
781
- "interleaved": {
782
- "field": "reasoning_content"
783
- },
784
1020
  "structured_output": true,
785
1021
  "temperature": true,
786
- "knowledge": "2024-12",
787
- "release_date": "2025-12-02",
788
- "last_updated": "2025-12-02",
1022
+ "knowledge": "2025-04",
1023
+ "release_date": "2025-09-23",
1024
+ "last_updated": "2025-09-23",
789
1025
  "modalities": {
790
1026
  "input": [
791
1027
  "text"
@@ -796,31 +1032,47 @@
796
1032
  },
797
1033
  "open_weights": false,
798
1034
  "limit": {
799
- "context": 163840,
800
- "output": 64000
1035
+ "context": 256000,
1036
+ "output": 65536
801
1037
  },
802
1038
  "cost": {
803
- "input": 0.26,
804
- "output": 0.38,
805
- "cache_read": 0.13
1039
+ "input": 1.2,
1040
+ "output": 6,
1041
+ "cache_read": 0.24,
1042
+ "tiers": [
1043
+ {
1044
+ "input": 2.4,
1045
+ "output": 12,
1046
+ "cache_read": 0.48,
1047
+ "tier": {
1048
+ "type": "context",
1049
+ "size": 32000
1050
+ }
1051
+ },
1052
+ {
1053
+ "input": 3,
1054
+ "output": 15,
1055
+ "cache_read": 0.6,
1056
+ "tier": {
1057
+ "type": "context",
1058
+ "size": 128000
1059
+ }
1060
+ }
1061
+ ]
806
1062
  }
807
1063
  },
808
- "MiniMaxAI/MiniMax-M2.5": {
809
- "id": "MiniMaxAI/MiniMax-M2.5",
810
- "name": "MiniMax M2.5",
811
- "description": "MiniMax model for chat, coding, office work, and agentic tasks",
1064
+ "MiniMaxAI/MiniMax-M2.7": {
1065
+ "id": "MiniMaxAI/MiniMax-M2.7",
1066
+ "name": "MiniMax-M2.7",
1067
+ "description": "Open MiniMax flagship for coding agents, office automation, and complex environments",
812
1068
  "family": "minimax",
813
1069
  "attachment": false,
814
1070
  "reasoning": true,
815
1071
  "reasoning_options": [],
816
1072
  "tool_call": true,
817
- "interleaved": {
818
- "field": "reasoning_content"
819
- },
820
1073
  "temperature": true,
821
- "knowledge": "2025-06",
822
- "release_date": "2026-02-12",
823
- "last_updated": "2026-02-12",
1074
+ "release_date": "2026-03-18",
1075
+ "last_updated": "2026-03-18",
824
1076
  "modalities": {
825
1077
  "input": [
826
1078
  "text"
@@ -834,25 +1086,28 @@
834
1086
  "context": 196608,
835
1087
  "output": 131072
836
1088
  },
837
- "status": "deprecated",
838
1089
  "cost": {
839
- "input": 0.15,
840
- "output": 1.15,
841
- "cache_read": 0.03
1090
+ "input": 0.25,
1091
+ "output": 1,
1092
+ "cache_read": 0.05
842
1093
  }
843
1094
  },
844
- "MiniMaxAI/MiniMax-M2.7": {
845
- "id": "MiniMaxAI/MiniMax-M2.7",
846
- "name": "MiniMax-M2.7",
847
- "description": "Open MiniMax flagship for coding agents, office automation, and complex environments",
1095
+ "MiniMaxAI/MiniMax-M2.5": {
1096
+ "id": "MiniMaxAI/MiniMax-M2.5",
1097
+ "name": "MiniMax M2.5",
1098
+ "description": "MiniMax model for chat, coding, office work, and agentic tasks",
848
1099
  "family": "minimax",
849
1100
  "attachment": false,
850
1101
  "reasoning": true,
851
1102
  "reasoning_options": [],
852
1103
  "tool_call": true,
1104
+ "interleaved": {
1105
+ "field": "reasoning_content"
1106
+ },
853
1107
  "temperature": true,
854
- "release_date": "2026-03-18",
855
- "last_updated": "2026-03-18",
1108
+ "knowledge": "2025-06",
1109
+ "release_date": "2026-02-12",
1110
+ "last_updated": "2026-02-12",
856
1111
  "modalities": {
857
1112
  "input": [
858
1113
  "text"
@@ -866,10 +1121,11 @@
866
1121
  "context": 196608,
867
1122
  "output": 131072
868
1123
  },
1124
+ "status": "deprecated",
869
1125
  "cost": {
870
- "input": 0.25,
871
- "output": 1,
872
- "cache_read": 0.05
1126
+ "input": 0.15,
1127
+ "output": 1.15,
1128
+ "cache_read": 0.03
873
1129
  }
874
1130
  },
875
1131
  "MiniMaxAI/MiniMax-M3": {
@@ -881,6 +1137,7 @@
881
1137
  "reasoning": true,
882
1138
  "reasoning_options": [],
883
1139
  "tool_call": true,
1140
+ "structured_output": true,
884
1141
  "temperature": true,
885
1142
  "release_date": "2026-06-01",
886
1143
  "last_updated": "2026-06-01",
@@ -905,23 +1162,18 @@
905
1162
  "cache_read": 0.06
906
1163
  }
907
1164
  },
908
- "zai-org/GLM-4.7-Flash": {
909
- "id": "zai-org/GLM-4.7-Flash",
910
- "name": "GLM-4.7-Flash",
911
- "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
912
- "family": "glm-flash",
1165
+ "deepseek-ai/DeepSeek-V3": {
1166
+ "id": "deepseek-ai/DeepSeek-V3",
1167
+ "name": "DeepSeek-V3",
1168
+ "description": "Open DeepSeek MoE chat model for coding, math, and general reasoning",
1169
+ "family": "deepseek",
913
1170
  "attachment": false,
914
- "reasoning": true,
915
- "reasoning_options": [],
916
- "tool_call": true,
917
- "interleaved": {
918
- "field": "reasoning_content"
919
- },
1171
+ "reasoning": false,
1172
+ "tool_call": true,
920
1173
  "structured_output": true,
921
1174
  "temperature": true,
922
- "knowledge": "2025-04",
923
- "release_date": "2026-01-19",
924
- "last_updated": "2026-01-19",
1175
+ "release_date": "2024-12-26",
1176
+ "last_updated": "2024-12-26",
925
1177
  "modalities": {
926
1178
  "input": [
927
1179
  "text"
@@ -932,20 +1184,19 @@
932
1184
  },
933
1185
  "open_weights": true,
934
1186
  "limit": {
935
- "context": 202752,
936
- "output": 16384
1187
+ "context": 163840,
1188
+ "output": 8192
937
1189
  },
938
1190
  "cost": {
939
- "input": 0.06,
940
- "output": 0.4,
941
- "cache_read": 0.01
1191
+ "input": 0.32,
1192
+ "output": 0.89
942
1193
  }
943
1194
  },
944
- "zai-org/GLM-4.6": {
945
- "id": "zai-org/GLM-4.6",
946
- "name": "GLM-4.6",
947
- "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
948
- "family": "glm",
1195
+ "deepseek-ai/DeepSeek-V4-Flash-0731": {
1196
+ "id": "deepseek-ai/DeepSeek-V4-Flash-0731",
1197
+ "name": "DeepSeek V4 Flash 0731",
1198
+ "description": "Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding",
1199
+ "family": "deepseek-flash",
949
1200
  "attachment": false,
950
1201
  "reasoning": true,
951
1202
  "reasoning_options": [
@@ -954,14 +1205,11 @@
954
1205
  }
955
1206
  ],
956
1207
  "tool_call": true,
957
- "interleaved": {
958
- "field": "reasoning_content"
959
- },
960
1208
  "structured_output": true,
961
1209
  "temperature": true,
962
- "knowledge": "2025-04",
963
- "release_date": "2025-09-30",
964
- "last_updated": "2025-09-30",
1210
+ "knowledge": "2025-05",
1211
+ "release_date": "2026-07-31",
1212
+ "last_updated": "2026-07-31",
965
1213
  "modalities": {
966
1214
  "input": [
967
1215
  "text"
@@ -972,44 +1220,31 @@
972
1220
  },
973
1221
  "open_weights": true,
974
1222
  "limit": {
975
- "context": 202752,
976
- "output": 131072
1223
+ "context": 1048576,
1224
+ "output": 384000
977
1225
  },
978
1226
  "cost": {
979
- "input": 0.5,
980
- "output": 2,
981
- "cache_read": 0.1
1227
+ "input": 0.09,
1228
+ "output": 0.18,
1229
+ "cache_read": 0.018
982
1230
  }
983
1231
  },
984
- "zai-org/GLM-5.2": {
985
- "id": "zai-org/GLM-5.2",
986
- "name": "GLM-5.2",
987
- "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
988
- "family": "glm",
1232
+ "deepseek-ai/DeepSeek-R1-0528": {
1233
+ "id": "deepseek-ai/DeepSeek-R1-0528",
1234
+ "name": "DeepSeek-R1-0528",
1235
+ "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
989
1236
  "attachment": false,
990
1237
  "reasoning": true,
991
- "reasoning_options": [
992
- {
993
- "type": "toggle"
994
- },
995
- {
996
- "type": "effort",
997
- "values": [
998
- "low",
999
- "medium",
1000
- "high",
1001
- "xhigh"
1002
- ]
1003
- }
1004
- ],
1238
+ "reasoning_options": [],
1005
1239
  "tool_call": true,
1006
1240
  "interleaved": {
1007
1241
  "field": "reasoning_content"
1008
1242
  },
1009
1243
  "structured_output": true,
1010
1244
  "temperature": true,
1011
- "release_date": "2026-06-13",
1012
- "last_updated": "2026-06-13",
1245
+ "knowledge": "2024-07",
1246
+ "release_date": "2025-05-28",
1247
+ "last_updated": "2025-05-28",
1013
1248
  "modalities": {
1014
1249
  "input": [
1015
1250
  "text"
@@ -1018,22 +1253,21 @@
1018
1253
  "text"
1019
1254
  ]
1020
1255
  },
1021
- "open_weights": true,
1256
+ "open_weights": false,
1022
1257
  "limit": {
1023
- "context": 1048576,
1024
- "output": 32768
1258
+ "context": 163840,
1259
+ "output": 64000
1025
1260
  },
1026
1261
  "cost": {
1027
- "input": 0.93,
1028
- "output": 3,
1029
- "cache_read": 0.18
1262
+ "input": 0.5,
1263
+ "output": 2.15,
1264
+ "cache_read": 0.35
1030
1265
  }
1031
1266
  },
1032
- "zai-org/GLM-5": {
1033
- "id": "zai-org/GLM-5",
1034
- "name": "GLM-5",
1035
- "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
1036
- "family": "glm",
1267
+ "deepseek-ai/DeepSeek-V3.2": {
1268
+ "id": "deepseek-ai/DeepSeek-V3.2",
1269
+ "name": "DeepSeek-V3.2",
1270
+ "description": "DeepSeek chat model for instruction following, coding, and analysis",
1037
1271
  "attachment": false,
1038
1272
  "reasoning": true,
1039
1273
  "reasoning_options": [
@@ -1047,9 +1281,9 @@
1047
1281
  },
1048
1282
  "structured_output": true,
1049
1283
  "temperature": true,
1050
- "knowledge": "2025-12",
1051
- "release_date": "2026-02-12",
1052
- "last_updated": "2026-02-12",
1284
+ "knowledge": "2024-12",
1285
+ "release_date": "2025-12-02",
1286
+ "last_updated": "2025-12-02",
1053
1287
  "modalities": {
1054
1288
  "input": [
1055
1289
  "text"
@@ -1058,22 +1292,22 @@
1058
1292
  "text"
1059
1293
  ]
1060
1294
  },
1061
- "open_weights": true,
1295
+ "open_weights": false,
1062
1296
  "limit": {
1063
- "context": 202752,
1064
- "output": 16384
1297
+ "context": 163840,
1298
+ "output": 64000
1065
1299
  },
1066
1300
  "cost": {
1067
- "input": 0.6,
1068
- "output": 2.08,
1069
- "cache_read": 0.12
1301
+ "input": 0.26,
1302
+ "output": 0.38,
1303
+ "cache_read": 0.13
1070
1304
  }
1071
1305
  },
1072
- "zai-org/GLM-4.7": {
1073
- "id": "zai-org/GLM-4.7",
1074
- "name": "GLM-4.7",
1075
- "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
1076
- "family": "glm",
1306
+ "deepseek-ai/DeepSeek-V3.1": {
1307
+ "id": "deepseek-ai/DeepSeek-V3.1",
1308
+ "name": "DeepSeek-V3.1",
1309
+ "description": "Hybrid-reasoning DeepSeek model with thinking and non-thinking modes",
1310
+ "family": "deepseek",
1077
1311
  "attachment": false,
1078
1312
  "reasoning": true,
1079
1313
  "reasoning_options": [
@@ -1082,14 +1316,10 @@
1082
1316
  }
1083
1317
  ],
1084
1318
  "tool_call": true,
1085
- "interleaved": {
1086
- "field": "reasoning_content"
1087
- },
1088
1319
  "structured_output": true,
1089
1320
  "temperature": true,
1090
- "knowledge": "2025-04",
1091
- "release_date": "2025-12-22",
1092
- "last_updated": "2025-12-22",
1321
+ "release_date": "2025-08-21",
1322
+ "last_updated": "2025-08-21",
1093
1323
  "modalities": {
1094
1324
  "input": [
1095
1325
  "text"
@@ -1100,25 +1330,34 @@
1100
1330
  },
1101
1331
  "open_weights": true,
1102
1332
  "limit": {
1103
- "context": 202752,
1104
- "output": 16384
1333
+ "context": 163840,
1334
+ "output": 8192
1105
1335
  },
1106
1336
  "cost": {
1107
- "input": 0.4,
1108
- "output": 1.75,
1109
- "cache_read": 0.08
1337
+ "input": 0.25,
1338
+ "output": 0.95,
1339
+ "cache_read": 0.13
1110
1340
  }
1111
1341
  },
1112
- "zai-org/GLM-5.1": {
1113
- "id": "zai-org/GLM-5.1",
1114
- "name": "GLM-5.1",
1115
- "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
1116
- "family": "glm",
1342
+ "deepseek-ai/DeepSeek-V4-Flash": {
1343
+ "id": "deepseek-ai/DeepSeek-V4-Flash",
1344
+ "name": "DeepSeek V4 Flash",
1345
+ "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
1346
+ "family": "deepseek-flash",
1117
1347
  "attachment": false,
1118
1348
  "reasoning": true,
1119
1349
  "reasoning_options": [
1120
1350
  {
1121
1351
  "type": "toggle"
1352
+ },
1353
+ {
1354
+ "type": "effort",
1355
+ "values": [
1356
+ "low",
1357
+ "medium",
1358
+ "high",
1359
+ "xhigh"
1360
+ ]
1122
1361
  }
1123
1362
  ],
1124
1363
  "tool_call": true,
@@ -1127,9 +1366,9 @@
1127
1366
  },
1128
1367
  "structured_output": true,
1129
1368
  "temperature": true,
1130
- "knowledge": "2025-04",
1131
- "release_date": "2026-04-07",
1132
- "last_updated": "2026-04-07",
1369
+ "knowledge": "2025-05",
1370
+ "release_date": "2026-04-24",
1371
+ "last_updated": "2026-04-24",
1133
1372
  "modalities": {
1134
1373
  "input": [
1135
1374
  "text"
@@ -1140,25 +1379,34 @@
1140
1379
  },
1141
1380
  "open_weights": true,
1142
1381
  "limit": {
1143
- "context": 202752,
1382
+ "context": 1048576,
1144
1383
  "output": 16384
1145
1384
  },
1146
1385
  "cost": {
1147
- "input": 1.05,
1148
- "output": 3.5,
1149
- "cache_read": 0.205
1386
+ "input": 0.09,
1387
+ "output": 0.18,
1388
+ "cache_read": 0.018
1150
1389
  }
1151
1390
  },
1152
- "moonshotai/Kimi-K2.6": {
1153
- "id": "moonshotai/Kimi-K2.6",
1154
- "name": "Kimi K2.6",
1155
- "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
1156
- "family": "kimi-k2",
1157
- "attachment": true,
1391
+ "deepseek-ai/DeepSeek-V4-Pro": {
1392
+ "id": "deepseek-ai/DeepSeek-V4-Pro",
1393
+ "name": "DeepSeek V4 Pro",
1394
+ "description": "Open MoE flagship with million-token context for coding and long agent runs",
1395
+ "family": "deepseek-thinking",
1396
+ "attachment": false,
1158
1397
  "reasoning": true,
1159
1398
  "reasoning_options": [
1160
1399
  {
1161
1400
  "type": "toggle"
1401
+ },
1402
+ {
1403
+ "type": "effort",
1404
+ "values": [
1405
+ "low",
1406
+ "medium",
1407
+ "high",
1408
+ "xhigh"
1409
+ ]
1162
1410
  }
1163
1411
  ],
1164
1412
  "tool_call": true,
@@ -1167,14 +1415,12 @@
1167
1415
  },
1168
1416
  "structured_output": true,
1169
1417
  "temperature": true,
1170
- "knowledge": "2024-04",
1171
- "release_date": "2026-04-21",
1172
- "last_updated": "2026-04-21",
1418
+ "knowledge": "2025-05",
1419
+ "release_date": "2026-04-24",
1420
+ "last_updated": "2026-04-24",
1173
1421
  "modalities": {
1174
1422
  "input": [
1175
- "text",
1176
- "image",
1177
- "video"
1423
+ "text"
1178
1424
  ],
1179
1425
  "output": [
1180
1426
  "text"
@@ -1182,36 +1428,27 @@
1182
1428
  },
1183
1429
  "open_weights": true,
1184
1430
  "limit": {
1185
- "context": 262144,
1431
+ "context": 1048576,
1186
1432
  "output": 16384
1187
1433
  },
1188
1434
  "cost": {
1189
- "input": 0.75,
1190
- "output": 3.5,
1191
- "cache_read": 0.15
1435
+ "input": 1.3,
1436
+ "output": 2.6,
1437
+ "cache_read": 0.1
1192
1438
  }
1193
1439
  },
1194
- "moonshotai/Kimi-K2.5": {
1195
- "id": "moonshotai/Kimi-K2.5",
1196
- "name": "Kimi K2.5",
1197
- "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
1198
- "family": "kimi-k2",
1440
+ "stepfun-ai/Step-3.7-Flash": {
1441
+ "id": "stepfun-ai/Step-3.7-Flash",
1442
+ "name": "Step 3.7 Flash",
1443
+ "description": "Newer StepFun flash model for faster agents, coding, and multimodal prompts",
1199
1444
  "attachment": true,
1200
1445
  "reasoning": true,
1201
- "reasoning_options": [
1202
- {
1203
- "type": "toggle"
1204
- }
1205
- ],
1446
+ "reasoning_options": [],
1206
1447
  "tool_call": true,
1207
- "interleaved": {
1208
- "field": "reasoning_content"
1209
- },
1210
- "structured_output": true,
1211
1448
  "temperature": true,
1212
- "knowledge": "2025-01",
1213
- "release_date": "2026-01-27",
1214
- "last_updated": "2026-01-27",
1449
+ "knowledge": "2026-03-01",
1450
+ "release_date": "2026-05-29",
1451
+ "last_updated": "2026-05-29",
1215
1452
  "modalities": {
1216
1453
  "input": [
1217
1454
  "text",
@@ -1225,18 +1462,18 @@
1225
1462
  "open_weights": true,
1226
1463
  "limit": {
1227
1464
  "context": 262144,
1228
- "output": 32768
1465
+ "output": 256000
1229
1466
  },
1230
1467
  "cost": {
1231
- "input": 0.45,
1232
- "output": 2.25,
1233
- "cache_read": 0.07
1468
+ "input": 0.2,
1469
+ "output": 1.15,
1470
+ "cache_read": 0.04
1234
1471
  }
1235
1472
  },
1236
- "moonshotai/Kimi-K2.7-Code": {
1237
- "id": "moonshotai/Kimi-K2.7-Code",
1238
- "name": "Kimi K2.7 Code",
1239
- "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
1473
+ "moonshotai/Kimi-K2.6": {
1474
+ "id": "moonshotai/Kimi-K2.6",
1475
+ "name": "Kimi K2.6",
1476
+ "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
1240
1477
  "family": "kimi-k2",
1241
1478
  "attachment": true,
1242
1479
  "reasoning": true,
@@ -1250,10 +1487,10 @@
1250
1487
  "field": "reasoning_content"
1251
1488
  },
1252
1489
  "structured_output": true,
1253
- "temperature": false,
1254
- "knowledge": "2025-01",
1255
- "release_date": "2026-06-12",
1256
- "last_updated": "2026-06-12",
1490
+ "temperature": true,
1491
+ "knowledge": "2024-04",
1492
+ "release_date": "2026-04-21",
1493
+ "last_updated": "2026-04-21",
1257
1494
  "modalities": {
1258
1495
  "input": [
1259
1496
  "text",
@@ -1267,19 +1504,19 @@
1267
1504
  "open_weights": true,
1268
1505
  "limit": {
1269
1506
  "context": 262144,
1270
- "output": 262144
1507
+ "output": 16384
1271
1508
  },
1272
1509
  "cost": {
1273
- "input": 0.74,
1510
+ "input": 0.75,
1274
1511
  "output": 3.5,
1275
1512
  "cache_read": 0.15
1276
1513
  }
1277
1514
  },
1278
- "XiaomiMiMo/MiMo-V2.5-Pro": {
1279
- "id": "XiaomiMiMo/MiMo-V2.5-Pro",
1280
- "name": "MiMo-V2.5-Pro",
1281
- "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
1282
- "family": "mimo",
1515
+ "moonshotai/Kimi-K2.5": {
1516
+ "id": "moonshotai/Kimi-K2.5",
1517
+ "name": "Kimi K2.5",
1518
+ "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
1519
+ "family": "kimi-k2",
1283
1520
  "attachment": true,
1284
1521
  "reasoning": true,
1285
1522
  "reasoning_options": [
@@ -1293,13 +1530,14 @@
1293
1530
  },
1294
1531
  "structured_output": true,
1295
1532
  "temperature": true,
1296
- "knowledge": "2024-12",
1297
- "release_date": "2026-04-22",
1298
- "last_updated": "2026-04-22",
1533
+ "knowledge": "2025-01",
1534
+ "release_date": "2026-01-27",
1535
+ "last_updated": "2026-01-27",
1299
1536
  "modalities": {
1300
1537
  "input": [
1301
1538
  "text",
1302
- "audio"
1539
+ "image",
1540
+ "video"
1303
1541
  ],
1304
1542
  "output": [
1305
1543
  "text"
@@ -1307,20 +1545,20 @@
1307
1545
  },
1308
1546
  "open_weights": true,
1309
1547
  "limit": {
1310
- "context": 1048576,
1311
- "output": 16384
1548
+ "context": 262144,
1549
+ "output": 32768
1312
1550
  },
1313
1551
  "cost": {
1314
- "input": 1,
1315
- "output": 3,
1316
- "cache_read": 0.2
1552
+ "input": 0.45,
1553
+ "output": 2.25,
1554
+ "cache_read": 0.07
1317
1555
  }
1318
1556
  },
1319
- "XiaomiMiMo/MiMo-V2.5": {
1320
- "id": "XiaomiMiMo/MiMo-V2.5",
1321
- "name": "MiMo-V2.5",
1322
- "description": "Open MiMo model for multimodal coding agents and long-context automation",
1323
- "family": "mimo",
1557
+ "moonshotai/Kimi-K2.7-Code": {
1558
+ "id": "moonshotai/Kimi-K2.7-Code",
1559
+ "name": "Kimi K2.7 Code",
1560
+ "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
1561
+ "family": "kimi-k2",
1324
1562
  "attachment": true,
1325
1563
  "reasoning": true,
1326
1564
  "reasoning_options": [
@@ -1333,15 +1571,14 @@
1333
1571
  "field": "reasoning_content"
1334
1572
  },
1335
1573
  "structured_output": true,
1336
- "temperature": true,
1337
- "knowledge": "2024-12",
1338
- "release_date": "2026-04-22",
1339
- "last_updated": "2026-04-22",
1574
+ "temperature": false,
1575
+ "knowledge": "2025-01",
1576
+ "release_date": "2026-06-12",
1577
+ "last_updated": "2026-06-12",
1340
1578
  "modalities": {
1341
1579
  "input": [
1342
1580
  "text",
1343
1581
  "image",
1344
- "audio",
1345
1582
  "video"
1346
1583
  ],
1347
1584
  "output": [
@@ -1351,31 +1588,36 @@
1351
1588
  "open_weights": true,
1352
1589
  "limit": {
1353
1590
  "context": 262144,
1354
- "output": 16384
1591
+ "output": 262144
1355
1592
  },
1356
1593
  "cost": {
1357
- "input": 0.4,
1358
- "output": 2,
1359
- "cache_read": 0.08
1594
+ "input": 0.74,
1595
+ "output": 3.5,
1596
+ "cache_read": 0.15
1360
1597
  }
1361
1598
  },
1362
- "google/gemma-4-26B-A4B-it": {
1363
- "id": "google/gemma-4-26B-A4B-it",
1364
- "name": "Gemma 4 26B A4B IT",
1365
- "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
1366
- "family": "gemma",
1599
+ "moonshotai/Kimi-K3": {
1600
+ "id": "moonshotai/Kimi-K3",
1601
+ "name": "Kimi K3",
1602
+ "description": "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work",
1603
+ "family": "kimi-k3",
1367
1604
  "attachment": true,
1368
1605
  "reasoning": true,
1369
1606
  "reasoning_options": [
1370
1607
  {
1371
- "type": "toggle"
1608
+ "type": "effort",
1609
+ "values": [
1610
+ "low",
1611
+ "high",
1612
+ "max"
1613
+ ]
1372
1614
  }
1373
1615
  ],
1374
1616
  "tool_call": true,
1375
1617
  "structured_output": true,
1376
- "temperature": true,
1377
- "release_date": "2026-04-02",
1378
- "last_updated": "2026-04-02",
1618
+ "temperature": false,
1619
+ "release_date": "2026-07-16",
1620
+ "last_updated": "2026-07-16",
1379
1621
  "modalities": {
1380
1622
  "input": [
1381
1623
  "text",
@@ -1387,36 +1629,40 @@
1387
1629
  },
1388
1630
  "open_weights": true,
1389
1631
  "limit": {
1390
- "context": 262144,
1391
- "output": 32768
1632
+ "context": 1048576,
1633
+ "output": 131072
1392
1634
  },
1393
1635
  "cost": {
1394
- "input": 0.07,
1395
- "output": 0.34
1636
+ "input": 2.7,
1637
+ "output": 13.5,
1638
+ "cache_read": 0.27
1396
1639
  }
1397
1640
  },
1398
- "google/gemma-4-31B-it": {
1399
- "id": "google/gemma-4-31B-it",
1400
- "name": "Gemma 4 31B IT",
1401
- "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
1402
- "family": "gemma",
1403
- "attachment": true,
1641
+ "openai/gpt-oss-20b": {
1642
+ "id": "openai/gpt-oss-20b",
1643
+ "name": "GPT OSS 20B",
1644
+ "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
1645
+ "family": "gpt-oss",
1646
+ "attachment": false,
1404
1647
  "reasoning": true,
1405
1648
  "reasoning_options": [
1406
1649
  {
1407
- "type": "toggle"
1650
+ "type": "effort",
1651
+ "values": [
1652
+ "low",
1653
+ "medium",
1654
+ "high"
1655
+ ]
1408
1656
  }
1409
1657
  ],
1410
1658
  "tool_call": true,
1411
1659
  "structured_output": true,
1412
1660
  "temperature": true,
1413
- "release_date": "2026-04-02",
1414
- "last_updated": "2026-04-02",
1661
+ "release_date": "2025-08-05",
1662
+ "last_updated": "2025-08-05",
1415
1663
  "modalities": {
1416
1664
  "input": [
1417
- "text",
1418
- "image",
1419
- "video"
1665
+ "text"
1420
1666
  ],
1421
1667
  "output": [
1422
1668
  "text"
@@ -1424,12 +1670,12 @@
1424
1670
  },
1425
1671
  "open_weights": true,
1426
1672
  "limit": {
1427
- "context": 262144,
1428
- "output": 32768
1673
+ "context": 131072,
1674
+ "output": 16384
1429
1675
  },
1430
1676
  "cost": {
1431
- "input": 0.13,
1432
- "output": 0.38
1677
+ "input": 0.03,
1678
+ "output": 0.14
1433
1679
  }
1434
1680
  },
1435
1681
  "openai/gpt-oss-120b": {
@@ -1472,31 +1718,163 @@
1472
1718
  "output": 0.17
1473
1719
  }
1474
1720
  },
1475
- "openai/gpt-oss-20b": {
1476
- "id": "openai/gpt-oss-20b",
1477
- "name": "GPT OSS 20B",
1478
- "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
1479
- "family": "gpt-oss",
1721
+ "meta-llama/Llama-4-Scout-17B-16E-Instruct": {
1722
+ "id": "meta-llama/Llama-4-Scout-17B-16E-Instruct",
1723
+ "name": "Llama 4 Scout 17B",
1724
+ "description": "Open multimodal Llama model for long-context analysis and efficient agents",
1725
+ "family": "llama",
1726
+ "attachment": true,
1727
+ "reasoning": false,
1728
+ "tool_call": true,
1729
+ "structured_output": true,
1730
+ "release_date": "2025-04-05",
1731
+ "last_updated": "2025-04-05",
1732
+ "modalities": {
1733
+ "input": [
1734
+ "text",
1735
+ "image"
1736
+ ],
1737
+ "output": [
1738
+ "text"
1739
+ ]
1740
+ },
1741
+ "open_weights": true,
1742
+ "limit": {
1743
+ "context": 327680,
1744
+ "output": 16384
1745
+ },
1746
+ "cost": {
1747
+ "input": 0.1,
1748
+ "output": 0.3
1749
+ }
1750
+ },
1751
+ "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8": {
1752
+ "id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
1753
+ "name": "Llama 4 Maverick 17B FP8",
1754
+ "description": "Open multimodal Llama model for strong reasoning and fast responses",
1755
+ "family": "llama",
1756
+ "attachment": true,
1757
+ "reasoning": false,
1758
+ "tool_call": false,
1759
+ "structured_output": true,
1760
+ "release_date": "2025-04-05",
1761
+ "last_updated": "2025-04-05",
1762
+ "modalities": {
1763
+ "input": [
1764
+ "text",
1765
+ "image"
1766
+ ],
1767
+ "output": [
1768
+ "text"
1769
+ ]
1770
+ },
1771
+ "open_weights": true,
1772
+ "limit": {
1773
+ "context": 1048576,
1774
+ "output": 16384
1775
+ },
1776
+ "cost": {
1777
+ "input": 0.2,
1778
+ "output": 0.8
1779
+ }
1780
+ },
1781
+ "meta-llama/Llama-3.3-70B-Instruct-Turbo": {
1782
+ "id": "meta-llama/Llama-3.3-70B-Instruct-Turbo",
1783
+ "name": "Llama 3.3 70B Turbo",
1784
+ "description": "Compact Llama instruction model for fast chat and local deployment",
1785
+ "family": "llama",
1480
1786
  "attachment": false,
1787
+ "reasoning": false,
1788
+ "tool_call": true,
1789
+ "structured_output": true,
1790
+ "release_date": "2024-12-06",
1791
+ "last_updated": "2024-12-06",
1792
+ "modalities": {
1793
+ "input": [
1794
+ "text"
1795
+ ],
1796
+ "output": [
1797
+ "text"
1798
+ ]
1799
+ },
1800
+ "open_weights": true,
1801
+ "limit": {
1802
+ "context": 131072,
1803
+ "output": 16384
1804
+ },
1805
+ "cost": {
1806
+ "input": 0.1,
1807
+ "output": 0.32
1808
+ }
1809
+ },
1810
+ "XiaomiMiMo/MiMo-V2.5-Pro": {
1811
+ "id": "XiaomiMiMo/MiMo-V2.5-Pro",
1812
+ "name": "MiMo-V2.5-Pro",
1813
+ "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
1814
+ "family": "mimo",
1815
+ "attachment": true,
1481
1816
  "reasoning": true,
1482
1817
  "reasoning_options": [
1483
1818
  {
1484
- "type": "effort",
1485
- "values": [
1486
- "low",
1487
- "medium",
1488
- "high"
1489
- ]
1819
+ "type": "toggle"
1490
1820
  }
1491
1821
  ],
1492
1822
  "tool_call": true,
1823
+ "interleaved": {
1824
+ "field": "reasoning_content"
1825
+ },
1493
1826
  "structured_output": true,
1494
1827
  "temperature": true,
1495
- "release_date": "2025-08-05",
1496
- "last_updated": "2025-08-05",
1828
+ "knowledge": "2024-12",
1829
+ "release_date": "2026-04-22",
1830
+ "last_updated": "2026-04-22",
1497
1831
  "modalities": {
1498
1832
  "input": [
1833
+ "text",
1834
+ "audio"
1835
+ ],
1836
+ "output": [
1499
1837
  "text"
1838
+ ]
1839
+ },
1840
+ "open_weights": true,
1841
+ "limit": {
1842
+ "context": 1048576,
1843
+ "output": 16384
1844
+ },
1845
+ "cost": {
1846
+ "input": 1,
1847
+ "output": 3,
1848
+ "cache_read": 0.2
1849
+ }
1850
+ },
1851
+ "XiaomiMiMo/MiMo-V2.5": {
1852
+ "id": "XiaomiMiMo/MiMo-V2.5",
1853
+ "name": "MiMo-V2.5",
1854
+ "description": "Open MiMo model for multimodal coding agents and long-context automation",
1855
+ "family": "mimo",
1856
+ "attachment": true,
1857
+ "reasoning": true,
1858
+ "reasoning_options": [
1859
+ {
1860
+ "type": "toggle"
1861
+ }
1862
+ ],
1863
+ "tool_call": true,
1864
+ "interleaved": {
1865
+ "field": "reasoning_content"
1866
+ },
1867
+ "structured_output": true,
1868
+ "temperature": true,
1869
+ "knowledge": "2024-12",
1870
+ "release_date": "2026-04-22",
1871
+ "last_updated": "2026-04-22",
1872
+ "modalities": {
1873
+ "input": [
1874
+ "text",
1875
+ "image",
1876
+ "audio",
1877
+ "video"
1500
1878
  ],
1501
1879
  "output": [
1502
1880
  "text"
@@ -1504,12 +1882,13 @@
1504
1882
  },
1505
1883
  "open_weights": true,
1506
1884
  "limit": {
1507
- "context": 131072,
1885
+ "context": 262144,
1508
1886
  "output": 16384
1509
1887
  },
1510
1888
  "cost": {
1511
- "input": 0.03,
1512
- "output": 0.14
1889
+ "input": 0.4,
1890
+ "output": 2,
1891
+ "cache_read": 0.08
1513
1892
  }
1514
1893
  }
1515
1894
  }