llm.rb 12.4.0 → 12.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
data/data/deepinfra.json CHANGED
@@ -7,21 +7,24 @@
7
7
  "name": "Deep Infra",
8
8
  "doc": "https://deepinfra.com/models",
9
9
  "models": {
10
- "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8": {
11
- "id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
12
- "name": "Llama 4 Maverick 17B FP8",
13
- "description": "Open multimodal Llama model for strong reasoning and fast responses",
14
- "family": "llama",
10
+ "Qwen/Qwen3.5-9B": {
11
+ "id": "Qwen/Qwen3.5-9B",
12
+ "name": "Qwen3.5 9B",
13
+ "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
14
+ "family": "qwen",
15
15
  "attachment": true,
16
- "reasoning": false,
17
- "tool_call": false,
16
+ "reasoning": true,
17
+ "reasoning_options": [],
18
+ "tool_call": true,
18
19
  "structured_output": true,
19
- "release_date": "2025-04-05",
20
- "last_updated": "2025-04-05",
20
+ "temperature": true,
21
+ "release_date": "2026-02-23",
22
+ "last_updated": "2026-02-23",
21
23
  "modalities": {
22
24
  "input": [
23
25
  "text",
24
- "image"
26
+ "image",
27
+ "video"
25
28
  ],
26
29
  "output": [
27
30
  "text"
@@ -29,29 +32,33 @@
29
32
  },
30
33
  "open_weights": true,
31
34
  "limit": {
32
- "context": 1048576,
33
- "output": 16384
35
+ "context": 262144,
36
+ "output": 65536
34
37
  },
35
38
  "cost": {
36
- "input": 0.2,
37
- "output": 0.8
39
+ "input": 0.1,
40
+ "output": 0.15
38
41
  }
39
42
  },
40
- "meta-llama/Llama-4-Scout-17B-16E-Instruct": {
41
- "id": "meta-llama/Llama-4-Scout-17B-16E-Instruct",
42
- "name": "Llama 4 Scout 17B",
43
- "description": "Open multimodal Llama model for long-context analysis and efficient agents",
44
- "family": "llama",
43
+ "Qwen/Qwen3.5-35B-A3B": {
44
+ "id": "Qwen/Qwen3.5-35B-A3B",
45
+ "name": "Qwen 3.5 35B A3B",
46
+ "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
47
+ "family": "qwen",
45
48
  "attachment": true,
46
- "reasoning": false,
49
+ "reasoning": true,
50
+ "reasoning_options": [],
47
51
  "tool_call": true,
48
52
  "structured_output": true,
49
- "release_date": "2025-04-05",
50
- "last_updated": "2025-04-05",
53
+ "temperature": true,
54
+ "knowledge": "2025-01",
55
+ "release_date": "2026-02-01",
56
+ "last_updated": "2026-04-20",
51
57
  "modalities": {
52
58
  "input": [
53
59
  "text",
54
- "image"
60
+ "image",
61
+ "video"
55
62
  ],
56
63
  "output": [
57
64
  "text"
@@ -59,25 +66,28 @@
59
66
  },
60
67
  "open_weights": true,
61
68
  "limit": {
62
- "context": 327680,
63
- "output": 16384
69
+ "context": 262144,
70
+ "output": 81920
64
71
  },
65
72
  "cost": {
66
- "input": 0.1,
67
- "output": 0.3
73
+ "input": 0.14,
74
+ "output": 1,
75
+ "cache_read": 0.05
68
76
  }
69
77
  },
70
- "meta-llama/Llama-3.3-70B-Instruct-Turbo": {
71
- "id": "meta-llama/Llama-3.3-70B-Instruct-Turbo",
72
- "name": "Llama 3.3 70B Turbo",
73
- "description": "Compact Llama instruction model for fast chat and local deployment",
74
- "family": "llama",
78
+ "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo": {
79
+ "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo",
80
+ "name": "Qwen3 Coder 480B A35B Instruct Turbo",
81
+ "description": "Qwen coding model for software agents, repository edits, and code reasoning",
82
+ "family": "qwen",
75
83
  "attachment": false,
76
84
  "reasoning": false,
77
85
  "tool_call": true,
78
86
  "structured_output": true,
79
- "release_date": "2024-12-06",
80
- "last_updated": "2024-12-06",
87
+ "temperature": true,
88
+ "knowledge": "2025-04",
89
+ "release_date": "2025-07-23",
90
+ "last_updated": "2025-07-23",
81
91
  "modalities": {
82
92
  "input": [
83
93
  "text"
@@ -88,40 +98,32 @@
88
98
  },
89
99
  "open_weights": true,
90
100
  "limit": {
91
- "context": 131072,
92
- "output": 16384
101
+ "context": 262144,
102
+ "output": 66536
93
103
  },
94
104
  "cost": {
95
- "input": 0.1,
96
- "output": 0.32
105
+ "input": 0.3,
106
+ "output": 1,
107
+ "cache_read": 0.1
97
108
  }
98
109
  },
99
- "moonshotai/Kimi-K2.6": {
100
- "id": "moonshotai/Kimi-K2.6",
101
- "name": "Kimi K2.6",
102
- "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
103
- "family": "kimi-k2",
104
- "attachment": true,
110
+ "Qwen/Qwen3-32B": {
111
+ "id": "Qwen/Qwen3-32B",
112
+ "name": "Qwen3 32B",
113
+ "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding",
114
+ "family": "qwen",
115
+ "attachment": false,
105
116
  "reasoning": true,
106
- "reasoning_options": [
107
- {
108
- "type": "toggle"
109
- }
110
- ],
117
+ "reasoning_options": [],
111
118
  "tool_call": true,
112
- "interleaved": {
113
- "field": "reasoning_content"
114
- },
115
119
  "structured_output": true,
116
120
  "temperature": true,
117
- "knowledge": "2024-04",
118
- "release_date": "2026-04-21",
119
- "last_updated": "2026-04-21",
121
+ "knowledge": "2025-04",
122
+ "release_date": "2025-04",
123
+ "last_updated": "2025-04",
120
124
  "modalities": {
121
125
  "input": [
122
- "text",
123
- "image",
124
- "video"
126
+ "text"
125
127
  ],
126
128
  "output": [
127
129
  "text"
@@ -129,36 +131,28 @@
129
131
  },
130
132
  "open_weights": true,
131
133
  "limit": {
132
- "context": 262144,
134
+ "context": 40960,
133
135
  "output": 16384
134
136
  },
135
137
  "cost": {
136
- "input": 0.75,
137
- "output": 3.5,
138
- "cache_read": 0.15
138
+ "input": 0.08,
139
+ "output": 0.28
139
140
  }
140
141
  },
141
- "moonshotai/Kimi-K2.5": {
142
- "id": "moonshotai/Kimi-K2.5",
143
- "name": "Kimi K2.5",
144
- "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
145
- "family": "kimi-k2",
142
+ "Qwen/Qwen3.5-397B-A17B": {
143
+ "id": "Qwen/Qwen3.5-397B-A17B",
144
+ "name": "Qwen 3.5 397B A17B",
145
+ "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
146
+ "family": "qwen",
146
147
  "attachment": true,
147
148
  "reasoning": true,
148
- "reasoning_options": [
149
- {
150
- "type": "toggle"
151
- }
152
- ],
149
+ "reasoning_options": [],
153
150
  "tool_call": true,
154
- "interleaved": {
155
- "field": "reasoning_content"
156
- },
157
151
  "structured_output": true,
158
152
  "temperature": true,
159
153
  "knowledge": "2025-01",
160
- "release_date": "2026-01-27",
161
- "last_updated": "2026-01-27",
154
+ "release_date": "2026-02-01",
155
+ "last_updated": "2026-04-20",
162
156
  "modalities": {
163
157
  "input": [
164
158
  "text",
@@ -172,40 +166,33 @@
172
166
  "open_weights": true,
173
167
  "limit": {
174
168
  "context": 262144,
175
- "output": 32768
169
+ "output": 81920
176
170
  },
177
171
  "cost": {
178
172
  "input": 0.45,
179
- "output": 2.25,
180
- "cache_read": 0.07
173
+ "output": 3,
174
+ "cache_read": 0.22
181
175
  }
182
176
  },
183
- "moonshotai/Kimi-K2.7-Code": {
184
- "id": "moonshotai/Kimi-K2.7-Code",
185
- "name": "Kimi K2.7 Code",
186
- "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
187
- "family": "kimi-k2",
177
+ "Qwen/Qwen3.5-27B": {
178
+ "id": "Qwen/Qwen3.5-27B",
179
+ "name": "Qwen3.5 27B",
180
+ "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
181
+ "family": "qwen",
188
182
  "attachment": true,
189
183
  "reasoning": true,
190
- "reasoning_options": [
191
- {
192
- "type": "toggle"
193
- }
194
- ],
184
+ "reasoning_options": [],
195
185
  "tool_call": true,
196
- "interleaved": {
197
- "field": "reasoning_content"
198
- },
199
186
  "structured_output": true,
200
- "temperature": false,
201
- "knowledge": "2025-01",
202
- "release_date": "2026-06-12",
203
- "last_updated": "2026-06-12",
187
+ "temperature": true,
188
+ "release_date": "2026-02-23",
189
+ "last_updated": "2026-02-23",
204
190
  "modalities": {
205
191
  "input": [
206
192
  "text",
207
193
  "image",
208
- "video"
194
+ "video",
195
+ "audio"
209
196
  ],
210
197
  "output": [
211
198
  "text"
@@ -214,36 +201,32 @@
214
201
  "open_weights": true,
215
202
  "limit": {
216
203
  "context": 262144,
217
- "output": 262144
204
+ "output": 65536
218
205
  },
219
206
  "cost": {
220
- "input": 0.74,
221
- "output": 3.5,
222
- "cache_read": 0.15
207
+ "input": 0.26,
208
+ "output": 2.6
223
209
  }
224
210
  },
225
- "google/gemma-4-31B-it": {
226
- "id": "google/gemma-4-31B-it",
227
- "name": "Gemma 4 31B IT",
228
- "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
229
- "family": "gemma",
211
+ "Qwen/Qwen3.6-27B": {
212
+ "id": "Qwen/Qwen3.6-27B",
213
+ "name": "Qwen3.6 27B",
214
+ "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
215
+ "family": "qwen",
230
216
  "attachment": true,
231
217
  "reasoning": true,
232
- "reasoning_options": [
233
- {
234
- "type": "toggle"
235
- }
236
- ],
218
+ "reasoning_options": [],
237
219
  "tool_call": true,
238
220
  "structured_output": true,
239
221
  "temperature": true,
240
- "release_date": "2026-04-02",
241
- "last_updated": "2026-04-02",
222
+ "release_date": "2026-04-22",
223
+ "last_updated": "2026-04-22",
242
224
  "modalities": {
243
225
  "input": [
244
226
  "text",
245
227
  "image",
246
- "video"
228
+ "video",
229
+ "audio"
247
230
  ],
248
231
  "output": [
249
232
  "text"
@@ -252,34 +235,32 @@
252
235
  "open_weights": true,
253
236
  "limit": {
254
237
  "context": 262144,
255
- "output": 32768
238
+ "output": 65536
256
239
  },
257
240
  "cost": {
258
- "input": 0.13,
259
- "output": 0.38
241
+ "input": 0.32,
242
+ "output": 3.2
260
243
  }
261
244
  },
262
- "google/gemma-4-26B-A4B-it": {
263
- "id": "google/gemma-4-26B-A4B-it",
264
- "name": "Gemma 4 26B A4B IT",
265
- "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
266
- "family": "gemma",
245
+ "Qwen/Qwen3.5-122B-A10B": {
246
+ "id": "Qwen/Qwen3.5-122B-A10B",
247
+ "name": "Qwen3.5 122B-A10B",
248
+ "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
249
+ "family": "qwen",
267
250
  "attachment": true,
268
251
  "reasoning": true,
269
- "reasoning_options": [
270
- {
271
- "type": "toggle"
272
- }
273
- ],
252
+ "reasoning_options": [],
274
253
  "tool_call": true,
275
254
  "structured_output": true,
276
255
  "temperature": true,
277
- "release_date": "2026-04-02",
278
- "last_updated": "2026-04-02",
256
+ "release_date": "2026-02-23",
257
+ "last_updated": "2026-02-23",
279
258
  "modalities": {
280
259
  "input": [
281
260
  "text",
282
- "image"
261
+ "image",
262
+ "video",
263
+ "audio"
283
264
  ],
284
265
  "output": [
285
266
  "text"
@@ -288,31 +269,29 @@
288
269
  "open_weights": true,
289
270
  "limit": {
290
271
  "context": 262144,
291
- "output": 32768
272
+ "output": 65536
292
273
  },
293
274
  "cost": {
294
- "input": 0.07,
295
- "output": 0.34
275
+ "input": 0.29,
276
+ "output": 2.4
296
277
  }
297
278
  },
298
- "Qwen/Qwen3.6-35B-A3B": {
299
- "id": "Qwen/Qwen3.6-35B-A3B",
300
- "name": "Qwen3.6 35B A3B",
301
- "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
302
- "family": "qwen",
303
- "attachment": true,
304
- "reasoning": true,
305
- "reasoning_options": [],
279
+ "Qwen/Qwen3-Next-80B-A3B-Instruct": {
280
+ "id": "Qwen/Qwen3-Next-80B-A3B-Instruct",
281
+ "name": "Qwen3-Next 80B-A3B Instruct",
282
+ "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
283
+ "family": "qwen",
284
+ "attachment": false,
285
+ "reasoning": false,
306
286
  "tool_call": true,
307
287
  "structured_output": true,
308
288
  "temperature": true,
309
- "release_date": "2026-04-01",
310
- "last_updated": "2026-04-01",
289
+ "knowledge": "2025-04",
290
+ "release_date": "2025-09",
291
+ "last_updated": "2025-09",
311
292
  "modalities": {
312
293
  "input": [
313
- "text",
314
- "image",
315
- "video"
294
+ "text"
316
295
  ],
317
296
  "output": [
318
297
  "text"
@@ -321,11 +300,11 @@
321
300
  "open_weights": true,
322
301
  "limit": {
323
302
  "context": 262144,
324
- "output": 81920
303
+ "output": 32768
325
304
  },
326
305
  "cost": {
327
- "input": 0.15,
328
- "output": 0.95
306
+ "input": 0.09,
307
+ "output": 1.1
329
308
  }
330
309
  },
331
310
  "Qwen/Qwen3.7-Max": {
@@ -379,71 +358,6 @@
379
358
  ]
380
359
  }
381
360
  },
382
- "Qwen/Qwen3.6-27B": {
383
- "id": "Qwen/Qwen3.6-27B",
384
- "name": "Qwen3.6 27B",
385
- "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
386
- "family": "qwen",
387
- "attachment": true,
388
- "reasoning": true,
389
- "reasoning_options": [],
390
- "tool_call": true,
391
- "structured_output": true,
392
- "temperature": true,
393
- "release_date": "2026-04-22",
394
- "last_updated": "2026-04-22",
395
- "modalities": {
396
- "input": [
397
- "text",
398
- "image",
399
- "video",
400
- "audio"
401
- ],
402
- "output": [
403
- "text"
404
- ]
405
- },
406
- "open_weights": true,
407
- "limit": {
408
- "context": 262144,
409
- "output": 65536
410
- },
411
- "cost": {
412
- "input": 0.32,
413
- "output": 3.2
414
- }
415
- },
416
- "Qwen/Qwen3-Next-80B-A3B-Instruct": {
417
- "id": "Qwen/Qwen3-Next-80B-A3B-Instruct",
418
- "name": "Qwen3-Next 80B-A3B Instruct",
419
- "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
420
- "family": "qwen",
421
- "attachment": false,
422
- "reasoning": false,
423
- "tool_call": true,
424
- "structured_output": true,
425
- "temperature": true,
426
- "knowledge": "2025-04",
427
- "release_date": "2025-09",
428
- "last_updated": "2025-09",
429
- "modalities": {
430
- "input": [
431
- "text"
432
- ],
433
- "output": [
434
- "text"
435
- ]
436
- },
437
- "open_weights": true,
438
- "limit": {
439
- "context": 262144,
440
- "output": 32768
441
- },
442
- "cost": {
443
- "input": 0.09,
444
- "output": 1.1
445
- }
446
- },
447
361
  "Qwen/Qwen3-Max": {
448
362
  "id": "Qwen/Qwen3-Max",
449
363
  "name": "Qwen3 Max",
@@ -496,9 +410,9 @@
496
410
  ]
497
411
  }
498
412
  },
499
- "Qwen/Qwen3.5-397B-A17B": {
500
- "id": "Qwen/Qwen3.5-397B-A17B",
501
- "name": "Qwen 3.5 397B A17B",
413
+ "Qwen/Qwen3.6-35B-A3B": {
414
+ "id": "Qwen/Qwen3.6-35B-A3B",
415
+ "name": "Qwen3.6 35B A3B",
502
416
  "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
503
417
  "family": "qwen",
504
418
  "attachment": true,
@@ -507,9 +421,8 @@
507
421
  "tool_call": true,
508
422
  "structured_output": true,
509
423
  "temperature": true,
510
- "knowledge": "2025-01",
511
- "release_date": "2026-02-01",
512
- "last_updated": "2026-04-20",
424
+ "release_date": "2026-04-01",
425
+ "last_updated": "2026-04-01",
513
426
  "modalities": {
514
427
  "input": [
515
428
  "text",
@@ -526,30 +439,29 @@
526
439
  "output": 81920
527
440
  },
528
441
  "cost": {
529
- "input": 0.45,
530
- "output": 3,
531
- "cache_read": 0.22
442
+ "input": 0.15,
443
+ "output": 0.95
532
444
  }
533
445
  },
534
- "Qwen/Qwen3.5-122B-A10B": {
535
- "id": "Qwen/Qwen3.5-122B-A10B",
536
- "name": "Qwen3.5 122B-A10B",
537
- "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
538
- "family": "qwen",
539
- "attachment": true,
446
+ "nvidia/Nemotron-3-Nano-30B-A3B": {
447
+ "id": "nvidia/Nemotron-3-Nano-30B-A3B",
448
+ "name": "Nemotron 3 Nano 30B A3B",
449
+ "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents",
450
+ "family": "nemotron",
451
+ "attachment": false,
540
452
  "reasoning": true,
541
- "reasoning_options": [],
453
+ "reasoning_options": [
454
+ {
455
+ "type": "toggle"
456
+ }
457
+ ],
542
458
  "tool_call": true,
543
- "structured_output": true,
544
459
  "temperature": true,
545
- "release_date": "2026-02-23",
546
- "last_updated": "2026-02-23",
460
+ "release_date": "2025-12-15",
461
+ "last_updated": "2025-12-15",
547
462
  "modalities": {
548
463
  "input": [
549
- "text",
550
- "image",
551
- "video",
552
- "audio"
464
+ "text"
553
465
  ],
554
466
  "output": [
555
467
  "text"
@@ -558,32 +470,29 @@
558
470
  "open_weights": true,
559
471
  "limit": {
560
472
  "context": 262144,
561
- "output": 65536
473
+ "output": 262144
562
474
  },
563
475
  "cost": {
564
- "input": 0.29,
565
- "output": 2.4
476
+ "input": 0.05,
477
+ "output": 0.2
566
478
  }
567
479
  },
568
- "Qwen/Qwen3.5-27B": {
569
- "id": "Qwen/Qwen3.5-27B",
570
- "name": "Qwen3.5 27B",
571
- "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
572
- "family": "qwen",
573
- "attachment": true,
480
+ "nvidia/Llama-3.3-Nemotron-Super-49B-v1.5": {
481
+ "id": "nvidia/Llama-3.3-Nemotron-Super-49B-v1.5",
482
+ "name": "Llama 3.3 Nemotron Super 49B v1.5",
483
+ "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
484
+ "family": "nemotron",
485
+ "attachment": false,
574
486
  "reasoning": true,
575
487
  "reasoning_options": [],
576
488
  "tool_call": true,
577
489
  "structured_output": true,
578
490
  "temperature": true,
579
- "release_date": "2026-02-23",
580
- "last_updated": "2026-02-23",
491
+ "release_date": "2025-07-25",
492
+ "last_updated": "2025-07-25",
581
493
  "modalities": {
582
494
  "input": [
583
- "text",
584
- "image",
585
- "video",
586
- "audio"
495
+ "text"
587
496
  ],
588
497
  "output": [
589
498
  "text"
@@ -591,32 +500,33 @@
591
500
  },
592
501
  "open_weights": true,
593
502
  "limit": {
594
- "context": 262144,
595
- "output": 65536
503
+ "context": 131072,
504
+ "output": 131072
596
505
  },
597
506
  "cost": {
598
- "input": 0.26,
599
- "output": 2.6
507
+ "input": 0.4,
508
+ "output": 0.4
600
509
  }
601
510
  },
602
- "Qwen/Qwen3.5-9B": {
603
- "id": "Qwen/Qwen3.5-9B",
604
- "name": "Qwen3.5 9B",
605
- "description": "Qwen instruction model for multilingual chat, reasoning, and tool use",
606
- "family": "qwen",
511
+ "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning": {
512
+ "id": "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning",
513
+ "name": "Nemotron 3 Nano Omni 30B A3B Reasoning",
514
+ "description": "Open Nemotron omni model combining reasoning with text, vision, and audio",
515
+ "family": "nemotron",
607
516
  "attachment": true,
608
517
  "reasoning": true,
609
518
  "reasoning_options": [],
610
519
  "tool_call": true,
611
520
  "structured_output": true,
612
521
  "temperature": true,
613
- "release_date": "2026-02-23",
614
- "last_updated": "2026-02-23",
522
+ "release_date": "2026-04-28",
523
+ "last_updated": "2026-04-28",
615
524
  "modalities": {
616
525
  "input": [
617
526
  "text",
618
527
  "image",
619
- "video"
528
+ "video",
529
+ "audio"
620
530
  ],
621
531
  "output": [
622
532
  "text"
@@ -628,27 +538,25 @@
628
538
  "output": 65536
629
539
  },
630
540
  "cost": {
631
- "input": 0.1,
632
- "output": 0.15
541
+ "input": 0.2,
542
+ "output": 0.8
633
543
  }
634
544
  },
635
- "Qwen/Qwen3-32B": {
636
- "id": "Qwen/Qwen3-32B",
637
- "name": "Qwen3 32B",
638
- "description": "Dense open Qwen model for self-hosted chat, reasoning, and coding",
639
- "family": "qwen",
640
- "attachment": false,
641
- "reasoning": true,
642
- "reasoning_options": [],
545
+ "meta-llama/Llama-4-Scout-17B-16E-Instruct": {
546
+ "id": "meta-llama/Llama-4-Scout-17B-16E-Instruct",
547
+ "name": "Llama 4 Scout 17B",
548
+ "description": "Open multimodal Llama model for long-context analysis and efficient agents",
549
+ "family": "llama",
550
+ "attachment": true,
551
+ "reasoning": false,
643
552
  "tool_call": true,
644
553
  "structured_output": true,
645
- "temperature": true,
646
- "knowledge": "2025-04",
647
- "release_date": "2025-04",
648
- "last_updated": "2025-04",
554
+ "release_date": "2025-04-05",
555
+ "last_updated": "2025-04-05",
649
556
  "modalities": {
650
557
  "input": [
651
- "text"
558
+ "text",
559
+ "image"
652
560
  ],
653
561
  "output": [
654
562
  "text"
@@ -656,30 +564,29 @@
656
564
  },
657
565
  "open_weights": true,
658
566
  "limit": {
659
- "context": 40960,
567
+ "context": 327680,
660
568
  "output": 16384
661
569
  },
662
570
  "cost": {
663
- "input": 0.08,
664
- "output": 0.28
571
+ "input": 0.1,
572
+ "output": 0.3
665
573
  }
666
574
  },
667
- "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo": {
668
- "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo",
669
- "name": "Qwen3 Coder 480B A35B Instruct Turbo",
670
- "description": "Qwen coding model for software agents, repository edits, and code reasoning",
671
- "family": "qwen",
672
- "attachment": false,
575
+ "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8": {
576
+ "id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8",
577
+ "name": "Llama 4 Maverick 17B FP8",
578
+ "description": "Open multimodal Llama model for strong reasoning and fast responses",
579
+ "family": "llama",
580
+ "attachment": true,
673
581
  "reasoning": false,
674
- "tool_call": true,
582
+ "tool_call": false,
675
583
  "structured_output": true,
676
- "temperature": true,
677
- "knowledge": "2025-04",
678
- "release_date": "2025-07-23",
679
- "last_updated": "2025-07-23",
584
+ "release_date": "2025-04-05",
585
+ "last_updated": "2025-04-05",
680
586
  "modalities": {
681
587
  "input": [
682
- "text"
588
+ "text",
589
+ "image"
683
590
  ],
684
591
  "output": [
685
592
  "text"
@@ -687,34 +594,28 @@
687
594
  },
688
595
  "open_weights": true,
689
596
  "limit": {
690
- "context": 262144,
691
- "output": 66536
597
+ "context": 1048576,
598
+ "output": 16384
692
599
  },
693
600
  "cost": {
694
- "input": 0.3,
695
- "output": 1,
696
- "cache_read": 0.1
601
+ "input": 0.2,
602
+ "output": 0.8
697
603
  }
698
604
  },
699
- "Qwen/Qwen3.5-35B-A3B": {
700
- "id": "Qwen/Qwen3.5-35B-A3B",
701
- "name": "Qwen 3.5 35B A3B",
702
- "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
703
- "family": "qwen",
704
- "attachment": true,
705
- "reasoning": true,
706
- "reasoning_options": [],
605
+ "meta-llama/Llama-3.3-70B-Instruct-Turbo": {
606
+ "id": "meta-llama/Llama-3.3-70B-Instruct-Turbo",
607
+ "name": "Llama 3.3 70B Turbo",
608
+ "description": "Compact Llama instruction model for fast chat and local deployment",
609
+ "family": "llama",
610
+ "attachment": false,
611
+ "reasoning": false,
707
612
  "tool_call": true,
708
613
  "structured_output": true,
709
- "temperature": true,
710
- "knowledge": "2025-01",
711
- "release_date": "2026-02-01",
712
- "last_updated": "2026-04-20",
614
+ "release_date": "2024-12-06",
615
+ "last_updated": "2024-12-06",
713
616
  "modalities": {
714
617
  "input": [
715
- "text",
716
- "image",
717
- "video"
618
+ "text"
718
619
  ],
719
620
  "output": [
720
621
  "text"
@@ -722,37 +623,30 @@
722
623
  },
723
624
  "open_weights": true,
724
625
  "limit": {
725
- "context": 262144,
726
- "output": 81920
626
+ "context": 131072,
627
+ "output": 16384
727
628
  },
728
629
  "cost": {
729
- "input": 0.14,
730
- "output": 1,
731
- "cache_read": 0.05
630
+ "input": 0.1,
631
+ "output": 0.32
732
632
  }
733
633
  },
734
- "openai/gpt-oss-120b": {
735
- "id": "openai/gpt-oss-120b",
736
- "name": "GPT OSS 120B",
737
- "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
738
- "family": "gpt-oss",
634
+ "deepseek-ai/DeepSeek-R1-0528": {
635
+ "id": "deepseek-ai/DeepSeek-R1-0528",
636
+ "name": "DeepSeek-R1-0528",
637
+ "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
739
638
  "attachment": false,
740
639
  "reasoning": true,
741
- "reasoning_options": [
742
- {
743
- "type": "effort",
744
- "values": [
745
- "low",
746
- "medium",
747
- "high"
748
- ]
749
- }
750
- ],
640
+ "reasoning_options": [],
751
641
  "tool_call": true,
642
+ "interleaved": {
643
+ "field": "reasoning_content"
644
+ },
752
645
  "structured_output": true,
753
646
  "temperature": true,
754
- "release_date": "2025-08-05",
755
- "last_updated": "2025-08-05",
647
+ "knowledge": "2024-07",
648
+ "release_date": "2025-05-28",
649
+ "last_updated": "2025-05-28",
756
650
  "modalities": {
757
651
  "input": [
758
652
  "text"
@@ -761,38 +655,47 @@
761
655
  "text"
762
656
  ]
763
657
  },
764
- "open_weights": true,
658
+ "open_weights": false,
765
659
  "limit": {
766
- "context": 131072,
767
- "output": 16384
660
+ "context": 163840,
661
+ "output": 64000
768
662
  },
769
663
  "cost": {
770
- "input": 0.037,
771
- "output": 0.17
664
+ "input": 0.5,
665
+ "output": 2.15,
666
+ "cache_read": 0.35
772
667
  }
773
668
  },
774
- "openai/gpt-oss-20b": {
775
- "id": "openai/gpt-oss-20b",
776
- "name": "GPT OSS 20B",
777
- "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
778
- "family": "gpt-oss",
669
+ "deepseek-ai/DeepSeek-V4-Pro": {
670
+ "id": "deepseek-ai/DeepSeek-V4-Pro",
671
+ "name": "DeepSeek V4 Pro",
672
+ "description": "Open MoE flagship with million-token context for coding and long agent runs",
673
+ "family": "deepseek-thinking",
779
674
  "attachment": false,
780
675
  "reasoning": true,
781
676
  "reasoning_options": [
677
+ {
678
+ "type": "toggle"
679
+ },
782
680
  {
783
681
  "type": "effort",
784
682
  "values": [
785
683
  "low",
786
684
  "medium",
787
- "high"
685
+ "high",
686
+ "xhigh"
788
687
  ]
789
688
  }
790
689
  ],
791
690
  "tool_call": true,
691
+ "interleaved": {
692
+ "field": "reasoning_content"
693
+ },
792
694
  "structured_output": true,
793
695
  "temperature": true,
794
- "release_date": "2025-08-05",
795
- "last_updated": "2025-08-05",
696
+ "knowledge": "2025-05",
697
+ "release_date": "2026-04-24",
698
+ "last_updated": "2026-04-24",
796
699
  "modalities": {
797
700
  "input": [
798
701
  "text"
@@ -803,24 +706,34 @@
803
706
  },
804
707
  "open_weights": true,
805
708
  "limit": {
806
- "context": 131072,
709
+ "context": 1048576,
807
710
  "output": 16384
808
711
  },
809
712
  "cost": {
810
- "input": 0.03,
811
- "output": 0.14
713
+ "input": 1.3,
714
+ "output": 2.6,
715
+ "cache_read": 0.1
812
716
  }
813
717
  },
814
- "XiaomiMiMo/MiMo-V2.5": {
815
- "id": "XiaomiMiMo/MiMo-V2.5",
816
- "name": "MiMo-V2.5",
817
- "description": "Open MiMo model for multimodal coding agents and long-context automation",
818
- "family": "mimo",
819
- "attachment": true,
718
+ "deepseek-ai/DeepSeek-V4-Flash": {
719
+ "id": "deepseek-ai/DeepSeek-V4-Flash",
720
+ "name": "DeepSeek V4 Flash",
721
+ "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
722
+ "family": "deepseek-flash",
723
+ "attachment": false,
820
724
  "reasoning": true,
821
725
  "reasoning_options": [
822
726
  {
823
727
  "type": "toggle"
728
+ },
729
+ {
730
+ "type": "effort",
731
+ "values": [
732
+ "low",
733
+ "medium",
734
+ "high",
735
+ "xhigh"
736
+ ]
824
737
  }
825
738
  ],
826
739
  "tool_call": true,
@@ -829,15 +742,12 @@
829
742
  },
830
743
  "structured_output": true,
831
744
  "temperature": true,
832
- "knowledge": "2024-12",
833
- "release_date": "2026-04-22",
834
- "last_updated": "2026-04-22",
745
+ "knowledge": "2025-05",
746
+ "release_date": "2026-04-24",
747
+ "last_updated": "2026-04-24",
835
748
  "modalities": {
836
749
  "input": [
837
- "text",
838
- "image",
839
- "audio",
840
- "video"
750
+ "text"
841
751
  ],
842
752
  "output": [
843
753
  "text"
@@ -845,21 +755,20 @@
845
755
  },
846
756
  "open_weights": true,
847
757
  "limit": {
848
- "context": 262144,
758
+ "context": 1048576,
849
759
  "output": 16384
850
760
  },
851
761
  "cost": {
852
- "input": 0.4,
853
- "output": 2,
854
- "cache_read": 0.08
762
+ "input": 0.09,
763
+ "output": 0.18,
764
+ "cache_read": 0.018
855
765
  }
856
766
  },
857
- "XiaomiMiMo/MiMo-V2.5-Pro": {
858
- "id": "XiaomiMiMo/MiMo-V2.5-Pro",
859
- "name": "MiMo-V2.5-Pro",
860
- "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
861
- "family": "mimo",
862
- "attachment": true,
767
+ "deepseek-ai/DeepSeek-V3.2": {
768
+ "id": "deepseek-ai/DeepSeek-V3.2",
769
+ "name": "DeepSeek-V3.2",
770
+ "description": "DeepSeek chat model for instruction following, coding, and analysis",
771
+ "attachment": false,
863
772
  "reasoning": true,
864
773
  "reasoning_options": [
865
774
  {
@@ -873,47 +782,46 @@
873
782
  "structured_output": true,
874
783
  "temperature": true,
875
784
  "knowledge": "2024-12",
876
- "release_date": "2026-04-22",
877
- "last_updated": "2026-04-22",
785
+ "release_date": "2025-12-02",
786
+ "last_updated": "2025-12-02",
878
787
  "modalities": {
879
788
  "input": [
880
- "text",
881
- "audio"
789
+ "text"
882
790
  ],
883
791
  "output": [
884
792
  "text"
885
793
  ]
886
794
  },
887
- "open_weights": true,
795
+ "open_weights": false,
888
796
  "limit": {
889
- "context": 1048576,
890
- "output": 16384
797
+ "context": 163840,
798
+ "output": 64000
891
799
  },
892
800
  "cost": {
893
- "input": 1,
894
- "output": 3,
895
- "cache_read": 0.2
801
+ "input": 0.26,
802
+ "output": 0.38,
803
+ "cache_read": 0.13
896
804
  }
897
805
  },
898
- "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning": {
899
- "id": "nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning",
900
- "name": "Nemotron 3 Nano Omni 30B A3B Reasoning",
901
- "description": "Open Nemotron omni model combining reasoning with text, vision, and audio",
902
- "family": "nemotron",
903
- "attachment": true,
806
+ "MiniMaxAI/MiniMax-M2.5": {
807
+ "id": "MiniMaxAI/MiniMax-M2.5",
808
+ "name": "MiniMax M2.5",
809
+ "description": "MiniMax model for chat, coding, office work, and agentic tasks",
810
+ "family": "minimax",
811
+ "attachment": false,
904
812
  "reasoning": true,
905
813
  "reasoning_options": [],
906
814
  "tool_call": true,
907
- "structured_output": true,
815
+ "interleaved": {
816
+ "field": "reasoning_content"
817
+ },
908
818
  "temperature": true,
909
- "release_date": "2026-04-28",
910
- "last_updated": "2026-04-28",
819
+ "knowledge": "2025-06",
820
+ "release_date": "2026-02-12",
821
+ "last_updated": "2026-02-12",
911
822
  "modalities": {
912
823
  "input": [
913
- "text",
914
- "image",
915
- "video",
916
- "audio"
824
+ "text"
917
825
  ],
918
826
  "output": [
919
827
  "text"
@@ -921,30 +829,27 @@
921
829
  },
922
830
  "open_weights": true,
923
831
  "limit": {
924
- "context": 262144,
925
- "output": 65536
832
+ "context": 196608,
833
+ "output": 131072
926
834
  },
927
835
  "cost": {
928
- "input": 0.2,
929
- "output": 0.8
836
+ "input": 0.15,
837
+ "output": 1.15,
838
+ "cache_read": 0.03
930
839
  }
931
840
  },
932
- "nvidia/Nemotron-3-Nano-30B-A3B": {
933
- "id": "nvidia/Nemotron-3-Nano-30B-A3B",
934
- "name": "Nemotron 3 Nano 30B A3B",
935
- "description": "Small Nemotron 3 MoE for efficient coding, math, and long-context agents",
936
- "family": "nemotron",
841
+ "MiniMaxAI/MiniMax-M2.7": {
842
+ "id": "MiniMaxAI/MiniMax-M2.7",
843
+ "name": "MiniMax-M2.7",
844
+ "description": "Open MiniMax flagship for coding agents, office automation, and complex environments",
845
+ "family": "minimax",
937
846
  "attachment": false,
938
847
  "reasoning": true,
939
- "reasoning_options": [
940
- {
941
- "type": "toggle"
942
- }
943
- ],
848
+ "reasoning_options": [],
944
849
  "tool_call": true,
945
850
  "temperature": true,
946
- "release_date": "2025-12-15",
947
- "last_updated": "2025-12-15",
851
+ "release_date": "2026-03-18",
852
+ "last_updated": "2026-03-18",
948
853
  "modalities": {
949
854
  "input": [
950
855
  "text"
@@ -955,30 +860,32 @@
955
860
  },
956
861
  "open_weights": true,
957
862
  "limit": {
958
- "context": 262144,
959
- "output": 262144
863
+ "context": 196608,
864
+ "output": 131072
960
865
  },
961
866
  "cost": {
962
- "input": 0.05,
963
- "output": 0.2
867
+ "input": 0.25,
868
+ "output": 1,
869
+ "cache_read": 0.05
964
870
  }
965
871
  },
966
- "nvidia/Llama-3.3-Nemotron-Super-49B-v1.5": {
967
- "id": "nvidia/Llama-3.3-Nemotron-Super-49B-v1.5",
968
- "name": "Llama 3.3 Nemotron Super 49B v1.5",
969
- "description": "Nemotron model for efficient reasoning, coding, and specialized AI agents",
970
- "family": "nemotron",
971
- "attachment": false,
872
+ "MiniMaxAI/MiniMax-M3": {
873
+ "id": "MiniMaxAI/MiniMax-M3",
874
+ "name": "MiniMax-M3",
875
+ "description": "MiniMax multimodal model for long-context coding, perception, and agent planning",
876
+ "family": "minimax",
877
+ "attachment": true,
972
878
  "reasoning": true,
973
879
  "reasoning_options": [],
974
880
  "tool_call": true,
975
- "structured_output": true,
976
881
  "temperature": true,
977
- "release_date": "2025-07-25",
978
- "last_updated": "2025-07-25",
882
+ "release_date": "2026-06-01",
883
+ "last_updated": "2026-06-01",
979
884
  "modalities": {
980
885
  "input": [
981
- "text"
886
+ "text",
887
+ "image",
888
+ "video"
982
889
  ],
983
890
  "output": [
984
891
  "text"
@@ -986,12 +893,13 @@
986
893
  },
987
894
  "open_weights": true,
988
895
  "limit": {
989
- "context": 131072,
990
- "output": 131072
991
- },
896
+ "context": 524288,
897
+ "output": 128000
898
+ },
992
899
  "cost": {
993
- "input": 0.4,
994
- "output": 0.4
900
+ "input": 0.3,
901
+ "output": 1.2,
902
+ "cache_read": 0.06
995
903
  }
996
904
  },
997
905
  "zai-org/GLM-4.7-Flash": {
@@ -1070,6 +978,54 @@
1070
978
  "cache_read": 0.1
1071
979
  }
1072
980
  },
981
+ "zai-org/GLM-5.2": {
982
+ "id": "zai-org/GLM-5.2",
983
+ "name": "GLM-5.2",
984
+ "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
985
+ "family": "glm",
986
+ "attachment": false,
987
+ "reasoning": true,
988
+ "reasoning_options": [
989
+ {
990
+ "type": "toggle"
991
+ },
992
+ {
993
+ "type": "effort",
994
+ "values": [
995
+ "low",
996
+ "medium",
997
+ "high",
998
+ "xhigh"
999
+ ]
1000
+ }
1001
+ ],
1002
+ "tool_call": true,
1003
+ "interleaved": {
1004
+ "field": "reasoning_content"
1005
+ },
1006
+ "structured_output": true,
1007
+ "temperature": true,
1008
+ "release_date": "2026-06-13",
1009
+ "last_updated": "2026-06-13",
1010
+ "modalities": {
1011
+ "input": [
1012
+ "text"
1013
+ ],
1014
+ "output": [
1015
+ "text"
1016
+ ]
1017
+ },
1018
+ "open_weights": true,
1019
+ "limit": {
1020
+ "context": 1048576,
1021
+ "output": 32768
1022
+ },
1023
+ "cost": {
1024
+ "input": 0.93,
1025
+ "output": 3,
1026
+ "cache_read": 0.18
1027
+ }
1028
+ },
1073
1029
  "zai-org/GLM-5": {
1074
1030
  "id": "zai-org/GLM-5",
1075
1031
  "name": "GLM-5",
@@ -1150,25 +1106,16 @@
1150
1106
  "cache_read": 0.08
1151
1107
  }
1152
1108
  },
1153
- "zai-org/GLM-5.2": {
1154
- "id": "zai-org/GLM-5.2",
1155
- "name": "GLM-5.2",
1156
- "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
1109
+ "zai-org/GLM-5.1": {
1110
+ "id": "zai-org/GLM-5.1",
1111
+ "name": "GLM-5.1",
1112
+ "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
1157
1113
  "family": "glm",
1158
1114
  "attachment": false,
1159
1115
  "reasoning": true,
1160
1116
  "reasoning_options": [
1161
1117
  {
1162
1118
  "type": "toggle"
1163
- },
1164
- {
1165
- "type": "effort",
1166
- "values": [
1167
- "low",
1168
- "medium",
1169
- "high",
1170
- "xhigh"
1171
- ]
1172
1119
  }
1173
1120
  ],
1174
1121
  "tool_call": true,
@@ -1177,8 +1124,9 @@
1177
1124
  },
1178
1125
  "structured_output": true,
1179
1126
  "temperature": true,
1180
- "release_date": "2026-06-13",
1181
- "last_updated": "2026-06-13",
1127
+ "knowledge": "2025-04",
1128
+ "release_date": "2026-04-07",
1129
+ "last_updated": "2026-04-07",
1182
1130
  "modalities": {
1183
1131
  "input": [
1184
1132
  "text"
@@ -1189,21 +1137,21 @@
1189
1137
  },
1190
1138
  "open_weights": true,
1191
1139
  "limit": {
1192
- "context": 1048576,
1193
- "output": 32768
1140
+ "context": 202752,
1141
+ "output": 16384
1194
1142
  },
1195
1143
  "cost": {
1196
- "input": 0.93,
1197
- "output": 3,
1198
- "cache_read": 0.18
1144
+ "input": 1.05,
1145
+ "output": 3.5,
1146
+ "cache_read": 0.205
1199
1147
  }
1200
1148
  },
1201
- "zai-org/GLM-5.1": {
1202
- "id": "zai-org/GLM-5.1",
1203
- "name": "GLM-5.1",
1204
- "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
1205
- "family": "glm",
1206
- "attachment": false,
1149
+ "moonshotai/Kimi-K2.6": {
1150
+ "id": "moonshotai/Kimi-K2.6",
1151
+ "name": "Kimi K2.6",
1152
+ "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
1153
+ "family": "kimi-k2",
1154
+ "attachment": true,
1207
1155
  "reasoning": true,
1208
1156
  "reasoning_options": [
1209
1157
  {
@@ -1216,12 +1164,14 @@
1216
1164
  },
1217
1165
  "structured_output": true,
1218
1166
  "temperature": true,
1219
- "knowledge": "2025-04",
1220
- "release_date": "2026-04-07",
1221
- "last_updated": "2026-04-07",
1167
+ "knowledge": "2024-04",
1168
+ "release_date": "2026-04-21",
1169
+ "last_updated": "2026-04-21",
1222
1170
  "modalities": {
1223
1171
  "input": [
1224
- "text"
1172
+ "text",
1173
+ "image",
1174
+ "video"
1225
1175
  ],
1226
1176
  "output": [
1227
1177
  "text"
@@ -1229,69 +1179,67 @@
1229
1179
  },
1230
1180
  "open_weights": true,
1231
1181
  "limit": {
1232
- "context": 202752,
1182
+ "context": 262144,
1233
1183
  "output": 16384
1234
1184
  },
1235
1185
  "cost": {
1236
- "input": 1.05,
1186
+ "input": 0.75,
1237
1187
  "output": 3.5,
1238
- "cache_read": 0.205
1188
+ "cache_read": 0.15
1239
1189
  }
1240
1190
  },
1241
- "deepseek-ai/DeepSeek-R1-0528": {
1242
- "id": "deepseek-ai/DeepSeek-R1-0528",
1243
- "name": "DeepSeek-R1-0528",
1244
- "description": "DeepSeek reasoning model for multi-step analysis, math, coding, and tools",
1245
- "attachment": false,
1191
+ "moonshotai/Kimi-K2.5": {
1192
+ "id": "moonshotai/Kimi-K2.5",
1193
+ "name": "Kimi K2.5",
1194
+ "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
1195
+ "family": "kimi-k2",
1196
+ "attachment": true,
1246
1197
  "reasoning": true,
1247
- "reasoning_options": [],
1198
+ "reasoning_options": [
1199
+ {
1200
+ "type": "toggle"
1201
+ }
1202
+ ],
1248
1203
  "tool_call": true,
1249
1204
  "interleaved": {
1250
1205
  "field": "reasoning_content"
1251
1206
  },
1252
1207
  "structured_output": true,
1253
1208
  "temperature": true,
1254
- "knowledge": "2024-07",
1255
- "release_date": "2025-05-28",
1256
- "last_updated": "2025-05-28",
1209
+ "knowledge": "2025-01",
1210
+ "release_date": "2026-01-27",
1211
+ "last_updated": "2026-01-27",
1257
1212
  "modalities": {
1258
1213
  "input": [
1259
- "text"
1214
+ "text",
1215
+ "image",
1216
+ "video"
1260
1217
  ],
1261
1218
  "output": [
1262
1219
  "text"
1263
1220
  ]
1264
1221
  },
1265
- "open_weights": false,
1222
+ "open_weights": true,
1266
1223
  "limit": {
1267
- "context": 163840,
1268
- "output": 64000
1224
+ "context": 262144,
1225
+ "output": 32768
1269
1226
  },
1270
1227
  "cost": {
1271
- "input": 0.5,
1272
- "output": 2.15,
1273
- "cache_read": 0.35
1228
+ "input": 0.45,
1229
+ "output": 2.25,
1230
+ "cache_read": 0.07
1274
1231
  }
1275
1232
  },
1276
- "deepseek-ai/DeepSeek-V4-Flash": {
1277
- "id": "deepseek-ai/DeepSeek-V4-Flash",
1278
- "name": "DeepSeek V4 Flash",
1279
- "description": "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work",
1280
- "family": "deepseek-flash",
1281
- "attachment": false,
1233
+ "moonshotai/Kimi-K2.7-Code": {
1234
+ "id": "moonshotai/Kimi-K2.7-Code",
1235
+ "name": "Kimi K2.7 Code",
1236
+ "description": "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking",
1237
+ "family": "kimi-k2",
1238
+ "attachment": true,
1282
1239
  "reasoning": true,
1283
1240
  "reasoning_options": [
1284
1241
  {
1285
1242
  "type": "toggle"
1286
- },
1287
- {
1288
- "type": "effort",
1289
- "values": [
1290
- "low",
1291
- "medium",
1292
- "high",
1293
- "xhigh"
1294
- ]
1295
1243
  }
1296
1244
  ],
1297
1245
  "tool_call": true,
@@ -1299,13 +1247,15 @@
1299
1247
  "field": "reasoning_content"
1300
1248
  },
1301
1249
  "structured_output": true,
1302
- "temperature": true,
1303
- "knowledge": "2025-05",
1304
- "release_date": "2026-04-24",
1305
- "last_updated": "2026-04-24",
1250
+ "temperature": false,
1251
+ "knowledge": "2025-01",
1252
+ "release_date": "2026-06-12",
1253
+ "last_updated": "2026-06-12",
1306
1254
  "modalities": {
1307
1255
  "input": [
1308
- "text"
1256
+ "text",
1257
+ "image",
1258
+ "video"
1309
1259
  ],
1310
1260
  "output": [
1311
1261
  "text"
@@ -1313,34 +1263,25 @@
1313
1263
  },
1314
1264
  "open_weights": true,
1315
1265
  "limit": {
1316
- "context": 1048576,
1317
- "output": 16384
1266
+ "context": 262144,
1267
+ "output": 262144
1318
1268
  },
1319
1269
  "cost": {
1320
- "input": 0.09,
1321
- "output": 0.18,
1322
- "cache_read": 0.018
1270
+ "input": 0.74,
1271
+ "output": 3.5,
1272
+ "cache_read": 0.15
1323
1273
  }
1324
1274
  },
1325
- "deepseek-ai/DeepSeek-V4-Pro": {
1326
- "id": "deepseek-ai/DeepSeek-V4-Pro",
1327
- "name": "DeepSeek V4 Pro",
1328
- "description": "Open MoE flagship with million-token context for coding and long agent runs",
1329
- "family": "deepseek-thinking",
1330
- "attachment": false,
1275
+ "XiaomiMiMo/MiMo-V2.5-Pro": {
1276
+ "id": "XiaomiMiMo/MiMo-V2.5-Pro",
1277
+ "name": "MiMo-V2.5-Pro",
1278
+ "description": "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution",
1279
+ "family": "mimo",
1280
+ "attachment": true,
1331
1281
  "reasoning": true,
1332
1282
  "reasoning_options": [
1333
1283
  {
1334
1284
  "type": "toggle"
1335
- },
1336
- {
1337
- "type": "effort",
1338
- "values": [
1339
- "low",
1340
- "medium",
1341
- "high",
1342
- "xhigh"
1343
- ]
1344
1285
  }
1345
1286
  ],
1346
1287
  "tool_call": true,
@@ -1349,12 +1290,13 @@
1349
1290
  },
1350
1291
  "structured_output": true,
1351
1292
  "temperature": true,
1352
- "knowledge": "2025-05",
1353
- "release_date": "2026-04-24",
1354
- "last_updated": "2026-04-24",
1293
+ "knowledge": "2024-12",
1294
+ "release_date": "2026-04-22",
1295
+ "last_updated": "2026-04-22",
1355
1296
  "modalities": {
1356
1297
  "input": [
1357
- "text"
1298
+ "text",
1299
+ "audio"
1358
1300
  ],
1359
1301
  "output": [
1360
1302
  "text"
@@ -1366,16 +1308,17 @@
1366
1308
  "output": 16384
1367
1309
  },
1368
1310
  "cost": {
1369
- "input": 1.3,
1370
- "output": 2.6,
1371
- "cache_read": 0.1
1311
+ "input": 1,
1312
+ "output": 3,
1313
+ "cache_read": 0.2
1372
1314
  }
1373
1315
  },
1374
- "deepseek-ai/DeepSeek-V3.2": {
1375
- "id": "deepseek-ai/DeepSeek-V3.2",
1376
- "name": "DeepSeek-V3.2",
1377
- "description": "DeepSeek chat model for instruction following, coding, and analysis",
1378
- "attachment": false,
1316
+ "XiaomiMiMo/MiMo-V2.5": {
1317
+ "id": "XiaomiMiMo/MiMo-V2.5",
1318
+ "name": "MiMo-V2.5",
1319
+ "description": "Open MiMo model for multimodal coding agents and long-context automation",
1320
+ "family": "mimo",
1321
+ "attachment": true,
1379
1322
  "reasoning": true,
1380
1323
  "reasoning_options": [
1381
1324
  {
@@ -1389,46 +1332,51 @@
1389
1332
  "structured_output": true,
1390
1333
  "temperature": true,
1391
1334
  "knowledge": "2024-12",
1392
- "release_date": "2025-12-02",
1393
- "last_updated": "2025-12-02",
1335
+ "release_date": "2026-04-22",
1336
+ "last_updated": "2026-04-22",
1394
1337
  "modalities": {
1395
1338
  "input": [
1396
- "text"
1339
+ "text",
1340
+ "image",
1341
+ "audio",
1342
+ "video"
1397
1343
  ],
1398
1344
  "output": [
1399
1345
  "text"
1400
1346
  ]
1401
1347
  },
1402
- "open_weights": false,
1348
+ "open_weights": true,
1403
1349
  "limit": {
1404
- "context": 163840,
1405
- "output": 64000
1350
+ "context": 262144,
1351
+ "output": 16384
1406
1352
  },
1407
1353
  "cost": {
1408
- "input": 0.26,
1409
- "output": 0.38,
1410
- "cache_read": 0.13
1354
+ "input": 0.4,
1355
+ "output": 2,
1356
+ "cache_read": 0.08
1411
1357
  }
1412
1358
  },
1413
- "MiniMaxAI/MiniMax-M2.5": {
1414
- "id": "MiniMaxAI/MiniMax-M2.5",
1415
- "name": "MiniMax M2.5",
1416
- "description": "MiniMax model for chat, coding, office work, and agentic tasks",
1417
- "family": "minimax",
1418
- "attachment": false,
1359
+ "google/gemma-4-26B-A4B-it": {
1360
+ "id": "google/gemma-4-26B-A4B-it",
1361
+ "name": "Gemma 4 26B A4B IT",
1362
+ "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
1363
+ "family": "gemma",
1364
+ "attachment": true,
1419
1365
  "reasoning": true,
1420
- "reasoning_options": [],
1366
+ "reasoning_options": [
1367
+ {
1368
+ "type": "toggle"
1369
+ }
1370
+ ],
1421
1371
  "tool_call": true,
1422
- "interleaved": {
1423
- "field": "reasoning_content"
1424
- },
1372
+ "structured_output": true,
1425
1373
  "temperature": true,
1426
- "knowledge": "2025-06",
1427
- "release_date": "2026-02-12",
1428
- "last_updated": "2026-02-12",
1374
+ "release_date": "2026-04-02",
1375
+ "last_updated": "2026-04-02",
1429
1376
  "modalities": {
1430
1377
  "input": [
1431
- "text"
1378
+ "text",
1379
+ "image"
1432
1380
  ],
1433
1381
  "output": [
1434
1382
  "text"
@@ -1436,27 +1384,31 @@
1436
1384
  },
1437
1385
  "open_weights": true,
1438
1386
  "limit": {
1439
- "context": 196608,
1440
- "output": 131072
1387
+ "context": 262144,
1388
+ "output": 32768
1441
1389
  },
1442
1390
  "cost": {
1443
- "input": 0.15,
1444
- "output": 1.15,
1445
- "cache_read": 0.03
1391
+ "input": 0.07,
1392
+ "output": 0.34
1446
1393
  }
1447
1394
  },
1448
- "MiniMaxAI/MiniMax-M3": {
1449
- "id": "MiniMaxAI/MiniMax-M3",
1450
- "name": "MiniMax-M3",
1451
- "description": "MiniMax multimodal model for long-context coding, perception, and agent planning",
1452
- "family": "minimax",
1395
+ "google/gemma-4-31B-it": {
1396
+ "id": "google/gemma-4-31B-it",
1397
+ "name": "Gemma 4 31B IT",
1398
+ "description": "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning",
1399
+ "family": "gemma",
1453
1400
  "attachment": true,
1454
1401
  "reasoning": true,
1455
- "reasoning_options": [],
1402
+ "reasoning_options": [
1403
+ {
1404
+ "type": "toggle"
1405
+ }
1406
+ ],
1456
1407
  "tool_call": true,
1408
+ "structured_output": true,
1457
1409
  "temperature": true,
1458
- "release_date": "2026-06-01",
1459
- "last_updated": "2026-06-01",
1410
+ "release_date": "2026-04-02",
1411
+ "last_updated": "2026-04-02",
1460
1412
  "modalities": {
1461
1413
  "input": [
1462
1414
  "text",
@@ -1469,27 +1421,36 @@
1469
1421
  },
1470
1422
  "open_weights": true,
1471
1423
  "limit": {
1472
- "context": 524288,
1473
- "output": 128000
1424
+ "context": 262144,
1425
+ "output": 32768
1474
1426
  },
1475
1427
  "cost": {
1476
- "input": 0.3,
1477
- "output": 1.2,
1478
- "cache_read": 0.06
1428
+ "input": 0.13,
1429
+ "output": 0.38
1479
1430
  }
1480
1431
  },
1481
- "MiniMaxAI/MiniMax-M2.7": {
1482
- "id": "MiniMaxAI/MiniMax-M2.7",
1483
- "name": "MiniMax-M2.7",
1484
- "description": "Open MiniMax flagship for coding agents, office automation, and complex environments",
1485
- "family": "minimax",
1432
+ "openai/gpt-oss-120b": {
1433
+ "id": "openai/gpt-oss-120b",
1434
+ "name": "GPT OSS 120B",
1435
+ "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
1436
+ "family": "gpt-oss",
1486
1437
  "attachment": false,
1487
1438
  "reasoning": true,
1488
- "reasoning_options": [],
1439
+ "reasoning_options": [
1440
+ {
1441
+ "type": "effort",
1442
+ "values": [
1443
+ "low",
1444
+ "medium",
1445
+ "high"
1446
+ ]
1447
+ }
1448
+ ],
1489
1449
  "tool_call": true,
1450
+ "structured_output": true,
1490
1451
  "temperature": true,
1491
- "release_date": "2026-03-18",
1492
- "last_updated": "2026-03-18",
1452
+ "release_date": "2025-08-05",
1453
+ "last_updated": "2025-08-05",
1493
1454
  "modalities": {
1494
1455
  "input": [
1495
1456
  "text"
@@ -1500,13 +1461,52 @@
1500
1461
  },
1501
1462
  "open_weights": true,
1502
1463
  "limit": {
1503
- "context": 196608,
1504
- "output": 131072
1464
+ "context": 131072,
1465
+ "output": 16384
1505
1466
  },
1506
1467
  "cost": {
1507
- "input": 0.25,
1508
- "output": 1,
1509
- "cache_read": 0.05
1468
+ "input": 0.037,
1469
+ "output": 0.17
1470
+ }
1471
+ },
1472
+ "openai/gpt-oss-20b": {
1473
+ "id": "openai/gpt-oss-20b",
1474
+ "name": "GPT OSS 20B",
1475
+ "description": "Open-weight GPT model for self-hosted reasoning and instruction-following workloads",
1476
+ "family": "gpt-oss",
1477
+ "attachment": false,
1478
+ "reasoning": true,
1479
+ "reasoning_options": [
1480
+ {
1481
+ "type": "effort",
1482
+ "values": [
1483
+ "low",
1484
+ "medium",
1485
+ "high"
1486
+ ]
1487
+ }
1488
+ ],
1489
+ "tool_call": true,
1490
+ "structured_output": true,
1491
+ "temperature": true,
1492
+ "release_date": "2025-08-05",
1493
+ "last_updated": "2025-08-05",
1494
+ "modalities": {
1495
+ "input": [
1496
+ "text"
1497
+ ],
1498
+ "output": [
1499
+ "text"
1500
+ ]
1501
+ },
1502
+ "open_weights": true,
1503
+ "limit": {
1504
+ "context": 131072,
1505
+ "output": 16384
1506
+ },
1507
+ "cost": {
1508
+ "input": 0.03,
1509
+ "output": 0.14
1510
1510
  }
1511
1511
  }
1512
1512
  }