llm.rb 15.2.2 → 15.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +197 -3
  3. data/README.md +180 -64
  4. data/bin/llm.rb +9 -2
  5. data/data/alibaba.json +45 -0
  6. data/data/anthropic.json +67 -0
  7. data/data/bedrock.json +1466 -397
  8. data/data/deepinfra.json +148 -16
  9. data/data/deepseek.json +3 -0
  10. data/data/mistral.json +42 -0
  11. data/data/openai.json +186 -0
  12. data/data/openrouter.json +1538 -378
  13. data/data/xai.json +53 -20
  14. data/data/zai.json +90 -4
  15. data/docs/deepdive/advanced/compaction.md +1 -2
  16. data/docs/deepdive/advanced/context.md +214 -1
  17. data/docs/deepdive/advanced/guard.md +9 -57
  18. data/docs/deepdive/features/builtin_tools.md +14 -16
  19. data/docs/deepdive/features/console.md +5 -0
  20. data/docs/deepdive/features/database.md +85 -10
  21. data/docs/deepdive/fundamentals/agents.md +7 -8
  22. data/docs/deepdive/fundamentals/providers.md +45 -5
  23. data/docs/deepdive/fundamentals/schema.md +73 -0
  24. data/docs/deepdive/fundamentals/tools.md +80 -27
  25. data/docs/deepdive/media/audio.md +8 -19
  26. data/docs/deepdive/media/images.md +8 -10
  27. data/docs/deepdive/media/ocr.md +1 -3
  28. data/docs/deepdive/reference/cost.md +48 -0
  29. data/docs/deepdive/reference/tracer.md +76 -0
  30. data/docs/deepdive.md +1 -1
  31. data/lib/llm/active_record/message.rb +113 -0
  32. data/lib/llm/active_record.rb +1 -0
  33. data/lib/llm/agent.rb +44 -19
  34. data/lib/llm/console/buffer.rb +9 -1
  35. data/lib/llm/console.rb +6 -1
  36. data/lib/llm/context/deserializer.rb +10 -3
  37. data/lib/llm/context.rb +62 -26
  38. data/lib/llm/guard.rb +2 -8
  39. data/lib/llm/message.rb +18 -7
  40. data/lib/llm/provider.rb +74 -16
  41. data/lib/llm/providers/alibaba.rb +15 -0
  42. data/lib/llm/providers/anthropic/error_handler.rb +5 -2
  43. data/lib/llm/providers/anthropic/files.rb +12 -12
  44. data/lib/llm/providers/anthropic/models.rb +2 -2
  45. data/lib/llm/providers/anthropic.rb +5 -3
  46. data/lib/llm/providers/bedrock/error_handler.rb +3 -2
  47. data/lib/llm/providers/bedrock/models.rb +5 -3
  48. data/lib/llm/providers/bedrock.rb +5 -3
  49. data/lib/llm/providers/deepinfra/audio.rb +4 -4
  50. data/lib/llm/providers/deepinfra/images.rb +4 -4
  51. data/lib/llm/providers/google/error_handler.rb +5 -2
  52. data/lib/llm/providers/google/files.rb +10 -10
  53. data/lib/llm/providers/google/images.rb +2 -2
  54. data/lib/llm/providers/google/models.rb +2 -2
  55. data/lib/llm/providers/google.rb +7 -7
  56. data/lib/llm/providers/mistral.rb +3 -1
  57. data/lib/llm/providers/ollama/error_handler.rb +5 -2
  58. data/lib/llm/providers/ollama/models.rb +2 -2
  59. data/lib/llm/providers/ollama.rb +7 -5
  60. data/lib/llm/providers/openai/audio.rb +6 -6
  61. data/lib/llm/providers/openai/error_handler.rb +5 -2
  62. data/lib/llm/providers/openai/files.rb +10 -10
  63. data/lib/llm/providers/openai/images.rb +4 -4
  64. data/lib/llm/providers/openai/models.rb +2 -2
  65. data/lib/llm/providers/openai/moderations.rb +2 -2
  66. data/lib/llm/providers/openai/request_adapter.rb +1 -1
  67. data/lib/llm/providers/openai/responses.rb +8 -8
  68. data/lib/llm/providers/openai/vector_stores.rb +22 -22
  69. data/lib/llm/providers/openai.rb +10 -8
  70. data/lib/llm/providers/xai/images.rb +4 -4
  71. data/lib/llm/schema.rb +24 -0
  72. data/lib/llm/tracer/telemetry.rb +4 -4
  73. data/lib/llm/tracer.rb +11 -3
  74. data/lib/llm/transport/execution.rb +8 -4
  75. data/lib/llm/utils.rb +13 -0
  76. data/lib/llm/version.rb +1 -1
  77. data/llm.gemspec +2 -2
  78. metadata +5 -5
  79. data/lib/llm/guard/loop.rb +0 -89
data/data/openrouter.json CHANGED
@@ -194,7 +194,7 @@
194
194
  "open_weights": true,
195
195
  "limit": {
196
196
  "context": 262144,
197
- "output": 235929
197
+ "output": 32768
198
198
  },
199
199
  "cost": {
200
200
  "input": 0.1,
@@ -316,11 +316,11 @@
316
316
  "open_weights": true,
317
317
  "limit": {
318
318
  "context": 131072,
319
- "output": 8192
319
+ "output": 16384
320
320
  },
321
321
  "cost": {
322
- "input": 0.2275,
323
- "output": 0.91
322
+ "input": 0.12,
323
+ "output": 0.24
324
324
  }
325
325
  },
326
326
  "qwen/qwen3.6-plus": {
@@ -838,11 +838,12 @@
838
838
  "open_weights": true,
839
839
  "limit": {
840
840
  "context": 262144,
841
- "output": 16384
841
+ "output": 235929
842
842
  },
843
843
  "cost": {
844
- "input": 0.22,
845
- "output": 0.88
844
+ "input": 0.0875,
845
+ "output": 0.35,
846
+ "cache_read": 0.0175
846
847
  }
847
848
  },
848
849
  "qwen/qwen3.5-flash-02-23": {
@@ -982,12 +983,12 @@
982
983
  "open_weights": true,
983
984
  "limit": {
984
985
  "context": 262144,
985
- "output": 65536
986
+ "output": 262140
986
987
  },
987
988
  "cost": {
988
- "input": 0.3,
989
- "output": 2,
990
- "cache_read": 0.03
989
+ "input": 0.32,
990
+ "output": 2.7,
991
+ "cache_read": 0.15
991
992
  }
992
993
  },
993
994
  "qwen/qwen3.7-flash": {
@@ -1087,6 +1088,48 @@
1087
1088
  "output": 2.4
1088
1089
  }
1089
1090
  },
1091
+ "qwen/qwen3.8-omni-flash": {
1092
+ "id": "qwen/qwen3.8-omni-flash",
1093
+ "name": "Qwen3.8 Omni Flash",
1094
+ "description": "Qwen omni model for text, vision, audio, and multimodal agent tasks",
1095
+ "family": "qwen",
1096
+ "attachment": true,
1097
+ "reasoning": true,
1098
+ "reasoning_options": [
1099
+ {
1100
+ "type": "toggle"
1101
+ },
1102
+ {
1103
+ "type": "budget_tokens"
1104
+ }
1105
+ ],
1106
+ "tool_call": true,
1107
+ "structured_output": true,
1108
+ "temperature": true,
1109
+ "release_date": "2026-09-17",
1110
+ "last_updated": "2026-09-17",
1111
+ "modalities": {
1112
+ "input": [
1113
+ "text",
1114
+ "image",
1115
+ "audio",
1116
+ "video"
1117
+ ],
1118
+ "output": [
1119
+ "text"
1120
+ ]
1121
+ },
1122
+ "open_weights": false,
1123
+ "limit": {
1124
+ "context": 1000000,
1125
+ "output": 131072
1126
+ },
1127
+ "cost": {
1128
+ "input": 0.15,
1129
+ "output": 0.47,
1130
+ "cache_read": 0.016
1131
+ }
1132
+ },
1090
1133
  "qwen/qwen3.6-35b-a3b": {
1091
1134
  "id": "qwen/qwen3.6-35b-a3b",
1092
1135
  "name": "Qwen3.6 35B-A3B",
@@ -1120,8 +1163,8 @@
1120
1163
  "output": 235929
1121
1164
  },
1122
1165
  "cost": {
1123
- "input": 0.1,
1124
- "output": 0.9,
1166
+ "input": 0.15,
1167
+ "output": 1,
1125
1168
  "cache_read": 0.05
1126
1169
  }
1127
1170
  },
@@ -1376,11 +1419,11 @@
1376
1419
  "open_weights": true,
1377
1420
  "limit": {
1378
1421
  "context": 262144,
1379
- "output": 16384
1422
+ "output": 32768
1380
1423
  },
1381
1424
  "cost": {
1382
- "input": 0.15,
1383
- "output": 0.6
1425
+ "input": 0.13,
1426
+ "output": 0.52
1384
1427
  }
1385
1428
  },
1386
1429
  "qwen/qwen-plus": {
@@ -1471,6 +1514,51 @@
1471
1514
  "output": 2.08
1472
1515
  }
1473
1516
  },
1517
+ "qwen/qwen3.8-27b:free": {
1518
+ "id": "qwen/qwen3.8-27b:free",
1519
+ "name": "Qwen3.8 27B (free)",
1520
+ "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
1521
+ "family": "qwen",
1522
+ "attachment": true,
1523
+ "reasoning": true,
1524
+ "reasoning_options": [
1525
+ {
1526
+ "type": "toggle"
1527
+ },
1528
+ {
1529
+ "type": "effort",
1530
+ "values": [
1531
+ "low",
1532
+ "medium",
1533
+ "xhigh"
1534
+ ]
1535
+ }
1536
+ ],
1537
+ "tool_call": true,
1538
+ "structured_output": true,
1539
+ "temperature": true,
1540
+ "release_date": "2026-08-14",
1541
+ "last_updated": "2026-08-14",
1542
+ "modalities": {
1543
+ "input": [
1544
+ "text",
1545
+ "image",
1546
+ "video"
1547
+ ],
1548
+ "output": [
1549
+ "text"
1550
+ ]
1551
+ },
1552
+ "open_weights": true,
1553
+ "limit": {
1554
+ "context": 262144,
1555
+ "output": 235929
1556
+ },
1557
+ "cost": {
1558
+ "input": 0,
1559
+ "output": 0
1560
+ }
1561
+ },
1474
1562
  "qwen/qwen-2.5-coder-32b-instruct": {
1475
1563
  "id": "qwen/qwen-2.5-coder-32b-instruct",
1476
1564
  "name": "Qwen2.5 Coder 32B Instruct",
@@ -2070,6 +2158,37 @@
2070
2158
  "output": 1.25
2071
2159
  }
2072
2160
  },
2161
+ "unbiased/pareto": {
2162
+ "id": "unbiased/pareto",
2163
+ "name": "Pareto",
2164
+ "description": "Blended multimodal model that runs multiple language models in parallel and returns a synthesized answer",
2165
+ "attachment": true,
2166
+ "reasoning": false,
2167
+ "tool_call": true,
2168
+ "structured_output": false,
2169
+ "temperature": true,
2170
+ "release_date": "2026-09-17",
2171
+ "last_updated": "2026-09-17",
2172
+ "modalities": {
2173
+ "input": [
2174
+ "text",
2175
+ "image"
2176
+ ],
2177
+ "output": [
2178
+ "text"
2179
+ ]
2180
+ },
2181
+ "open_weights": false,
2182
+ "limit": {
2183
+ "context": 262144,
2184
+ "output": 131072
2185
+ },
2186
+ "cost": {
2187
+ "input": 2.5,
2188
+ "output": 7.5,
2189
+ "cache_read": 0.25
2190
+ }
2191
+ },
2073
2192
  "aion-labs/aion-2.0": {
2074
2193
  "id": "aion-labs/aion-2.0",
2075
2194
  "name": "Aion-2.0",
@@ -2248,9 +2367,6 @@
2248
2367
  "attachment": true,
2249
2368
  "reasoning": true,
2250
2369
  "reasoning_options": [
2251
- {
2252
- "type": "toggle"
2253
- },
2254
2370
  {
2255
2371
  "type": "effort",
2256
2372
  "values": [
@@ -2283,15 +2399,15 @@
2283
2399
  "output": 128000
2284
2400
  },
2285
2401
  "cost": {
2286
- "input": 5,
2287
- "output": 25,
2288
- "cache_read": 0.5,
2289
- "cache_write": 6.25
2402
+ "input": 4,
2403
+ "output": 20,
2404
+ "cache_read": 0.2,
2405
+ "cache_write": 5
2290
2406
  }
2291
2407
  },
2292
2408
  "~anthropic/claude-haiku-latest": {
2293
2409
  "id": "~anthropic/claude-haiku-latest",
2294
- "name": "Anthropic Claude Haiku Latest",
2410
+ "name": "Claude Haiku Latest",
2295
2411
  "description": "Fast Claude model for responsive assistance, classification, and lightweight agents",
2296
2412
  "family": "claude-haiku",
2297
2413
  "attachment": true,
@@ -2330,7 +2446,7 @@
2330
2446
  },
2331
2447
  "~anthropic/claude-sonnet-latest": {
2332
2448
  "id": "~anthropic/claude-sonnet-latest",
2333
- "name": "Anthropic Claude Sonnet Latest",
2449
+ "name": "Claude Sonnet Latest",
2334
2450
  "description": "Balanced Claude model for coding, analysis, agent workflows, and cost control",
2335
2451
  "family": "claude-sonnet",
2336
2452
  "attachment": true,
@@ -2504,12 +2620,101 @@
2504
2620
  "open_weights": false,
2505
2621
  "limit": {
2506
2622
  "context": 1310720,
2507
- "output": 393216
2623
+ "output": 943718
2508
2624
  },
2509
2625
  "cost": {
2510
- "input": 0.05,
2511
- "output": 0.16,
2512
- "cache_read": 0.013
2626
+ "input": 0.03,
2627
+ "output": 0.8,
2628
+ "cache_read": 0.008
2629
+ }
2630
+ },
2631
+ "~deepseek/deepseek-pro-latest": {
2632
+ "id": "~deepseek/deepseek-pro-latest",
2633
+ "name": "DeepSeek Pro Latest",
2634
+ "description": "Flagship DeepSeek model for coding, reasoning, and agentic work",
2635
+ "family": "deepseek",
2636
+ "attachment": false,
2637
+ "reasoning": true,
2638
+ "reasoning_options": [
2639
+ {
2640
+ "type": "toggle"
2641
+ },
2642
+ {
2643
+ "type": "effort",
2644
+ "values": [
2645
+ "low",
2646
+ "high",
2647
+ "max"
2648
+ ]
2649
+ }
2650
+ ],
2651
+ "tool_call": true,
2652
+ "structured_output": true,
2653
+ "temperature": true,
2654
+ "release_date": "2026-09-14",
2655
+ "last_updated": "2026-09-14",
2656
+ "modalities": {
2657
+ "input": [
2658
+ "text"
2659
+ ],
2660
+ "output": [
2661
+ "text"
2662
+ ]
2663
+ },
2664
+ "open_weights": false,
2665
+ "limit": {
2666
+ "context": 1048576,
2667
+ "output": 384000
2668
+ },
2669
+ "cost": {
2670
+ "input": 0.4,
2671
+ "output": 4.3,
2672
+ "cache_read": 0.033
2673
+ }
2674
+ },
2675
+ "~deepseek/deepseek-flash-latest": {
2676
+ "id": "~deepseek/deepseek-flash-latest",
2677
+ "name": "DeepSeek Flash Latest",
2678
+ "description": "Fast DeepSeek model for efficient chat, coding help, and agent loops",
2679
+ "family": "deepseek-flash",
2680
+ "attachment": true,
2681
+ "reasoning": true,
2682
+ "reasoning_options": [
2683
+ {
2684
+ "type": "toggle"
2685
+ },
2686
+ {
2687
+ "type": "effort",
2688
+ "values": [
2689
+ "low",
2690
+ "high",
2691
+ "max"
2692
+ ]
2693
+ }
2694
+ ],
2695
+ "tool_call": true,
2696
+ "structured_output": true,
2697
+ "temperature": true,
2698
+ "release_date": "2026-09-14",
2699
+ "last_updated": "2026-09-14",
2700
+ "modalities": {
2701
+ "input": [
2702
+ "text",
2703
+ "image"
2704
+ ],
2705
+ "output": [
2706
+ "text"
2707
+ ]
2708
+ },
2709
+ "open_weights": false,
2710
+ "limit": {
2711
+ "context": 1048576,
2712
+ "output": 943718
2713
+ },
2714
+ "cost": {
2715
+ "input": 0.1,
2716
+ "output": 0.5,
2717
+ "cache_read": 0.01
2513
2718
  }
2514
2719
  },
2515
2720
  "dots-studio/dots-3-note-preview:free": {
@@ -2586,14 +2791,14 @@
2586
2791
  "output": 450000
2587
2792
  },
2588
2793
  "cost": {
2589
- "input": 2,
2590
- "output": 6,
2591
- "cache_read": 0.5,
2794
+ "input": 1.6,
2795
+ "output": 4.8,
2796
+ "cache_read": 0.4,
2592
2797
  "tiers": [
2593
2798
  {
2594
- "input": 4,
2595
- "output": 12,
2596
- "cache_read": 1,
2799
+ "input": 3.2,
2800
+ "output": 9.6,
2801
+ "cache_read": 0.8,
2597
2802
  "tier": {
2598
2803
  "type": "context",
2599
2804
  "size": 200000
@@ -2601,9 +2806,9 @@
2601
2806
  }
2602
2807
  ],
2603
2808
  "context_over_200k": {
2604
- "input": 4,
2605
- "output": 12,
2606
- "cache_read": 1
2809
+ "input": 3.2,
2810
+ "output": 9.6,
2811
+ "cache_read": 0.8
2607
2812
  }
2608
2813
  }
2609
2814
  },
@@ -2646,26 +2851,33 @@
2646
2851
  "cache_read": 0.006
2647
2852
  }
2648
2853
  },
2649
- "poolside/laguna-xs-2.1": {
2650
- "id": "poolside/laguna-xs-2.1",
2651
- "name": "Laguna XS 2.1",
2652
- "description": "Agentic coding model from Poolside in the XS size class for local deployment",
2653
- "family": "laguna",
2654
- "attachment": false,
2854
+ "prism-ml/ternary-bonsai-2-27b": {
2855
+ "id": "prism-ml/ternary-bonsai-2-27b",
2856
+ "name": "Ternary Bonsai 2 27B",
2857
+ "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
2858
+ "attachment": true,
2655
2859
  "reasoning": true,
2656
2860
  "reasoning_options": [
2657
2861
  {
2658
2862
  "type": "toggle"
2863
+ },
2864
+ {
2865
+ "type": "effort",
2866
+ "values": [
2867
+ "medium",
2868
+ "xhigh"
2869
+ ]
2659
2870
  }
2660
2871
  ],
2661
2872
  "tool_call": true,
2662
- "structured_output": false,
2873
+ "structured_output": true,
2663
2874
  "temperature": true,
2664
- "release_date": "2026-07-02",
2665
- "last_updated": "2026-07-02",
2875
+ "release_date": "2026-09-18",
2876
+ "last_updated": "2026-09-18",
2666
2877
  "modalities": {
2667
2878
  "input": [
2668
- "text"
2879
+ "text",
2880
+ "image"
2669
2881
  ],
2670
2882
  "output": [
2671
2883
  "text"
@@ -2677,17 +2889,52 @@
2677
2889
  "output": 32768
2678
2890
  },
2679
2891
  "cost": {
2680
- "input": 0.06,
2681
- "output": 0.12,
2682
- "cache_read": 0.03
2892
+ "input": 0.075,
2893
+ "output": 0.5
2683
2894
  }
2684
2895
  },
2685
- "poolside/laguna-xs-2.1:free": {
2686
- "id": "poolside/laguna-xs-2.1:free",
2687
- "name": "Laguna XS 2.1 (free)",
2688
- "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads",
2689
- "family": "laguna",
2690
- "attachment": false,
2896
+ "poolside/laguna-xs-2.1": {
2897
+ "id": "poolside/laguna-xs-2.1",
2898
+ "name": "Laguna XS 2.1",
2899
+ "description": "Agentic coding model from Poolside in the XS size class for local deployment",
2900
+ "family": "laguna",
2901
+ "attachment": false,
2902
+ "reasoning": true,
2903
+ "reasoning_options": [
2904
+ {
2905
+ "type": "toggle"
2906
+ }
2907
+ ],
2908
+ "tool_call": true,
2909
+ "structured_output": false,
2910
+ "temperature": true,
2911
+ "release_date": "2026-07-02",
2912
+ "last_updated": "2026-07-02",
2913
+ "modalities": {
2914
+ "input": [
2915
+ "text"
2916
+ ],
2917
+ "output": [
2918
+ "text"
2919
+ ]
2920
+ },
2921
+ "open_weights": true,
2922
+ "limit": {
2923
+ "context": 262144,
2924
+ "output": 32768
2925
+ },
2926
+ "cost": {
2927
+ "input": 0.06,
2928
+ "output": 0.12,
2929
+ "cache_read": 0.03
2930
+ }
2931
+ },
2932
+ "poolside/laguna-xs-2.1:free": {
2933
+ "id": "poolside/laguna-xs-2.1:free",
2934
+ "name": "Laguna XS 2.1 (free)",
2935
+ "description": "Free provider route for experiments, demos, and cost-sensitive chat workloads",
2936
+ "family": "laguna",
2937
+ "attachment": false,
2691
2938
  "reasoning": true,
2692
2939
  "reasoning_options": [
2693
2940
  {
@@ -2788,37 +3035,6 @@
2788
3035
  "cache_read": 0.009
2789
3036
  }
2790
3037
  },
2791
- "kwaipilot/kat-coder-pro-v2": {
2792
- "id": "kwaipilot/kat-coder-pro-v2",
2793
- "name": "KAT-Coder-Pro V2",
2794
- "description": "Coding model for repository understanding, refactors, and agentic engineering tasks",
2795
- "family": "kat-coder",
2796
- "attachment": false,
2797
- "reasoning": false,
2798
- "tool_call": true,
2799
- "structured_output": true,
2800
- "temperature": true,
2801
- "release_date": "2026-03-27",
2802
- "last_updated": "2026-03-27",
2803
- "modalities": {
2804
- "input": [
2805
- "text"
2806
- ],
2807
- "output": [
2808
- "text"
2809
- ]
2810
- },
2811
- "open_weights": false,
2812
- "limit": {
2813
- "context": 262144,
2814
- "output": 144000
2815
- },
2816
- "cost": {
2817
- "input": 0.3,
2818
- "output": 1.2,
2819
- "cache_read": 0.06
2820
- }
2821
- },
2822
3038
  "kwaipilot/kat-coder-pro-v2.5": {
2823
3039
  "id": "kwaipilot/kat-coder-pro-v2.5",
2824
3040
  "name": "KAT-Coder-Pro V2.5",
@@ -3154,12 +3370,12 @@
3154
3370
  },
3155
3371
  "open_weights": true,
3156
3372
  "limit": {
3157
- "context": 131072,
3373
+ "context": 256000,
3158
3374
  "output": 16384
3159
3375
  },
3160
3376
  "cost": {
3161
- "input": 0.075,
3162
- "output": 0.2
3377
+ "input": 0.09375,
3378
+ "output": 0.25
3163
3379
  }
3164
3380
  },
3165
3381
  "mistralai/mixtral-8x22b-instruct": {
@@ -3228,40 +3444,6 @@
3228
3444
  "cache_read": 0.02
3229
3445
  }
3230
3446
  },
3231
- "mistralai/mistral-large-2512": {
3232
- "id": "mistralai/mistral-large-2512",
3233
- "name": "Mistral Large 3",
3234
- "description": "Mistral's largest general model for enterprise agents, coding, and multilingual reasoning",
3235
- "family": "mistral-large",
3236
- "attachment": true,
3237
- "reasoning": false,
3238
- "tool_call": true,
3239
- "structured_output": true,
3240
- "temperature": true,
3241
- "knowledge": "2024-11",
3242
- "release_date": "2024-11-01",
3243
- "last_updated": "2025-12-02",
3244
- "modalities": {
3245
- "input": [
3246
- "text",
3247
- "image",
3248
- "pdf"
3249
- ],
3250
- "output": [
3251
- "text"
3252
- ]
3253
- },
3254
- "open_weights": true,
3255
- "limit": {
3256
- "context": 262144,
3257
- "output": 209715
3258
- },
3259
- "cost": {
3260
- "input": 0.5,
3261
- "output": 1.5,
3262
- "cache_read": 0.05
3263
- }
3264
- },
3265
3447
  "mistralai/ministral-3b-2512": {
3266
3448
  "id": "mistralai/ministral-3b-2512",
3267
3449
  "name": "Ministral 3 3B 2512",
@@ -3397,7 +3579,7 @@
3397
3579
  "family": "mistral-small",
3398
3580
  "attachment": true,
3399
3581
  "reasoning": false,
3400
- "tool_call": false,
3582
+ "tool_call": true,
3401
3583
  "structured_output": false,
3402
3584
  "temperature": true,
3403
3585
  "knowledge": "2023-10-31",
@@ -3500,14 +3682,14 @@
3500
3682
  "id": "mistralai/voxtral-small-24b-2507",
3501
3683
  "name": "Voxtral Small 24B 2507",
3502
3684
  "description": "Efficient Mistral model for fast chat, extraction, and production assistants",
3503
- "family": "mistral",
3685
+ "family": "voxtral",
3504
3686
  "attachment": true,
3505
3687
  "reasoning": false,
3506
3688
  "tool_call": true,
3507
3689
  "structured_output": true,
3508
3690
  "temperature": true,
3509
- "release_date": "2025-10-30",
3510
- "last_updated": "2025-10-30",
3691
+ "release_date": "2025-07-15",
3692
+ "last_updated": "2025-07-15",
3511
3693
  "modalities": {
3512
3694
  "input": [
3513
3695
  "text",
@@ -3563,6 +3745,125 @@
3563
3745
  "cache_read": 0.04
3564
3746
  }
3565
3747
  },
3748
+ "xiaomi/mimo-v2.6-pro": {
3749
+ "id": "xiaomi/mimo-v2.6-pro",
3750
+ "name": "MiMo-V2.6-Pro",
3751
+ "description": "MiMo pro model for strong multimodal reasoning and agent execution",
3752
+ "family": "mimo",
3753
+ "attachment": true,
3754
+ "reasoning": true,
3755
+ "reasoning_options": [
3756
+ {
3757
+ "type": "toggle"
3758
+ }
3759
+ ],
3760
+ "tool_call": true,
3761
+ "structured_output": true,
3762
+ "temperature": true,
3763
+ "knowledge": "2026-09-22",
3764
+ "release_date": "2026-09-22",
3765
+ "last_updated": "2026-09-22",
3766
+ "modalities": {
3767
+ "input": [
3768
+ "text",
3769
+ "image",
3770
+ "video",
3771
+ "audio"
3772
+ ],
3773
+ "output": [
3774
+ "text"
3775
+ ]
3776
+ },
3777
+ "open_weights": false,
3778
+ "limit": {
3779
+ "context": 1048576,
3780
+ "output": 131072
3781
+ },
3782
+ "cost": {
3783
+ "input": 0.435,
3784
+ "output": 0.87,
3785
+ "cache_read": 0.0036
3786
+ }
3787
+ },
3788
+ "xiaomi/mimo-v2.6-pro-ultraspeed": {
3789
+ "id": "xiaomi/mimo-v2.6-pro-ultraspeed",
3790
+ "name": "MiMo-V2.6-Pro-UltraSpeed",
3791
+ "description": "MiMo pro model for strong multimodal reasoning and agent execution",
3792
+ "family": "mimo",
3793
+ "attachment": true,
3794
+ "reasoning": true,
3795
+ "reasoning_options": [
3796
+ {
3797
+ "type": "toggle"
3798
+ }
3799
+ ],
3800
+ "tool_call": true,
3801
+ "structured_output": true,
3802
+ "temperature": true,
3803
+ "release_date": "2026-09-21",
3804
+ "last_updated": "2026-09-21",
3805
+ "modalities": {
3806
+ "input": [
3807
+ "text",
3808
+ "image",
3809
+ "video",
3810
+ "audio"
3811
+ ],
3812
+ "output": [
3813
+ "text"
3814
+ ]
3815
+ },
3816
+ "open_weights": false,
3817
+ "limit": {
3818
+ "context": 1048576,
3819
+ "output": 131072
3820
+ },
3821
+ "cost": {
3822
+ "input": 4.35,
3823
+ "output": 8.7,
3824
+ "cache_read": 0.036
3825
+ }
3826
+ },
3827
+ "xiaomi/mimo-v2.6-flash": {
3828
+ "id": "xiaomi/mimo-v2.6-flash",
3829
+ "name": "MiMo-V2.6-Flash",
3830
+ "description": "MiMo flash model for fast multimodal assistance and agent workflows",
3831
+ "family": "mimo",
3832
+ "attachment": true,
3833
+ "reasoning": true,
3834
+ "reasoning_options": [
3835
+ {
3836
+ "type": "toggle"
3837
+ }
3838
+ ],
3839
+ "tool_call": true,
3840
+ "structured_output": true,
3841
+ "temperature": true,
3842
+ "knowledge": "2026-09-22",
3843
+ "release_date": "2026-09-22",
3844
+ "last_updated": "2026-09-22",
3845
+ "modalities": {
3846
+ "input": [
3847
+ "text",
3848
+ "image",
3849
+ "video",
3850
+ "audio"
3851
+ ],
3852
+ "output": [
3853
+ "text"
3854
+ ]
3855
+ },
3856
+ "open_weights": false,
3857
+ "limit": {
3858
+ "context": 1048576,
3859
+ "output": 131072
3860
+ },
3861
+ "cost": {
3862
+ "input": 0.14,
3863
+ "output": 0.28,
3864
+ "cache_read": 0.0028
3865
+ }
3866
+ },
3566
3867
  "xiaomi/mimo-v2.5": {
3567
3868
  "id": "xiaomi/mimo-v2.5",
3568
3869
  "name": "MiMo-V2.5",
@@ -3883,7 +4184,7 @@
3883
4184
  "output": 40000
3884
4185
  },
3885
4186
  "cost": {
3886
- "input": 0.55,
4187
+ "input": 0.4,
3887
4188
  "output": 2.2
3888
4189
  }
3889
4190
  },
@@ -4018,10 +4319,10 @@
4018
4319
  "open_weights": true,
4019
4320
  "limit": {
4020
4321
  "context": 262144,
4021
- "output": 131072
4322
+ "output": 235929
4022
4323
  },
4023
4324
  "cost": {
4024
- "input": 0.08,
4325
+ "input": 0.07,
4025
4326
  "output": 0.2,
4026
4327
  "cache_read": 0.04
4027
4328
  }
@@ -4090,7 +4391,7 @@
4090
4391
  }
4091
4392
  ],
4092
4393
  "tool_call": true,
4093
- "structured_output": false,
4394
+ "structured_output": true,
4094
4395
  "temperature": true,
4095
4396
  "release_date": "2026-03-11",
4096
4397
  "last_updated": "2026-03-11",
@@ -4105,11 +4406,11 @@
4105
4406
  "open_weights": true,
4106
4407
  "limit": {
4107
4408
  "context": 262144,
4108
- "output": 16384
4409
+ "output": 235929
4109
4410
  },
4110
4411
  "cost": {
4111
- "input": 0.085,
4112
- "output": 0.4
4412
+ "input": 0.08,
4413
+ "output": 0.45
4113
4414
  }
4114
4415
  },
4115
4416
  "nvidia/nemotron-3-ultra-550b-a55b:free": {
@@ -4276,12 +4577,12 @@
4276
4577
  "open_weights": true,
4277
4578
  "limit": {
4278
4579
  "context": 262144,
4279
- "output": 32768
4580
+ "output": 182520
4280
4581
  },
4281
4582
  "cost": {
4282
- "input": 0.625,
4283
- "output": 3.125,
4284
- "cache_read": 0.1875
4583
+ "input": 0.6,
4584
+ "output": 2.4,
4585
+ "cache_read": 0.12
4285
4586
  }
4286
4587
  },
4287
4588
  "nvidia/nemotron-3-nano-30b-a3b": {
@@ -4595,6 +4896,53 @@
4595
4896
  }
4596
4897
  }
4597
4898
  },
4899
+ "anthropic/claude-opus-5.5": {
4900
+ "id": "anthropic/claude-opus-5.5",
4901
+ "name": "Claude Opus 5.5",
4902
+ "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
4903
+ "family": "claude-opus",
4904
+ "attachment": true,
4905
+ "reasoning": true,
4906
+ "reasoning_options": [
4907
+ {
4908
+ "type": "effort",
4909
+ "values": [
4910
+ "low",
4911
+ "medium",
4912
+ "high",
4913
+ "xhigh",
4914
+ "max"
4915
+ ]
4916
+ }
4917
+ ],
4918
+ "tool_call": true,
4919
+ "structured_output": true,
4920
+ "temperature": true,
4921
+ "knowledge": "2026-06",
4922
+ "release_date": "2026-09-22",
4923
+ "last_updated": "2026-09-22",
4924
+ "modalities": {
4925
+ "input": [
4926
+ "text",
4927
+ "image",
4928
+ "pdf"
4929
+ ],
4930
+ "output": [
4931
+ "text"
4932
+ ]
4933
+ },
4934
+ "open_weights": false,
4935
+ "limit": {
4936
+ "context": 1000000,
4937
+ "output": 128000
4938
+ },
4939
+ "cost": {
4940
+ "input": 4,
4941
+ "output": 20,
4942
+ "cache_read": 0.2,
4943
+ "cache_write": 5
4944
+ }
4945
+ },
4598
4946
  "anthropic/claude-3-haiku": {
4599
4947
  "id": "anthropic/claude-3-haiku",
4600
4948
  "name": "Claude 3 Haiku",
@@ -4786,46 +5134,6 @@
4786
5134
  "cache_write": 12.5
4787
5135
  }
4788
5136
  },
4789
- "anthropic/claude-opus-4": {
4790
- "id": "anthropic/claude-opus-4",
4791
- "name": "Claude Opus 4",
4792
- "description": "Flagship Claude model for deep reasoning, coding, and long-horizon agents",
4793
- "family": "claude-opus",
4794
- "attachment": true,
4795
- "reasoning": true,
4796
- "reasoning_options": [
4797
- {
4798
- "type": "toggle"
4799
- }
4800
- ],
4801
- "tool_call": true,
4802
- "structured_output": false,
4803
- "temperature": true,
4804
- "knowledge": "2025-01-31",
4805
- "release_date": "2025-05-22",
4806
- "last_updated": "2025-05-22",
4807
- "modalities": {
4808
- "input": [
4809
- "image",
4810
- "text",
4811
- "pdf"
4812
- ],
4813
- "output": [
4814
- "text"
4815
- ]
4816
- },
4817
- "open_weights": false,
4818
- "limit": {
4819
- "context": 200000,
4820
- "output": 32000
4821
- },
4822
- "cost": {
4823
- "input": 15,
4824
- "output": 75,
4825
- "cache_read": 1.5,
4826
- "cache_write": 18.75
4827
- }
4828
- },
4829
5137
  "anthropic/claude-sonnet-4.5": {
4830
5138
  "id": "anthropic/claude-sonnet-4.5",
4831
5139
  "name": "Claude Sonnet 4.5 (latest)",
@@ -4954,7 +5262,7 @@
4954
5262
  },
4955
5263
  "open_weights": false,
4956
5264
  "limit": {
4957
- "context": 1000000,
5265
+ "context": 200000,
4958
5266
  "output": 64000
4959
5267
  },
4960
5268
  "cost": {
@@ -5109,11 +5417,12 @@
5109
5417
  "open_weights": true,
5110
5418
  "limit": {
5111
5419
  "context": 262144,
5112
- "output": 32768
5420
+ "output": 235929
5113
5421
  },
5114
5422
  "cost": {
5115
- "input": 0.042,
5116
- "output": 0.22
5423
+ "input": 0.09,
5424
+ "output": 0.3,
5425
+ "cache_read": 0.05
5117
5426
  }
5118
5427
  },
5119
5428
  "google/gemini-3.1-pro-preview-customtools": {
@@ -5230,7 +5539,7 @@
5230
5539
  },
5231
5540
  "google/gemma-3-4b-it": {
5232
5541
  "id": "google/gemma-3-4b-it",
5233
- "name": "Gemma 3 4B",
5542
+ "name": "Gemma 3 4B IT",
5234
5543
  "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
5235
5544
  "family": "gemma",
5236
5545
  "attachment": true,
@@ -5238,9 +5547,9 @@
5238
5547
  "tool_call": false,
5239
5548
  "structured_output": true,
5240
5549
  "temperature": true,
5241
- "knowledge": "2024-08-31",
5242
- "release_date": "2025-03-13",
5243
- "last_updated": "2025-03-13",
5550
+ "knowledge": "2024-08",
5551
+ "release_date": "2025-03-12",
5552
+ "last_updated": "2025-03-12",
5244
5553
  "modalities": {
5245
5554
  "input": [
5246
5555
  "text",
@@ -5678,7 +5987,7 @@
5678
5987
  },
5679
5988
  "google/gemma-3-27b-it": {
5680
5989
  "id": "google/gemma-3-27b-it",
5681
- "name": "Gemma 3 27B",
5990
+ "name": "Gemma 3 27B IT",
5682
5991
  "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
5683
5992
  "family": "gemma",
5684
5993
  "attachment": true,
@@ -5686,7 +5995,7 @@
5686
5995
  "tool_call": true,
5687
5996
  "structured_output": true,
5688
5997
  "temperature": true,
5689
- "knowledge": "2024-08-31",
5998
+ "knowledge": "2024-08",
5690
5999
  "release_date": "2025-03-12",
5691
6000
  "last_updated": "2025-03-12",
5692
6001
  "modalities": {
@@ -6306,70 +6615,9 @@
6306
6615
  "cache_write": 0.083333
6307
6616
  }
6308
6617
  },
6309
- "google/gemini-2.5-pro-preview-05-06": {
6310
- "id": "google/gemini-2.5-pro-preview-05-06",
6311
- "name": "Gemini 2.5 Pro Preview 05-06",
6312
- "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
6313
- "family": "gemini-pro",
6314
- "attachment": true,
6315
- "reasoning": true,
6316
- "reasoning_options": [
6317
- {
6318
- "type": "budget_tokens",
6319
- "min": 128,
6320
- "max": 32768
6321
- }
6322
- ],
6323
- "tool_call": true,
6324
- "structured_output": true,
6325
- "temperature": true,
6326
- "knowledge": "2025-01-31",
6327
- "release_date": "2025-05-07",
6328
- "last_updated": "2025-05-07",
6329
- "modalities": {
6330
- "input": [
6331
- "text",
6332
- "image",
6333
- "pdf",
6334
- "audio",
6335
- "video"
6336
- ],
6337
- "output": [
6338
- "text"
6339
- ]
6340
- },
6341
- "open_weights": false,
6342
- "limit": {
6343
- "context": 1048576,
6344
- "output": 65535
6345
- },
6346
- "cost": {
6347
- "input": 1.25,
6348
- "output": 10,
6349
- "reasoning": 10,
6350
- "cache_read": 0.125,
6351
- "cache_write": 0.375,
6352
- "tiers": [
6353
- {
6354
- "input": 2.5,
6355
- "output": 15,
6356
- "cache_read": 0.25,
6357
- "tier": {
6358
- "type": "context",
6359
- "size": 200000
6360
- }
6361
- }
6362
- ],
6363
- "context_over_200k": {
6364
- "input": 2.5,
6365
- "output": 15,
6366
- "cache_read": 0.25
6367
- }
6368
- }
6369
- },
6370
6618
  "google/gemma-3-12b-it": {
6371
6619
  "id": "google/gemma-3-12b-it",
6372
- "name": "Gemma 3 12B",
6620
+ "name": "Gemma 3 12B IT",
6373
6621
  "description": "Open Gemma instruction model for efficient chat and self-hosted deployments",
6374
6622
  "family": "gemma",
6375
6623
  "attachment": true,
@@ -6377,9 +6625,9 @@
6377
6625
  "tool_call": true,
6378
6626
  "structured_output": true,
6379
6627
  "temperature": true,
6380
- "knowledge": "2024-08-31",
6381
- "release_date": "2025-03-13",
6382
- "last_updated": "2025-03-13",
6628
+ "knowledge": "2024-08",
6629
+ "release_date": "2025-03-12",
6630
+ "last_updated": "2025-03-12",
6383
6631
  "modalities": {
6384
6632
  "input": [
6385
6633
  "text",
@@ -6566,6 +6814,48 @@
6566
6814
  "output": 0
6567
6815
  }
6568
6816
  },
6817
+ "nex-agi/nex-n2.5-pro": {
6818
+ "id": "nex-agi/nex-n2.5-pro",
6819
+ "name": "Nex-N2.5-Pro",
6820
+ "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
6821
+ "family": "agi",
6822
+ "attachment": true,
6823
+ "reasoning": true,
6824
+ "reasoning_options": [
6825
+ {
6826
+ "type": "effort",
6827
+ "values": [
6828
+ "none",
6829
+ "medium",
6830
+ "high"
6831
+ ]
6832
+ }
6833
+ ],
6834
+ "tool_call": true,
6835
+ "structured_output": true,
6836
+ "temperature": true,
6837
+ "release_date": "2026-09-08",
6838
+ "last_updated": "2026-09-08",
6839
+ "modalities": {
6840
+ "input": [
6841
+ "text",
6842
+ "image"
6843
+ ],
6844
+ "output": [
6845
+ "text"
6846
+ ]
6847
+ },
6848
+ "open_weights": true,
6849
+ "limit": {
6850
+ "context": 262144,
6851
+ "output": 235929
6852
+ },
6853
+ "cost": {
6854
+ "input": 0.075,
6855
+ "output": 0.25,
6856
+ "cache_read": 0.015
6857
+ }
6858
+ },
6569
6859
  "nex-agi/nex-n2.5-pro:free": {
6570
6860
  "id": "nex-agi/nex-n2.5-pro:free",
6571
6861
  "name": "Nex-N2.5-Pro (free)",
@@ -6607,6 +6897,48 @@
6607
6897
  "output": 0
6608
6898
  }
6609
6899
  },
6900
+ "nex-agi/nex-n2.5-mini": {
6901
+ "id": "nex-agi/nex-n2.5-mini",
6902
+ "name": "Nex-N2.5-Mini",
6903
+ "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
6904
+ "family": "agi",
6905
+ "attachment": true,
6906
+ "reasoning": true,
6907
+ "reasoning_options": [
6908
+ {
6909
+ "type": "effort",
6910
+ "values": [
6911
+ "none",
6912
+ "medium",
6913
+ "high"
6914
+ ]
6915
+ }
6916
+ ],
6917
+ "tool_call": false,
6918
+ "structured_output": true,
6919
+ "temperature": true,
6920
+ "release_date": "2026-09-08",
6921
+ "last_updated": "2026-09-08",
6922
+ "modalities": {
6923
+ "input": [
6924
+ "text",
6925
+ "image"
6926
+ ],
6927
+ "output": [
6928
+ "text"
6929
+ ]
6930
+ },
6931
+ "open_weights": true,
6932
+ "limit": {
6933
+ "context": 262144,
6934
+ "output": 235929
6935
+ },
6936
+ "cost": {
6937
+ "input": 0.025,
6938
+ "output": 0.1,
6939
+ "cache_read": 0.0025
6940
+ }
6941
+ },
6610
6942
  "thinkingmachines/inkling-small": {
6611
6943
  "id": "thinkingmachines/inkling-small",
6612
6944
  "name": "Inkling Small",
@@ -6781,7 +7113,7 @@
6781
7113
  "open_weights": true,
6782
7114
  "limit": {
6783
7115
  "context": 1048576,
6784
- "output": 32768
7116
+ "output": 471859
6785
7117
  },
6786
7118
  "cost": {
6787
7119
  "input": 1,
@@ -6815,8 +7147,8 @@
6815
7147
  "output": 3686
6816
7148
  },
6817
7149
  "cost": {
6818
- "input": 0.06,
6819
- "output": 0.06
7150
+ "input": 0.08,
7151
+ "output": 0.11
6820
7152
  }
6821
7153
  },
6822
7154
  "meta/muse-spark-1.3": {
@@ -7092,11 +7424,11 @@
7092
7424
  "open_weights": true,
7093
7425
  "limit": {
7094
7426
  "context": 131072,
7095
- "output": 117964
7427
+ "output": 16384
7096
7428
  },
7097
7429
  "cost": {
7098
7430
  "input": 0.3,
7099
- "output": 1.1,
7431
+ "output": 1.2,
7100
7432
  "cache_read": 0.04
7101
7433
  }
7102
7434
  },
@@ -7173,7 +7505,7 @@
7173
7505
  "description": "Open-weight instruction model for adaptable chat and self-hosted production workloads",
7174
7506
  "attachment": false,
7175
7507
  "reasoning": false,
7176
- "tool_call": true,
7508
+ "tool_call": false,
7177
7509
  "structured_output": true,
7178
7510
  "temperature": true,
7179
7511
  "knowledge": "2024-04-30",
@@ -7674,7 +8006,7 @@
7674
8006
  },
7675
8007
  "~google/gemini-pro-latest": {
7676
8008
  "id": "~google/gemini-pro-latest",
7677
- "name": "Google Gemini Pro Latest",
8009
+ "name": "Gemini Pro Latest",
7678
8010
  "description": "Advanced Gemini model for complex reasoning, coding, and multimodal analysis",
7679
8011
  "family": "gemini-pro",
7680
8012
  "attachment": true,
@@ -7738,7 +8070,7 @@
7738
8070
  },
7739
8071
  "~google/gemini-flash-latest": {
7740
8072
  "id": "~google/gemini-flash-latest",
7741
- "name": "Google Gemini Flash Latest",
8073
+ "name": "Gemini Flash Latest",
7742
8074
  "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
7743
8075
  "family": "gemini-flash",
7744
8076
  "attachment": true,
@@ -7903,6 +8235,49 @@
7903
8235
  }
7904
8236
  }
7905
8237
  },
8238
+ "sakana/fugu-max": {
8239
+ "id": "sakana/fugu-max",
8240
+ "name": "Fugu Max",
8241
+ "description": "Multi-agent model for routing expert agents across complex analytical tasks",
8242
+ "family": "fugu",
8243
+ "attachment": true,
8244
+ "reasoning": true,
8245
+ "reasoning_options": [
8246
+ {
8247
+ "type": "effort",
8248
+ "values": [
8249
+ "high",
8250
+ "xhigh",
8251
+ "max"
8252
+ ]
8253
+ }
8254
+ ],
8255
+ "tool_call": true,
8256
+ "structured_output": true,
8257
+ "temperature": false,
8258
+ "release_date": "2026-09-11",
8259
+ "last_updated": "2026-09-11",
8260
+ "modalities": {
8261
+ "input": [
8262
+ "text",
8263
+ "image",
8264
+ "pdf"
8265
+ ],
8266
+ "output": [
8267
+ "text"
8268
+ ]
8269
+ },
8270
+ "open_weights": false,
8271
+ "limit": {
8272
+ "context": 1000000,
8273
+ "output": 128000
8274
+ },
8275
+ "cost": {
8276
+ "input": 2,
8277
+ "output": 6,
8278
+ "cache_read": 0.25
8279
+ }
8280
+ },
7906
8281
  "sakana/sakana-namazu": {
7907
8282
  "id": "sakana/sakana-namazu",
7908
8283
  "name": "Sakana Namazu",
@@ -7945,36 +8320,34 @@
7945
8320
  "cache_read": 0.15
7946
8321
  }
7947
8322
  },
7948
- "~moonshotai/kimi-latest": {
7949
- "id": "~moonshotai/kimi-latest",
7950
- "name": "MoonshotAI Kimi Latest",
7951
- "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
7952
- "family": "kimi",
8323
+ "sakana/fugu-ultra-v2": {
8324
+ "id": "sakana/fugu-ultra-v2",
8325
+ "name": "Fugu Ultra v2",
8326
+ "description": "Quality-first multi-agent model for hard research, analysis, and competitions",
8327
+ "family": "fugu",
7953
8328
  "attachment": true,
7954
8329
  "reasoning": true,
7955
8330
  "reasoning_options": [
7956
- {
7957
- "type": "toggle"
7958
- },
7959
8331
  {
7960
8332
  "type": "effort",
7961
8333
  "values": [
7962
- "low",
7963
8334
  "high",
8335
+ "xhigh",
7964
8336
  "max"
7965
8337
  ]
7966
8338
  }
7967
8339
  ],
7968
8340
  "tool_call": true,
7969
8341
  "structured_output": true,
7970
- "temperature": true,
7971
- "release_date": "2026-04-27",
7972
- "last_updated": "2026-04-27",
8342
+ "temperature": false,
8343
+ "knowledge": "2026-08-28",
8344
+ "release_date": "2026-09-11",
8345
+ "last_updated": "2026-09-11",
7973
8346
  "modalities": {
7974
8347
  "input": [
7975
8348
  "text",
7976
8349
  "image",
7977
- "video"
8350
+ "pdf"
7978
8351
  ],
7979
8352
  "output": [
7980
8353
  "text"
@@ -7982,13 +8355,75 @@
7982
8355
  },
7983
8356
  "open_weights": false,
7984
8357
  "limit": {
7985
- "context": 1048576,
7986
- "output": 943718
7987
- },
8358
+ "context": 1000000,
8359
+ "output": 128000
8360
+ },
7988
8361
  "cost": {
7989
- "input": 2.4,
7990
- "output": 12,
7991
- "cache_read": 0.24
8362
+ "input": 5,
8363
+ "output": 30,
8364
+ "cache_read": 0.5,
8365
+ "tiers": [
8366
+ {
8367
+ "input": 10,
8368
+ "output": 45,
8369
+ "cache_read": 1,
8370
+ "tier": {
8371
+ "type": "context",
8372
+ "size": 272000
8373
+ }
8374
+ }
8375
+ ],
8376
+ "context_over_200k": {
8377
+ "input": 10,
8378
+ "output": 45,
8379
+ "cache_read": 1
8380
+ }
8381
+ }
8382
+ },
8383
+ "~moonshotai/kimi-latest": {
8384
+ "id": "~moonshotai/kimi-latest",
8385
+ "name": "Kimi Latest",
8386
+ "description": "Kimi multimodal agent model for visual understanding, coding, and planning",
8387
+ "family": "kimi",
8388
+ "attachment": true,
8389
+ "reasoning": true,
8390
+ "reasoning_options": [
8391
+ {
8392
+ "type": "toggle"
8393
+ },
8394
+ {
8395
+ "type": "effort",
8396
+ "values": [
8397
+ "low",
8398
+ "high",
8399
+ "max"
8400
+ ]
8401
+ }
8402
+ ],
8403
+ "tool_call": true,
8404
+ "structured_output": true,
8405
+ "temperature": true,
8406
+ "release_date": "2026-04-27",
8407
+ "last_updated": "2026-04-27",
8408
+ "modalities": {
8409
+ "input": [
8410
+ "text",
8411
+ "image",
8412
+ "video"
8413
+ ],
8414
+ "output": [
8415
+ "text"
8416
+ ]
8417
+ },
8418
+ "open_weights": false,
8419
+ "limit": {
8420
+ "context": 1048576,
8421
+ "output": 943718
8422
+ },
8423
+ "cost": {
8424
+ "input": 1.05,
8425
+ "output": 13,
8426
+ "cache_read": 0.3
7992
8427
  }
7993
8428
  },
7994
8429
  "ibm-granite/granite-4.2-8b": {
@@ -8123,7 +8558,7 @@
8123
8558
  "structured_output": true,
8124
8559
  "temperature": true,
8125
8560
  "release_date": "2026-08-21",
8126
- "last_updated": "2026-08-21",
8561
+ "last_updated": "2026-09-01",
8127
8562
  "modalities": {
8128
8563
  "input": [
8129
8564
  "text",
@@ -8133,7 +8568,7 @@
8133
8568
  "text"
8134
8569
  ]
8135
8570
  },
8136
- "open_weights": false,
8571
+ "open_weights": true,
8137
8572
  "limit": {
8138
8573
  "context": 1048576,
8139
8574
  "output": 943718
@@ -8180,12 +8615,12 @@
8180
8615
  "open_weights": true,
8181
8616
  "limit": {
8182
8617
  "context": 1048576,
8183
- "output": 393216
8618
+ "output": 384000
8184
8619
  },
8185
8620
  "cost": {
8186
- "input": 0.57948,
8187
- "output": 1.73844,
8188
- "cache_read": 0.018438
8621
+ "input": 1.32,
8622
+ "output": 3.96,
8623
+ "cache_read": 0.044
8189
8624
  }
8190
8625
  },
8191
8626
  "deepseek/deepseek-v4-flash-0731": {
@@ -8228,8 +8663,8 @@
8228
8663
  "output": 943718
8229
8664
  },
8230
8665
  "cost": {
8231
- "input": 0.065,
8232
- "output": 0.18,
8666
+ "input": 0.04,
8667
+ "output": 0.64,
8233
8668
  "cache_read": 0.016
8234
8669
  }
8235
8670
  },
@@ -8275,9 +8710,9 @@
8275
8710
  "output": 384000
8276
8711
  },
8277
8712
  "cost": {
8278
- "input": 0.08554,
8279
- "output": 0.17108,
8280
- "cache_read": 0.017108
8713
+ "input": 0.088606,
8714
+ "output": 0.177212,
8715
+ "cache_read": 0.017721
8281
8716
  }
8282
8717
  },
8283
8718
  "deepseek/deepseek-v4.1-flash": {
@@ -8318,12 +8753,12 @@
8318
8753
  "open_weights": true,
8319
8754
  "limit": {
8320
8755
  "context": 1048576,
8321
- "output": 384000
8756
+ "output": 943718
8322
8757
  },
8323
8758
  "cost": {
8324
- "input": 0.15,
8325
- "output": 0.6,
8326
- "cache_read": 0.003
8759
+ "input": 0.1,
8760
+ "output": 0.5,
8761
+ "cache_read": 0.01
8327
8762
  }
8328
8763
  },
8329
8764
  "deepseek/deepseek-r1": {
@@ -8335,7 +8770,7 @@
8335
8770
  "reasoning": true,
8336
8771
  "reasoning_options": [],
8337
8772
  "tool_call": true,
8338
- "structured_output": true,
8773
+ "structured_output": false,
8339
8774
  "temperature": true,
8340
8775
  "knowledge": "2024-07",
8341
8776
  "release_date": "2025-01-20",
@@ -8382,11 +8817,11 @@
8382
8817
  "open_weights": true,
8383
8818
  "limit": {
8384
8819
  "context": 163840,
8385
- "output": 16000
8820
+ "output": 16384
8386
8821
  },
8387
8822
  "cost": {
8388
- "input": 0.2574,
8389
- "output": 1.0287
8823
+ "input": 0.32,
8824
+ "output": 0.89
8390
8825
  }
8391
8826
  },
8392
8827
  "deepseek/deepseek-r1-0528": {
@@ -8610,9 +9045,9 @@
8610
9045
  "output": 384000
8611
9046
  },
8612
9047
  "cost": {
8613
- "input": 0.87,
8614
- "output": 1.74,
8615
- "cache_read": 0.0725
9048
+ "input": 0.95526,
9049
+ "output": 1.91052,
9050
+ "cache_read": 0.079605
8616
9051
  }
8617
9052
  },
8618
9053
  "deepseek/deepseek-chat-v3-0324": {
@@ -8642,16 +9077,15 @@
8642
9077
  "output": 147456
8643
9078
  },
8644
9079
  "cost": {
8645
- "input": 0.29,
8646
- "output": 1.14,
8647
- "cache_read": 0.11
9080
+ "input": 0.25,
9081
+ "output": 1
8648
9082
  }
8649
9083
  },
8650
- "~openai/gpt-latest": {
8651
- "id": "~openai/gpt-latest",
8652
- "name": "OpenAI GPT Latest",
9084
+ "~openai/gpt-terra-latest": {
9085
+ "id": "~openai/gpt-terra-latest",
9086
+ "name": "GPT Terra Latest",
8653
9087
  "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
8654
- "family": "gpt",
9088
+ "family": "gpt-terra",
8655
9089
  "attachment": true,
8656
9090
  "reasoning": true,
8657
9091
  "reasoning_options": [
@@ -8671,8 +9105,74 @@
8671
9105
  "structured_output": true,
8672
9106
  "temperature": false,
8673
9107
  "knowledge": "2026-02-16",
8674
- "release_date": "2026-04-27",
8675
- "last_updated": "2026-04-27",
9108
+ "release_date": "2026-09-11",
9109
+ "last_updated": "2026-09-11",
9110
+ "modalities": {
9111
+ "input": [
9112
+ "pdf",
9113
+ "image",
9114
+ "text"
9115
+ ],
9116
+ "output": [
9117
+ "text"
9118
+ ]
9119
+ },
9120
+ "open_weights": false,
9121
+ "limit": {
9122
+ "context": 1050000,
9123
+ "output": 128000
9124
+ },
9125
+ "cost": {
9126
+ "input": 2,
9127
+ "output": 12,
9128
+ "cache_read": 0.2,
9129
+ "cache_write": 2.5,
9130
+ "tiers": [
9131
+ {
9132
+ "input": 4,
9133
+ "output": 18,
9134
+ "cache_read": 0.4,
9135
+ "cache_write": 5,
9136
+ "tier": {
9137
+ "type": "context",
9138
+ "size": 272000
9139
+ }
9140
+ }
9141
+ ],
9142
+ "context_over_200k": {
9143
+ "input": 4,
9144
+ "output": 18,
9145
+ "cache_read": 0.4,
9146
+ "cache_write": 5
9147
+ }
9148
+ }
9149
+ },
9150
+ "~openai/gpt-sol-latest": {
9151
+ "id": "~openai/gpt-sol-latest",
9152
+ "name": "GPT Sol Latest",
9153
+ "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
9154
+ "family": "gpt-sol",
9155
+ "attachment": true,
9156
+ "reasoning": true,
9157
+ "reasoning_options": [
9158
+ {
9159
+ "type": "effort",
9160
+ "values": [
9161
+ "none",
9162
+ "low",
9163
+ "medium",
9164
+ "high",
9165
+ "xhigh",
9166
+ "max"
9167
+ ]
9168
+ }
9169
+ ],
9170
+ "tool_call": true,
9171
+ "structured_output": true,
9172
+ "temperature": false,
9173
+ "knowledge": "2026-02-16",
9174
+ "release_date": "2026-09-11",
9175
+ "last_updated": "2026-09-11",
8676
9176
  "modalities": {
8677
9177
  "input": [
8678
9178
  "pdf",
@@ -8713,9 +9213,139 @@
8713
9213
  }
8714
9214
  }
8715
9215
  },
9216
+ "~openai/gpt-luna-latest": {
9217
+ "id": "~openai/gpt-luna-latest",
9218
+ "name": "GPT Luna Latest",
9219
+ "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
9220
+ "family": "gpt-luna",
9221
+ "attachment": true,
9222
+ "reasoning": true,
9223
+ "reasoning_options": [
9224
+ {
9225
+ "type": "effort",
9226
+ "values": [
9227
+ "none",
9228
+ "low",
9229
+ "medium",
9230
+ "high",
9231
+ "xhigh",
9232
+ "max"
9233
+ ]
9234
+ }
9235
+ ],
9236
+ "tool_call": true,
9237
+ "structured_output": true,
9238
+ "temperature": false,
9239
+ "knowledge": "2026-02-16",
9240
+ "release_date": "2026-09-11",
9241
+ "last_updated": "2026-09-11",
9242
+ "modalities": {
9243
+ "input": [
9244
+ "pdf",
9245
+ "image",
9246
+ "text"
9247
+ ],
9248
+ "output": [
9249
+ "text"
9250
+ ]
9251
+ },
9252
+ "open_weights": false,
9253
+ "limit": {
9254
+ "context": 1050000,
9255
+ "output": 128000
9256
+ },
9257
+ "cost": {
9258
+ "input": 0.1,
9259
+ "output": 0.5,
9260
+ "cache_read": 0.01,
9261
+ "cache_write": 0.125,
9262
+ "tiers": [
9263
+ {
9264
+ "input": 0.2,
9265
+ "output": 0.75,
9266
+ "cache_read": 0.02,
9267
+ "cache_write": 0.25,
9268
+ "tier": {
9269
+ "type": "context",
9270
+ "size": 272000
9271
+ }
9272
+ }
9273
+ ],
9274
+ "context_over_200k": {
9275
+ "input": 0.2,
9276
+ "output": 0.75,
9277
+ "cache_read": 0.02,
9278
+ "cache_write": 0.25
9279
+ }
9280
+ }
9281
+ },
9282
+ "~openai/gpt-astra-latest": {
9283
+ "id": "~openai/gpt-astra-latest",
9284
+ "name": "GPT Astra Latest",
9285
+ "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
9286
+ "family": "gpt-astra",
9287
+ "attachment": true,
9288
+ "reasoning": true,
9289
+ "reasoning_options": [
9290
+ {
9291
+ "type": "effort",
9292
+ "values": [
9293
+ "low",
9294
+ "medium",
9295
+ "high",
9296
+ "xhigh",
9297
+ "max"
9298
+ ]
9299
+ }
9300
+ ],
9301
+ "tool_call": true,
9302
+ "structured_output": true,
9303
+ "temperature": false,
9304
+ "release_date": "2026-09-11",
9305
+ "last_updated": "2026-09-11",
9306
+ "modalities": {
9307
+ "input": [
9308
+ "pdf",
9309
+ "image",
9310
+ "text"
9311
+ ],
9312
+ "output": [
9313
+ "text"
9314
+ ]
9315
+ },
9316
+ "open_weights": false,
9317
+ "limit": {
9318
+ "context": 1050000,
9319
+ "output": 128000
9320
+ },
9321
+ "cost": {
9322
+ "input": 10,
9323
+ "output": 50,
9324
+ "cache_read": 1,
9325
+ "cache_write": 12.5,
9326
+ "tiers": [
9327
+ {
9328
+ "input": 20,
9329
+ "output": 75,
9330
+ "cache_read": 2,
9331
+ "cache_write": 25,
9332
+ "tier": {
9333
+ "type": "context",
9334
+ "size": 272000
9335
+ }
9336
+ }
9337
+ ],
9338
+ "context_over_200k": {
9339
+ "input": 20,
9340
+ "output": 75,
9341
+ "cache_read": 2,
9342
+ "cache_write": 25
9343
+ }
9344
+ }
9345
+ },
8716
9346
  "~openai/gpt-mini-latest": {
8717
9347
  "id": "~openai/gpt-mini-latest",
8718
- "name": "OpenAI GPT Mini Latest",
9348
+ "name": "GPT Mini Latest",
8719
9349
  "description": "Compact GPT model for low-latency assistance and high-volume workloads",
8720
9350
  "family": "gpt-mini",
8721
9351
  "attachment": true,
@@ -9103,6 +9733,44 @@
9103
9733
  "output": 0
9104
9734
  }
9105
9735
  },
9736
+ "inclusionai/ling-3.0-flash-vl": {
9737
+ "id": "inclusionai/ling-3.0-flash-vl",
9738
+ "name": "Ling 3.0 Flash VL",
9739
+ "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
9740
+ "family": "ling",
9741
+ "attachment": true,
9742
+ "reasoning": true,
9743
+ "reasoning_options": [
9744
+ {
9745
+ "type": "toggle"
9746
+ }
9747
+ ],
9748
+ "tool_call": true,
9749
+ "structured_output": true,
9750
+ "temperature": true,
9751
+ "release_date": "2026-09-10",
9752
+ "last_updated": "2026-09-10",
9753
+ "modalities": {
9754
+ "input": [
9755
+ "text",
9756
+ "image",
9757
+ "video"
9758
+ ],
9759
+ "output": [
9760
+ "text"
9761
+ ]
9762
+ },
9763
+ "open_weights": true,
9764
+ "limit": {
9765
+ "context": 131072,
9766
+ "output": 32768
9767
+ },
9768
+ "cost": {
9769
+ "input": 0.06,
9770
+ "output": 0.18,
9771
+ "cache_read": 0.012
9772
+ }
9773
+ },
9106
9774
  "anthracite-org/magnum-v4-72b": {
9107
9775
  "id": "anthracite-org/magnum-v4-72b",
9108
9776
  "name": "Magnum v4 72B",
@@ -9742,6 +10410,67 @@
9742
10410
  }
9743
10411
  }
9744
10412
  },
10413
+ "x-ai/grok-4.7": {
10414
+ "id": "x-ai/grok-4.7",
10415
+ "name": "Grok 4.7",
10416
+ "description": "Grok model for agentic tool use, reasoning, coding, and live assistance",
10417
+ "family": "grok",
10418
+ "attachment": true,
10419
+ "reasoning": true,
10420
+ "reasoning_options": [
10421
+ {
10422
+ "type": "effort",
10423
+ "values": [
10424
+ "low",
10425
+ "medium",
10426
+ "high",
10427
+ "xhigh"
10428
+ ]
10429
+ }
10430
+ ],
10431
+ "tool_call": true,
10432
+ "structured_output": true,
10433
+ "temperature": true,
10434
+ "knowledge": "2026-05",
10435
+ "release_date": "2026-09-21",
10436
+ "last_updated": "2026-09-21",
10437
+ "modalities": {
10438
+ "input": [
10439
+ "text",
10440
+ "image",
10441
+ "pdf"
10442
+ ],
10443
+ "output": [
10444
+ "text"
10445
+ ]
10446
+ },
10447
+ "open_weights": false,
10448
+ "limit": {
10449
+ "context": 500000,
10450
+ "output": 450000
10451
+ },
10452
+ "cost": {
10453
+ "input": 1.6,
10454
+ "output": 4.8,
10455
+ "cache_read": 0.4,
10456
+ "tiers": [
10457
+ {
10458
+ "input": 3.2,
10459
+ "output": 9.6,
10460
+ "cache_read": 0.8,
10461
+ "tier": {
10462
+ "type": "context",
10463
+ "size": 200000
10464
+ }
10465
+ }
10466
+ ],
10467
+ "context_over_200k": {
10468
+ "input": 3.2,
10469
+ "output": 9.6,
10470
+ "cache_read": 0.8
10471
+ }
10472
+ }
10473
+ },
9745
10474
  "meta-llama/llama-3.1-8b-instruct": {
9746
10475
  "id": "meta-llama/llama-3.1-8b-instruct",
9747
10476
  "name": "Llama-3.1-8B-Instruct",
@@ -9893,11 +10622,11 @@
9893
10622
  "open_weights": true,
9894
10623
  "limit": {
9895
10624
  "context": 1048576,
9896
- "output": 115200
10625
+ "output": 16384
9897
10626
  },
9898
10627
  "cost": {
9899
- "input": 0.2,
9900
- "output": 0.696
10628
+ "input": 0.1875,
10629
+ "output": 0.6525
9901
10630
  }
9902
10631
  },
9903
10632
  "meta-llama/llama-4-scout": {
@@ -10364,6 +11093,73 @@
10364
11093
  "cache_read": 0.55
10365
11094
  }
10366
11095
  },
11096
+ "openai/gpt-6-sol": {
11097
+ "id": "openai/gpt-6-sol",
11098
+ "name": "GPT-6 Sol",
11099
+ "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
11100
+ "family": "gpt-sol",
11101
+ "attachment": true,
11102
+ "reasoning": true,
11103
+ "reasoning_options": [
11104
+ {
11105
+ "type": "effort",
11106
+ "values": [
11107
+ "none",
11108
+ "low",
11109
+ "medium",
11110
+ "high",
11111
+ "xhigh",
11112
+ "max"
11113
+ ]
11114
+ }
11115
+ ],
11116
+ "tool_call": true,
11117
+ "structured_output": true,
11118
+ "temperature": false,
11119
+ "knowledge": "2026-04-20",
11120
+ "release_date": "2026-09-22",
11121
+ "last_updated": "2026-09-22",
11122
+ "modalities": {
11123
+ "input": [
11124
+ "text",
11125
+ "image",
11126
+ "pdf"
11127
+ ],
11128
+ "output": [
11129
+ "text"
11130
+ ]
11131
+ },
11132
+ "open_weights": false,
11133
+ "limit": {
11134
+ "context": 1050000,
11135
+ "input": 922000,
11136
+ "output": 128000
11137
+ },
11138
+ "cost": {
11139
+ "input": 2,
11140
+ "output": 10,
11141
+ "cache_read": 0.2,
11142
+ "cache_write": 2.5,
11143
+ "tiers": [
11144
+ {
11145
+ "input": 4,
11146
+ "output": 15,
11147
+ "cache_read": 0.4,
11148
+ "cache_write": 5,
11149
+ "tier": {
11150
+ "type": "context",
11151
+ "size": 272000
11152
+ }
11153
+ }
11154
+ ],
11155
+ "context_over_200k": {
11156
+ "input": 4,
11157
+ "output": 15,
11158
+ "cache_read": 0.4,
11159
+ "cache_write": 5
11160
+ }
11161
+ }
11162
+ },
10367
11163
  "openai/gpt-5.1-codex-mini": {
10368
11164
  "id": "openai/gpt-5.1-codex-mini",
10369
11165
  "name": "GPT-5.1 Codex mini",
@@ -10475,6 +11271,71 @@
10475
11271
  }
10476
11272
  }
10477
11273
  },
11274
+ "openai/gpt-6-luna-pro": {
11275
+ "id": "openai/gpt-6-luna-pro",
11276
+ "name": "GPT-6 Luna Pro",
11277
+ "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
11278
+ "family": "gpt",
11279
+ "attachment": true,
11280
+ "reasoning": true,
11281
+ "reasoning_options": [
11282
+ {
11283
+ "type": "effort",
11284
+ "values": [
11285
+ "none",
11286
+ "low",
11287
+ "medium",
11288
+ "high",
11289
+ "xhigh",
11290
+ "max"
11291
+ ]
11292
+ }
11293
+ ],
11294
+ "tool_call": true,
11295
+ "structured_output": true,
11296
+ "temperature": false,
11297
+ "release_date": "2026-09-22",
11298
+ "last_updated": "2026-09-22",
11299
+ "modalities": {
11300
+ "input": [
11301
+ "pdf",
11302
+ "image",
11303
+ "text"
11304
+ ],
11305
+ "output": [
11306
+ "text"
11307
+ ]
11308
+ },
11309
+ "open_weights": false,
11310
+ "limit": {
11311
+ "context": 1050000,
11312
+ "output": 128000
11313
+ },
11314
+ "cost": {
11315
+ "input": 0.1,
11316
+ "output": 0.5,
11317
+ "cache_read": 0.01,
11318
+ "cache_write": 0.125,
11319
+ "tiers": [
11320
+ {
11321
+ "input": 0.2,
11322
+ "output": 0.75,
11323
+ "cache_read": 0.02,
11324
+ "cache_write": 0.25,
11325
+ "tier": {
11326
+ "type": "context",
11327
+ "size": 272000
11328
+ }
11329
+ }
11330
+ ],
11331
+ "context_over_200k": {
11332
+ "input": 0.2,
11333
+ "output": 0.75,
11334
+ "cache_read": 0.02,
11335
+ "cache_write": 0.25
11336
+ }
11337
+ }
11338
+ },
10478
11339
  "openai/gpt-audio-mini": {
10479
11340
  "id": "openai/gpt-audio-mini",
10480
11341
  "name": "GPT Audio Mini",
@@ -10618,21 +11479,35 @@
10618
11479
  }
10619
11480
  }
10620
11481
  },
10621
- "openai/gpt-4-turbo-preview": {
10622
- "id": "openai/gpt-4-turbo-preview",
10623
- "name": "GPT-4 Turbo Preview",
10624
- "description": "Compact GPT model for low-latency assistance and high-volume workloads",
11482
+ "openai/gpt-6-sol-pro": {
11483
+ "id": "openai/gpt-6-sol-pro",
11484
+ "name": "GPT-6 Sol Pro",
11485
+ "description": "Frontier GPT model for professional reasoning, coding, and multimodal work",
10625
11486
  "family": "gpt",
10626
- "attachment": false,
10627
- "reasoning": false,
11487
+ "attachment": true,
11488
+ "reasoning": true,
11489
+ "reasoning_options": [
11490
+ {
11491
+ "type": "effort",
11492
+ "values": [
11493
+ "none",
11494
+ "low",
11495
+ "medium",
11496
+ "high",
11497
+ "xhigh",
11498
+ "max"
11499
+ ]
11500
+ }
11501
+ ],
10628
11502
  "tool_call": true,
10629
11503
  "structured_output": true,
10630
- "temperature": true,
10631
- "knowledge": "2023-12-31",
10632
- "release_date": "2024-01-25",
10633
- "last_updated": "2024-01-25",
11504
+ "temperature": false,
11505
+ "release_date": "2026-09-22",
11506
+ "last_updated": "2026-09-22",
10634
11507
  "modalities": {
10635
11508
  "input": [
11509
+ "pdf",
11510
+ "image",
10636
11511
  "text"
10637
11512
  ],
10638
11513
  "output": [
@@ -10641,12 +11516,32 @@
10641
11516
  },
10642
11517
  "open_weights": false,
10643
11518
  "limit": {
10644
- "context": 128000,
10645
- "output": 4096
11519
+ "context": 1050000,
11520
+ "output": 128000
10646
11521
  },
10647
11522
  "cost": {
10648
- "input": 10,
10649
- "output": 30
11523
+ "input": 2,
11524
+ "output": 10,
11525
+ "cache_read": 0.2,
11526
+ "cache_write": 2.5,
11527
+ "tiers": [
11528
+ {
11529
+ "input": 4,
11530
+ "output": 15,
11531
+ "cache_read": 0.4,
11532
+ "cache_write": 5,
11533
+ "tier": {
11534
+ "type": "context",
11535
+ "size": 272000
11536
+ }
11537
+ }
11538
+ ],
11539
+ "context_over_200k": {
11540
+ "input": 4,
11541
+ "output": 15,
11542
+ "cache_read": 0.4,
11543
+ "cache_write": 5
11544
+ }
10650
11545
  }
10651
11546
  },
10652
11547
  "openai/gpt-4o-2024-08-06": {
@@ -11069,12 +11964,11 @@
11069
11964
  "open_weights": true,
11070
11965
  "limit": {
11071
11966
  "context": 131072,
11072
- "output": 117964
11967
+ "output": 32768
11073
11968
  },
11074
11969
  "cost": {
11075
- "input": 0.03,
11076
- "output": 0.13,
11077
- "cache_read": 0.03
11970
+ "input": 0.018,
11971
+ "output": 0.09
11078
11972
  }
11079
11973
  },
11080
11974
  "openai/gpt-4-turbo": {
@@ -11214,7 +12108,7 @@
11214
12108
  },
11215
12109
  "openai/gpt-oss-safeguard-20b": {
11216
12110
  "id": "openai/gpt-oss-safeguard-20b",
11217
- "name": "gpt-oss-safeguard-20b",
12111
+ "name": "GPT OSS Safeguard 20B",
11218
12112
  "description": "Safety model for policy screening, moderation, and risk-aware routing workflows",
11219
12113
  "family": "gpt-oss",
11220
12114
  "attachment": false,
@@ -11498,13 +12392,60 @@
11498
12392
  "description": "Omni-era GPT for multimodal chat, practical coding, and general assistants",
11499
12393
  "family": "gpt",
11500
12394
  "attachment": true,
11501
- "reasoning": false,
12395
+ "reasoning": false,
12396
+ "tool_call": true,
12397
+ "structured_output": true,
12398
+ "temperature": true,
12399
+ "knowledge": "2023-09",
12400
+ "release_date": "2024-05-13",
12401
+ "last_updated": "2024-08-06",
12402
+ "modalities": {
12403
+ "input": [
12404
+ "text",
12405
+ "image",
12406
+ "pdf"
12407
+ ],
12408
+ "output": [
12409
+ "text"
12410
+ ]
12411
+ },
12412
+ "open_weights": false,
12413
+ "limit": {
12414
+ "context": 128000,
12415
+ "output": 16384
12416
+ },
12417
+ "cost": {
12418
+ "input": 2.5,
12419
+ "output": 10,
12420
+ "cache_read": 1.25
12421
+ }
12422
+ },
12423
+ "openai/gpt-6-luna": {
12424
+ "id": "openai/gpt-6-luna",
12425
+ "name": "GPT-6 Luna",
12426
+ "description": "GPT model for general reasoning, writing, coding, and tool-assisted tasks",
12427
+ "family": "gpt-luna",
12428
+ "attachment": true,
12429
+ "reasoning": true,
12430
+ "reasoning_options": [
12431
+ {
12432
+ "type": "effort",
12433
+ "values": [
12434
+ "none",
12435
+ "low",
12436
+ "medium",
12437
+ "high",
12438
+ "xhigh",
12439
+ "max"
12440
+ ]
12441
+ }
12442
+ ],
11502
12443
  "tool_call": true,
11503
12444
  "structured_output": true,
11504
- "temperature": true,
11505
- "knowledge": "2023-09",
11506
- "release_date": "2024-05-13",
11507
- "last_updated": "2024-08-06",
12445
+ "temperature": false,
12446
+ "knowledge": "2026-05-18",
12447
+ "release_date": "2026-09-22",
12448
+ "last_updated": "2026-09-22",
11508
12449
  "modalities": {
11509
12450
  "input": [
11510
12451
  "text",
@@ -11517,13 +12458,33 @@
11517
12458
  },
11518
12459
  "open_weights": false,
11519
12460
  "limit": {
11520
- "context": 128000,
11521
- "output": 16384
12461
+ "context": 1050000,
12462
+ "input": 922000,
12463
+ "output": 128000
11522
12464
  },
11523
12465
  "cost": {
11524
- "input": 2.5,
11525
- "output": 10,
11526
- "cache_read": 1.25
12466
+ "input": 0.1,
12467
+ "output": 0.5,
12468
+ "cache_read": 0.01,
12469
+ "cache_write": 0.125,
12470
+ "tiers": [
12471
+ {
12472
+ "input": 0.2,
12473
+ "output": 0.75,
12474
+ "cache_read": 0.02,
12475
+ "cache_write": 0.25,
12476
+ "tier": {
12477
+ "type": "context",
12478
+ "size": 272000
12479
+ }
12480
+ }
12481
+ ],
12482
+ "context_over_200k": {
12483
+ "input": 0.2,
12484
+ "output": 0.75,
12485
+ "cache_read": 0.02,
12486
+ "cache_write": 0.25
12487
+ }
11527
12488
  }
11528
12489
  },
11529
12490
  "openai/gpt-5.6-luna": {
@@ -12174,11 +13135,12 @@
12174
13135
  "open_weights": true,
12175
13136
  "limit": {
12176
13137
  "context": 131072,
12177
- "output": 117964
13138
+ "output": 65536
12178
13139
  },
12179
13140
  "cost": {
12180
- "input": 0.037,
12181
- "output": 0.17
13141
+ "input": 0.15,
13142
+ "output": 0.6,
13143
+ "cache_read": 0.075
12182
13144
  }
12183
13145
  },
12184
13146
  "openai/gpt-5.4-pro": {
@@ -12788,12 +13750,12 @@
12788
13750
  "open_weights": false,
12789
13751
  "limit": {
12790
13752
  "context": 1310720,
12791
- "output": 128000
13753
+ "output": 131072
12792
13754
  },
12793
13755
  "cost": {
12794
- "input": 0.7,
12795
- "output": 2.2,
12796
- "cache_read": 0.13
13756
+ "input": 0.5625,
13757
+ "output": 2.5,
13758
+ "cache_read": 0.125
12797
13759
  }
12798
13760
  },
12799
13761
  "moonshotai/kimi-k2-0905": {
@@ -12820,7 +13782,7 @@
12820
13782
  "open_weights": true,
12821
13783
  "limit": {
12822
13784
  "context": 262144,
12823
- "output": 100352
13785
+ "output": 98304
12824
13786
  },
12825
13787
  "cost": {
12826
13788
  "input": 0.6,
@@ -12897,9 +13859,9 @@
12897
13859
  "output": 235929
12898
13860
  },
12899
13861
  "cost": {
12900
- "input": 0.71,
12901
- "output": 3.5,
12902
- "cache_read": 0.15
13862
+ "input": 0.7062,
13863
+ "output": 3.3,
13864
+ "cache_read": 0.18
12903
13865
  }
12904
13866
  },
12905
13867
  "moonshotai/kimi-k2-thinking": {
@@ -12930,7 +13892,7 @@
12930
13892
  "open_weights": true,
12931
13893
  "limit": {
12932
13894
  "context": 262144,
12933
- "output": 100352
13895
+ "output": 98304
12934
13896
  },
12935
13897
  "cost": {
12936
13898
  "input": 0.6,
@@ -13008,7 +13970,7 @@
13008
13970
  "open_weights": true,
13009
13971
  "limit": {
13010
13972
  "context": 131072,
13011
- "output": 100352
13973
+ "output": 98304
13012
13974
  },
13013
13975
  "cost": {
13014
13976
  "input": 0.57,
@@ -13056,6 +14018,103 @@
13056
14018
  "cache_read": 0.07
13057
14019
  }
13058
14020
  },
14021
+ "inference-net/schematron-v2-small": {
14022
+ "id": "inference-net/schematron-v2-small",
14023
+ "name": "Schematron V2 Small",
14024
+ "description": "Efficient model for low-latency assistance, extraction, and routine automation",
14025
+ "attachment": false,
14026
+ "reasoning": false,
14027
+ "tool_call": false,
14028
+ "structured_output": true,
14029
+ "temperature": true,
14030
+ "release_date": "2026-09-12",
14031
+ "last_updated": "2026-09-12",
14032
+ "modalities": {
14033
+ "input": [
14034
+ "text"
14035
+ ],
14036
+ "output": [
14037
+ "text"
14038
+ ]
14039
+ },
14040
+ "open_weights": true,
14041
+ "limit": {
14042
+ "context": 128000,
14043
+ "output": 4096
14044
+ },
14045
+ "cost": {
14046
+ "input": 0.05,
14047
+ "output": 0.23,
14048
+ "cache_read": 0.05
14049
+ }
14050
+ },
14051
+ "inference-net/schematron-v2-turbo": {
14052
+ "id": "inference-net/schematron-v2-turbo",
14053
+ "name": "Schematron V2 Turbo",
14054
+ "description": "Efficient model for low-latency assistance, extraction, and routine automation",
14055
+ "attachment": false,
14056
+ "reasoning": false,
14057
+ "tool_call": false,
14058
+ "structured_output": true,
14059
+ "temperature": true,
14060
+ "release_date": "2026-09-12",
14061
+ "last_updated": "2026-09-12",
14062
+ "modalities": {
14063
+ "input": [
14064
+ "text"
14065
+ ],
14066
+ "output": [
14067
+ "text"
14068
+ ]
14069
+ },
14070
+ "open_weights": true,
14071
+ "limit": {
14072
+ "context": 128000,
14073
+ "output": 8192
14074
+ },
14075
+ "cost": {
14076
+ "input": 0.03,
14077
+ "output": 0.15,
14078
+ "cache_read": 0.03
14079
+ }
14080
+ },
14081
+ "cohere/command-a-plus": {
14082
+ "id": "cohere/command-a-plus",
14083
+ "name": "Command A+",
14084
+ "description": "Cohere command model for multilingual enterprise agents, tools, and chat",
14085
+ "family": "command-a",
14086
+ "attachment": true,
14087
+ "reasoning": true,
14088
+ "reasoning_options": [
14089
+ {
14090
+ "type": "toggle"
14091
+ }
14092
+ ],
14093
+ "tool_call": true,
14094
+ "structured_output": true,
14095
+ "temperature": true,
14096
+ "release_date": "2026-09-22",
14097
+ "last_updated": "2026-09-22",
14098
+ "modalities": {
14099
+ "input": [
14100
+ "text",
14101
+ "image"
14102
+ ],
14103
+ "output": [
14104
+ "text"
14105
+ ]
14106
+ },
14107
+ "open_weights": false,
14108
+ "limit": {
14109
+ "context": 192000,
14110
+ "output": 64000
14111
+ },
14112
+ "cost": {
14113
+ "input": 0.3,
14114
+ "output": 1.5,
14115
+ "cache_read": 0.15
14116
+ }
14117
+ },
13059
14118
  "cohere/north-mini-code:free": {
13060
14119
  "id": "cohere/north-mini-code:free",
13061
14120
  "name": "North Mini Code (free)",
@@ -13224,7 +14283,14 @@
13224
14283
  "reasoning": true,
13225
14284
  "reasoning_options": [
13226
14285
  {
13227
- "type": "toggle"
14286
+ "type": "effort",
14287
+ "values": [
14288
+ "none",
14289
+ "minimal",
14290
+ "low",
14291
+ "medium",
14292
+ "high"
14293
+ ]
13228
14294
  }
13229
14295
  ],
13230
14296
  "tool_call": true,
@@ -13260,7 +14326,16 @@
13260
14326
  "reasoning": true,
13261
14327
  "reasoning_options": [
13262
14328
  {
13263
- "type": "toggle"
14329
+ "type": "effort",
14330
+ "values": [
14331
+ "none",
14332
+ "minimal",
14333
+ "low",
14334
+ "medium",
14335
+ "high",
14336
+ "xhigh",
14337
+ "max"
14338
+ ]
13264
14339
  }
13265
14340
  ],
13266
14341
  "tool_call": true,
@@ -13282,9 +14357,9 @@
13282
14357
  "output": 131072
13283
14358
  },
13284
14359
  "cost": {
13285
- "input": 0.03,
13286
- "output": 0.12,
13287
- "cache_read": 0.006
14360
+ "input": 0.09,
14361
+ "output": 0.36,
14362
+ "cache_read": 0.018
13288
14363
  }
13289
14364
  },
13290
14365
  "arcee-ai/trinity-large-thinking": {
@@ -13356,9 +14431,9 @@
13356
14431
  "output": 128000
13357
14432
  },
13358
14433
  "cost": {
13359
- "input": 0.0825,
13360
- "output": 0.33,
13361
- "cache_read": 0.020625
14434
+ "input": 0.132,
14435
+ "output": 0.528,
14436
+ "cache_read": 0.033
13362
14437
  }
13363
14438
  },
13364
14439
  "tencent/hy4-preview": {
@@ -13791,19 +14866,62 @@
13791
14866
  "open_weights": true,
13792
14867
  "limit": {
13793
14868
  "context": 1048576,
13794
- "output": 128000
14869
+ "output": 131072
14870
+ },
14871
+ "cost": {
14872
+ "input": 0.6496,
14873
+ "output": 2.0416,
14874
+ "cache_read": 0.12064
14875
+ }
14876
+ },
14877
+ "z-ai/glm-5.3-flashx": {
14878
+ "id": "z-ai/glm-5.3-flashx",
14879
+ "name": "GLM 5.3 FlashX",
14880
+ "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
14881
+ "family": "glm",
14882
+ "attachment": true,
14883
+ "reasoning": true,
14884
+ "reasoning_options": [
14885
+ {
14886
+ "type": "effort",
14887
+ "values": [
14888
+ "low",
14889
+ "high",
14890
+ "max"
14891
+ ]
14892
+ }
14893
+ ],
14894
+ "tool_call": true,
14895
+ "structured_output": false,
14896
+ "temperature": true,
14897
+ "release_date": "2026-09-18",
14898
+ "last_updated": "2026-09-18",
14899
+ "modalities": {
14900
+ "input": [
14901
+ "text",
14902
+ "image",
14903
+ "video"
14904
+ ],
14905
+ "output": [
14906
+ "text"
14907
+ ]
14908
+ },
14909
+ "open_weights": false,
14910
+ "limit": {
14911
+ "context": 1048576,
14912
+ "output": 131072
13795
14913
  },
13796
14914
  "cost": {
13797
- "input": 0.28,
13798
- "output": 0.88,
13799
- "cache_read": 0.052
14915
+ "input": 0.37,
14916
+ "output": 1.25,
14917
+ "cache_read": 0.075
13800
14918
  }
13801
14919
  },
13802
14920
  "z-ai/glm-5.3-flash": {
13803
14921
  "id": "z-ai/glm-5.3-flash",
13804
14922
  "name": "GLM-5.3-Flash",
13805
14923
  "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
13806
- "family": "glm",
14924
+ "family": "glm-flash",
13807
14925
  "attachment": true,
13808
14926
  "reasoning": true,
13809
14927
  "reasoning_options": [
@@ -13834,12 +14952,12 @@
13834
14952
  "open_weights": true,
13835
14953
  "limit": {
13836
14954
  "context": 1310720,
13837
- "output": 131072
14955
+ "output": 943718
13838
14956
  },
13839
14957
  "cost": {
13840
14958
  "input": 0.15,
13841
14959
  "output": 0.5,
13842
- "cache_read": 0.03
14960
+ "cache_read": 0.05
13843
14961
  }
13844
14962
  },
13845
14963
  "z-ai/glm-4.5": {
@@ -13917,6 +15035,48 @@
13917
15035
  "cache_read": 0.11
13918
15036
  }
13919
15037
  },
15038
+ "z-ai/glm-5.2:free": {
15039
+ "id": "z-ai/glm-5.2:free",
15040
+ "name": "GLM 5.2 (free)",
15041
+ "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
15042
+ "family": "glm",
15043
+ "attachment": false,
15044
+ "reasoning": true,
15045
+ "reasoning_options": [
15046
+ {
15047
+ "type": "toggle"
15048
+ },
15049
+ {
15050
+ "type": "effort",
15051
+ "values": [
15052
+ "high",
15053
+ "xhigh"
15054
+ ]
15055
+ }
15056
+ ],
15057
+ "tool_call": false,
15058
+ "structured_output": false,
15059
+ "temperature": true,
15060
+ "release_date": "2026-06-13",
15061
+ "last_updated": "2026-06-13",
15062
+ "modalities": {
15063
+ "input": [
15064
+ "text"
15065
+ ],
15066
+ "output": [
15067
+ "text"
15068
+ ]
15069
+ },
15070
+ "open_weights": true,
15071
+ "limit": {
15072
+ "context": 32768,
15073
+ "output": 29491
15074
+ },
15075
+ "cost": {
15076
+ "input": 0,
15077
+ "output": 0
15078
+ }
15079
+ },
13920
15080
  "z-ai/glm-5": {
13921
15081
  "id": "z-ai/glm-5",
13922
15082
  "name": "GLM-5",
@@ -14067,12 +15227,12 @@
14067
15227
  "open_weights": true,
14068
15228
  "limit": {
14069
15229
  "context": 1310720,
14070
- "output": 943718
15230
+ "output": 131072
14071
15231
  },
14072
15232
  "cost": {
14073
- "input": 1.4,
14074
- "output": 4.4,
14075
- "cache_read": 0.26
15233
+ "input": 0.84,
15234
+ "output": 2.64,
15235
+ "cache_read": 0.156
14076
15236
  }
14077
15237
  },
14078
15238
  "z-ai/glm-5v-turbo": {