@molecule/api-resource-ai-models 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +115 -0
- package/dist/browser-guard.d.ts +2 -0
- package/dist/browser-guard.d.ts.map +1 -0
- package/dist/browser-guard.js +18 -0
- package/dist/handlers/index.d.ts +2 -0
- package/dist/handlers/index.d.ts.map +1 -0
- package/dist/handlers/index.js +1 -0
- package/dist/handlers/list.d.ts +34 -0
- package/dist/handlers/list.d.ts.map +1 -0
- package/dist/handlers/list.js +48 -0
- package/dist/index.d.ts +48 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +47 -0
- package/dist/lookup.d.ts +93 -0
- package/dist/lookup.d.ts.map +1 -0
- package/dist/lookup.js +121 -0
- package/dist/models.d.ts +78 -0
- package/dist/models.d.ts.map +1 -0
- package/dist/models.js +1233 -0
- package/dist/requestHandlerMap.d.ts +13 -0
- package/dist/requestHandlerMap.d.ts.map +1 -0
- package/dist/requestHandlerMap.js +12 -0
- package/dist/routes.d.ts +13 -0
- package/dist/routes.d.ts.map +1 -0
- package/dist/routes.js +14 -0
- package/dist/types.d.ts +315 -0
- package/dist/types.d.ts.map +1 -0
- package/dist/types.js +12 -0
- package/package.json +59 -0
package/dist/models.js
ADDED
|
@@ -0,0 +1,1233 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Available AI models.
|
|
3
|
+
*
|
|
4
|
+
* This is the single source of truth for which models Synthase can use.
|
|
5
|
+
* Server-side code (chat handler, compaction) consumes the full definitions
|
|
6
|
+
* directly; clients receive the `PublicModel` projection from `GET /ai/models`.
|
|
7
|
+
*
|
|
8
|
+
* @module
|
|
9
|
+
*/
|
|
10
|
+
/**
|
|
11
|
+
* All available AI models, grouped by provider, ordered from most to least capable.
|
|
12
|
+
*
|
|
13
|
+
* To add or remove a model, edit this array. Both the server-side validation
|
|
14
|
+
* and the public discovery endpoint will update automatically.
|
|
15
|
+
*
|
|
16
|
+
* Effort is each model's OWN native value — there is no abstract scale (see
|
|
17
|
+
* {@link ModelDefinition.supportedEffortLevels}):
|
|
18
|
+
* - A model driven by a provider-native effort/level param lists its provider
|
|
19
|
+
* values verbatim in `supportedEffortLevels` (ascending), with
|
|
20
|
+
* `defaultEffortLevel` = the provider's default/recommended level for agentic
|
|
21
|
+
* coding. NO `effortBudgetTokens`.
|
|
22
|
+
* - A model with a controllable token budget but no native level names (e.g.
|
|
23
|
+
* Claude Haiku 4.5's `budget_tokens`, Qwen's `thinking_budget`) lists
|
|
24
|
+
* scaled-budget labels (`['4K', '8K', '16K', '32K']`) with `effortBudgetTokens`
|
|
25
|
+
* mapping each label to the token budget it sends.
|
|
26
|
+
* - A model whose reasoning is fixed (always-on or on/off only, no depth
|
|
27
|
+
* control) carries `thinkingConfigurable: false` and OMITS both fields —
|
|
28
|
+
* there is nothing to tune.
|
|
29
|
+
*
|
|
30
|
+
* Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
31
|
+
* 2026-07-30 GPT-5.6 repricing — cross-check prices against models.dev with
|
|
32
|
+
* `npm run check:model-freshness` from the workspace root):
|
|
33
|
+
* - Anthropic: https://platform.claude.com/docs/en/about-claude/models/overview
|
|
34
|
+
* + /docs/en/build-with-claude/effort (fable-5 / opus-5 / sonnet-5 current;
|
|
35
|
+
* opus-4-8 superseded by opus-5 at identical pricing but still served — it is
|
|
36
|
+
* the recommended refusal-fallback model; effort ladder on all three current
|
|
37
|
+
* models is low|medium|high|xhigh|max; budget_tokens 400s on 4.7+)
|
|
38
|
+
* - OpenAI: https://developers.openai.com/api/docs/pricing (GPT-5.6 family GA
|
|
39
|
+
* 2026-07-09; REPRICED 2026-07-30: -luna cut 80% to $0.20/$1.20, -terra cut
|
|
40
|
+
* 20% to $2/$12, -sol unchanged $5/$30; cache read 0.1× input; gpt-5.5/
|
|
41
|
+
* gpt-5.4 still listed as current; long-context 2× variants exist upstream —
|
|
42
|
+
* not modeled, same as the Gemini/Grok tiers; Sol "Fast mode" 2.5× speed at
|
|
43
|
+
* 2× price announced 2026-07-30 — not yet modeled)
|
|
44
|
+
* - Google: https://ai.google.dev/gemini-api/docs/pricing (gemini-3.6-flash GA
|
|
45
|
+
* 2026-07-21 $1.50/$7.50 supersedes 3.5-flash as the agentic flagship;
|
|
46
|
+
* gemini-3.1-pro-preview still the pro tier — "3.5 Pro" has NOT shipped as
|
|
47
|
+
* of 2026-07-28 despite the coming-soon badge; do not add until it has an id)
|
|
48
|
+
* - xAI: https://docs.x.ai/developers/models + /developers/grok-4-5
|
|
49
|
+
* (grok-4.5 flagship 2026-07-08: $2/$6, 500K ctx, ≥200K prompts bill 2× —
|
|
50
|
+
* not modeled; reasoning_effort low|medium|high default high, image input;
|
|
51
|
+
* grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
|
|
52
|
+
* grok-code-fast-1 no longer listed — retires 2026-08-15)
|
|
53
|
+
* - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (unchanged V4
|
|
54
|
+
* Pro/Flash pricing; legacy deepseek-chat/-reasoner ids fully retired
|
|
55
|
+
* 2026-07-24 — never in this catalog; the announced peak-hour 2× surcharge is
|
|
56
|
+
* still NOT active as of 2026-07-28, see the entries)
|
|
57
|
+
* - Moonshot: https://platform.kimi.ai/docs/models (kimi-k3 flagship 2026-07-16
|
|
58
|
+
* — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
|
|
59
|
+
* reasoning_content that must be replayed through tool loops, the same
|
|
60
|
+
* constraint that keeps kimi-k2.7-code out; add BOTH once the moonshot bond
|
|
61
|
+
* supports preserved thinking + reasoning_effort low|high|max. kimi-k2.6
|
|
62
|
+
* remains the newest model the bond can run correctly.)
|
|
63
|
+
* - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
|
|
64
|
+
* minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
|
|
65
|
+
* - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
|
|
66
|
+
* (unchanged; qwen3.8-max-preview, 2026-07-19, is Token-Plan-subscriber-only
|
|
67
|
+
* — not on the pay-as-you-go API, so it cannot be added yet; qwen3.7-max
|
|
68
|
+
* currently runs a 50%-off promo — still billed here at list, $2.50/$7.50)
|
|
69
|
+
* - Zhipu: https://docs.z.ai/guides/overview/pricing (unchanged; glm-5.2 is
|
|
70
|
+
* the newest — "GLM-5.3/5.5" rumors have no released ids as of 2026-07-28)
|
|
71
|
+
*
|
|
72
|
+
* Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
|
|
73
|
+
* where the provider doesn't publish one; the provider sources above verify
|
|
74
|
+
* id / pricing / context window.
|
|
75
|
+
*/
|
|
76
|
+
export const MODELS = [
|
|
77
|
+
// ---------------------------------------------------------------------------
|
|
78
|
+
// Anthropic
|
|
79
|
+
// Verified: https://platform.claude.com/docs/en/about-claude/models/overview
|
|
80
|
+
// https://platform.claude.com/docs/en/build-with-claude/effort
|
|
81
|
+
// Effort is output_config.effort with adaptive thinking on 4.6+ models —
|
|
82
|
+
// budget_tokens is REJECTED (400) on fable-5 / opus-4-8 / sonnet-5 and
|
|
83
|
+
// deprecated on the 4.6 family. Only Haiku 4.5 still uses budget_tokens.
|
|
84
|
+
// ---------------------------------------------------------------------------
|
|
85
|
+
{
|
|
86
|
+
id: 'claude-fable-5',
|
|
87
|
+
provider: 'anthropic',
|
|
88
|
+
label: 'Claude Fable 5',
|
|
89
|
+
description: 'Most capable Anthropic — frontier reasoning & long-horizon agents',
|
|
90
|
+
contextWindow: 1_000_000,
|
|
91
|
+
maxOutputTokens: 128_000,
|
|
92
|
+
supportsThinking: true,
|
|
93
|
+
thinkingBudgetTokens: 16_000,
|
|
94
|
+
thinkingConfigurable: true,
|
|
95
|
+
supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'],
|
|
96
|
+
defaultEffortLevel: 'high',
|
|
97
|
+
// Thinking is ALWAYS ON (adaptive); effort is the only depth control.
|
|
98
|
+
// Full five-level ladder (medium was missing here — Fable supports low
|
|
99
|
+
// through max). Default/recommended is high; xhigh for the most
|
|
100
|
+
// capability-sensitive work; low still performs well on routine tasks.
|
|
101
|
+
supportsVision: true,
|
|
102
|
+
supportsPromptCaching: true,
|
|
103
|
+
supportsTools: true,
|
|
104
|
+
webSearchToolType: 'web_search_20260209',
|
|
105
|
+
codeExecutionToolType: 'code_execution_20250825',
|
|
106
|
+
webFetchToolType: 'web_fetch_20260209',
|
|
107
|
+
inputPricePerMTok: 10,
|
|
108
|
+
outputPricePerMTok: 50,
|
|
109
|
+
// Anthropic 5-minute prompt cache: read 0.1× input, write 1.25× input.
|
|
110
|
+
cacheReadPricePerMTok: 1,
|
|
111
|
+
cacheWritePricePerMTok: 12.5,
|
|
112
|
+
knowledgeCutoff: '2026-01-01',
|
|
113
|
+
},
|
|
114
|
+
{
|
|
115
|
+
id: 'claude-opus-5',
|
|
116
|
+
provider: 'anthropic',
|
|
117
|
+
label: 'Claude Opus 5',
|
|
118
|
+
description: 'Anthropic Opus flagship — step-change agentic coding at 4.8 pricing',
|
|
119
|
+
// Fast mode (research preview, Claude API only): same model at up to 2.5×
|
|
120
|
+
// output speed, $10/$50 per MTok (2× standard; platform.claude.com fast-mode
|
|
121
|
+
// docs, verified 2026-07-31). Cache rates follow Anthropic's standard
|
|
122
|
+
// ratios (read 0.1× input, write 1.25× input) applied to the fast input
|
|
123
|
+
// rate. Separate upstream rate limits from standard Opus.
|
|
124
|
+
fastPricing: {
|
|
125
|
+
inputPricePerMTok: 10,
|
|
126
|
+
outputPricePerMTok: 50,
|
|
127
|
+
cacheReadPricePerMTok: 1,
|
|
128
|
+
cacheWritePricePerMTok: 12.5,
|
|
129
|
+
},
|
|
130
|
+
contextWindow: 1_000_000,
|
|
131
|
+
maxOutputTokens: 128_000,
|
|
132
|
+
supportsThinking: true,
|
|
133
|
+
thinkingBudgetTokens: 16_000,
|
|
134
|
+
thinkingConfigurable: true,
|
|
135
|
+
supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'],
|
|
136
|
+
defaultEffortLevel: 'high',
|
|
137
|
+
// Thinking on by default (adaptive); disabling is only valid at effort
|
|
138
|
+
// high or below (xhigh/max + disabled → 400). Full five-level ladder;
|
|
139
|
+
// Anthropic's rec: xhigh for coding/agentic, but low/medium punch far
|
|
140
|
+
// above their weight on this model. 512-token prompt-cache minimum.
|
|
141
|
+
supportsVision: true,
|
|
142
|
+
supportsPromptCaching: true,
|
|
143
|
+
supportsTools: true,
|
|
144
|
+
webSearchToolType: 'web_search_20260209',
|
|
145
|
+
codeExecutionToolType: 'code_execution_20250825',
|
|
146
|
+
webFetchToolType: 'web_fetch_20260209',
|
|
147
|
+
// Drop-in successor to Opus 4.8 at identical pricing.
|
|
148
|
+
inputPricePerMTok: 5,
|
|
149
|
+
outputPricePerMTok: 25,
|
|
150
|
+
// Anthropic 5-minute prompt cache: read 0.1× input, write 1.25× input.
|
|
151
|
+
cacheReadPricePerMTok: 0.5,
|
|
152
|
+
cacheWritePricePerMTok: 6.25,
|
|
153
|
+
// Not published at verification time — best-effort estimate (≥ Opus 4.8's).
|
|
154
|
+
knowledgeCutoff: '2026-01-01',
|
|
155
|
+
},
|
|
156
|
+
{
|
|
157
|
+
id: 'claude-opus-4-8',
|
|
158
|
+
provider: 'anthropic',
|
|
159
|
+
label: 'Claude Opus 4.8',
|
|
160
|
+
description: 'Previous Opus — deep reasoning; the opus-5 refusal fallback',
|
|
161
|
+
contextWindow: 1_000_000,
|
|
162
|
+
maxOutputTokens: 128_000,
|
|
163
|
+
supportsThinking: true,
|
|
164
|
+
thinkingBudgetTokens: 16_000,
|
|
165
|
+
thinkingConfigurable: true,
|
|
166
|
+
supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'],
|
|
167
|
+
defaultEffortLevel: 'high',
|
|
168
|
+
// Full five-level ladder (medium was missing here). Anthropic's rec for
|
|
169
|
+
// coding/agentic on Opus 4.8 is xhigh; the API default is high.
|
|
170
|
+
supportsVision: true,
|
|
171
|
+
supportsPromptCaching: true,
|
|
172
|
+
supportsTools: true,
|
|
173
|
+
webSearchToolType: 'web_search_20260209',
|
|
174
|
+
codeExecutionToolType: 'code_execution_20250825',
|
|
175
|
+
webFetchToolType: 'web_fetch_20260209',
|
|
176
|
+
inputPricePerMTok: 5,
|
|
177
|
+
outputPricePerMTok: 25,
|
|
178
|
+
// Anthropic 5-minute prompt cache: read 0.1× input, write 1.25× input.
|
|
179
|
+
cacheReadPricePerMTok: 0.5,
|
|
180
|
+
cacheWritePricePerMTok: 6.25,
|
|
181
|
+
knowledgeCutoff: '2026-01-01',
|
|
182
|
+
// Superseded by claude-opus-5 (same price); still served upstream and the
|
|
183
|
+
// recommended refusal-fallback target. Selectable under "Older models".
|
|
184
|
+
deprecatedAt: '2026-07-28',
|
|
185
|
+
},
|
|
186
|
+
{
|
|
187
|
+
id: 'claude-sonnet-5',
|
|
188
|
+
provider: 'anthropic',
|
|
189
|
+
label: 'Claude Sonnet 5',
|
|
190
|
+
description: 'Fast & capable — near-Opus coding at Sonnet cost',
|
|
191
|
+
contextWindow: 1_000_000,
|
|
192
|
+
maxOutputTokens: 128_000,
|
|
193
|
+
supportsThinking: true,
|
|
194
|
+
thinkingBudgetTokens: 16_000,
|
|
195
|
+
thinkingConfigurable: true,
|
|
196
|
+
supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'],
|
|
197
|
+
defaultEffortLevel: 'high',
|
|
198
|
+
// Adaptive thinking on by default; full five-level ladder (medium was
|
|
199
|
+
// missing here). Default/rec is high, xhigh for the hardest coding/agentic
|
|
200
|
+
// tasks, medium ≈ Sonnet 4.6 at high.
|
|
201
|
+
supportsVision: true,
|
|
202
|
+
supportsPromptCaching: true,
|
|
203
|
+
supportsTools: true,
|
|
204
|
+
webSearchToolType: 'web_search_20260209',
|
|
205
|
+
codeExecutionToolType: 'code_execution_20250825',
|
|
206
|
+
webFetchToolType: 'web_fetch_20260209',
|
|
207
|
+
// Standard pricing. Intro pricing ($2/$10) applies through 2026-08-31 —
|
|
208
|
+
// billed here at standard so metering never under-charges; revisit after.
|
|
209
|
+
inputPricePerMTok: 3,
|
|
210
|
+
outputPricePerMTok: 15,
|
|
211
|
+
// Anthropic 5-minute prompt cache: read 0.1× input, write 1.25× input.
|
|
212
|
+
cacheReadPricePerMTok: 0.3,
|
|
213
|
+
cacheWritePricePerMTok: 3.75,
|
|
214
|
+
knowledgeCutoff: '2026-01-01',
|
|
215
|
+
},
|
|
216
|
+
{
|
|
217
|
+
id: 'claude-opus-4-6',
|
|
218
|
+
provider: 'anthropic',
|
|
219
|
+
label: 'Claude Opus 4.6',
|
|
220
|
+
description: 'Previous-generation Opus — deep reasoning & complex tasks',
|
|
221
|
+
contextWindow: 1_000_000,
|
|
222
|
+
maxOutputTokens: 128_000,
|
|
223
|
+
supportsThinking: true,
|
|
224
|
+
thinkingBudgetTokens: 16_000,
|
|
225
|
+
thinkingConfigurable: true,
|
|
226
|
+
supportedEffortLevels: ['low', 'medium', 'high', 'max'],
|
|
227
|
+
defaultEffortLevel: 'medium',
|
|
228
|
+
// No xhigh on the 4.6 family; budget_tokens still accepted but deprecated —
|
|
229
|
+
// adaptive + effort is the recommended control.
|
|
230
|
+
supportsVision: true,
|
|
231
|
+
supportsPromptCaching: true,
|
|
232
|
+
supportsTools: true,
|
|
233
|
+
webSearchToolType: 'web_search_20260209',
|
|
234
|
+
codeExecutionToolType: 'code_execution_20250825',
|
|
235
|
+
webFetchToolType: 'web_fetch_20260209',
|
|
236
|
+
inputPricePerMTok: 5,
|
|
237
|
+
outputPricePerMTok: 25,
|
|
238
|
+
// Anthropic 5-minute prompt cache: read 0.1× input, write 1.25× input.
|
|
239
|
+
cacheReadPricePerMTok: 0.5,
|
|
240
|
+
cacheWritePricePerMTok: 6.25,
|
|
241
|
+
knowledgeCutoff: '2025-05-01',
|
|
242
|
+
// Superseded by claude-opus-4-8; kept selectable (Older models) + priceable.
|
|
243
|
+
deprecatedAt: '2026-06-16',
|
|
244
|
+
},
|
|
245
|
+
{
|
|
246
|
+
id: 'claude-sonnet-4-6',
|
|
247
|
+
provider: 'anthropic',
|
|
248
|
+
label: 'Claude Sonnet 4.6',
|
|
249
|
+
description: 'Previous-generation Sonnet — fast & balanced',
|
|
250
|
+
contextWindow: 1_000_000,
|
|
251
|
+
// Was wrongly 64K — Sonnet 4.6 supports 128K output (models overview).
|
|
252
|
+
maxOutputTokens: 128_000,
|
|
253
|
+
supportsThinking: true,
|
|
254
|
+
thinkingBudgetTokens: 10_000,
|
|
255
|
+
thinkingConfigurable: true,
|
|
256
|
+
supportedEffortLevels: ['low', 'medium', 'high', 'max'],
|
|
257
|
+
defaultEffortLevel: 'medium',
|
|
258
|
+
// No xhigh on the 4.6 family. Anthropic's recommended default for agentic
|
|
259
|
+
// coding on Sonnet 4.6 is medium (= M); high is its API default (= L).
|
|
260
|
+
supportsVision: true,
|
|
261
|
+
supportsPromptCaching: true,
|
|
262
|
+
supportsTools: true,
|
|
263
|
+
webSearchToolType: 'web_search_20260209',
|
|
264
|
+
codeExecutionToolType: 'code_execution_20250825',
|
|
265
|
+
webFetchToolType: 'web_fetch_20260209',
|
|
266
|
+
inputPricePerMTok: 3,
|
|
267
|
+
outputPricePerMTok: 15,
|
|
268
|
+
// Anthropic 5-minute prompt cache: read 0.1× input, write 1.25× input.
|
|
269
|
+
cacheReadPricePerMTok: 0.3,
|
|
270
|
+
cacheWritePricePerMTok: 3.75,
|
|
271
|
+
knowledgeCutoff: '2025-08-01',
|
|
272
|
+
// Superseded by claude-sonnet-5; kept selectable (Older models) + priceable.
|
|
273
|
+
deprecatedAt: '2026-07-07',
|
|
274
|
+
},
|
|
275
|
+
{
|
|
276
|
+
id: 'claude-haiku-4-5-20251001',
|
|
277
|
+
provider: 'anthropic',
|
|
278
|
+
label: 'Claude Haiku 4.5',
|
|
279
|
+
description: 'Fastest Anthropic — quick tasks & iteration',
|
|
280
|
+
contextWindow: 200_000,
|
|
281
|
+
maxOutputTokens: 64_000,
|
|
282
|
+
supportsThinking: true,
|
|
283
|
+
thinkingBudgetTokens: 8_000,
|
|
284
|
+
thinkingConfigurable: true,
|
|
285
|
+
// Haiku 4.5 does NOT support output_config.effort (400) or adaptive
|
|
286
|
+
// thinking — it stays on the budget_tokens path, so its levels are
|
|
287
|
+
// budget labels (effortBudgetTokens), not a native effort param.
|
|
288
|
+
supportedEffortLevels: ['4K', '8K', '16K', '32K'],
|
|
289
|
+
defaultEffortLevel: '8K',
|
|
290
|
+
effortBudgetTokens: { '4K': 4000, '8K': 8000, '16K': 16000, '32K': 32000 },
|
|
291
|
+
supportsVision: true,
|
|
292
|
+
supportsPromptCaching: true,
|
|
293
|
+
supportsTools: true,
|
|
294
|
+
webSearchToolType: 'web_search_20250305',
|
|
295
|
+
webFetchToolType: 'web_fetch_20250910',
|
|
296
|
+
inputPricePerMTok: 1,
|
|
297
|
+
outputPricePerMTok: 5,
|
|
298
|
+
// Anthropic 5-minute prompt cache: read 0.1× input, write 1.25× input.
|
|
299
|
+
cacheReadPricePerMTok: 0.1,
|
|
300
|
+
cacheWritePricePerMTok: 1.25,
|
|
301
|
+
knowledgeCutoff: '2025-02-01',
|
|
302
|
+
},
|
|
303
|
+
// ---------------------------------------------------------------------------
|
|
304
|
+
// OpenAI
|
|
305
|
+
// Verified: https://developers.openai.com/api/docs/pricing (2026-07-28)
|
|
306
|
+
// GPT-5.6 family GA 2026-07-09: gpt-5.6 is an alias for gpt-5.6-sol; Sol is
|
|
307
|
+
// the frontier tier, Terra the balanced tier, Luna the cheap/fast tier.
|
|
308
|
+
// Cached input is billed at 0.1× input. The official pricing page shows no
|
|
309
|
+
// cache-WRITE premium, but third-party trackers report 1.25× with a 30-min
|
|
310
|
+
// cache life for the 5.6 family — billed here at 1.25× (conservative;
|
|
311
|
+
// re-verify against the official docs). reasoning_effort values for 5.6 are
|
|
312
|
+
// not yet on the docs page — carried over from gpt-5.5 (low|medium|high|
|
|
313
|
+
// xhigh, default medium); re-verify. Long-context 2× price variants exist
|
|
314
|
+
// upstream — not modeled (same as the Gemini/Grok >200K tiers).
|
|
315
|
+
// ---------------------------------------------------------------------------
|
|
316
|
+
{
|
|
317
|
+
id: 'gpt-5.6-sol',
|
|
318
|
+
provider: 'openai',
|
|
319
|
+
label: 'GPT-5.6 Sol',
|
|
320
|
+
description: 'OpenAI frontier — the hardest coding & reasoning work',
|
|
321
|
+
// Reported as 1.05M; floored to 1M until the docs state it (understating
|
|
322
|
+
// only makes compaction slightly earlier — overstating risks overflow).
|
|
323
|
+
contextWindow: 1_000_000,
|
|
324
|
+
maxOutputTokens: 128_000,
|
|
325
|
+
supportsThinking: true,
|
|
326
|
+
thinkingBudgetTokens: 16_000,
|
|
327
|
+
thinkingConfigurable: true,
|
|
328
|
+
supportedEffortLevels: ['low', 'medium', 'high', 'xhigh'],
|
|
329
|
+
defaultEffortLevel: 'medium',
|
|
330
|
+
supportsVision: true,
|
|
331
|
+
supportsPromptCaching: true,
|
|
332
|
+
supportsTools: true,
|
|
333
|
+
// Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
|
|
334
|
+
toolsRequireReasoningOff: true,
|
|
335
|
+
webSearchToolType: 'web_search',
|
|
336
|
+
codeExecutionToolType: 'code_interpreter',
|
|
337
|
+
inputPricePerMTok: 5,
|
|
338
|
+
outputPricePerMTok: 30,
|
|
339
|
+
// Cached input 0.1× input; write premium reported 1.25× (see section note).
|
|
340
|
+
cacheReadPricePerMTok: 0.5,
|
|
341
|
+
cacheWritePricePerMTok: 6.25,
|
|
342
|
+
// Not published — best-effort estimate.
|
|
343
|
+
knowledgeCutoff: '2026-03-01',
|
|
344
|
+
},
|
|
345
|
+
{
|
|
346
|
+
id: 'gpt-5.6-terra',
|
|
347
|
+
provider: 'openai',
|
|
348
|
+
label: 'GPT-5.6 Terra',
|
|
349
|
+
description: 'Balanced OpenAI — everyday coding & tool use',
|
|
350
|
+
contextWindow: 1_000_000,
|
|
351
|
+
maxOutputTokens: 128_000,
|
|
352
|
+
supportsThinking: true,
|
|
353
|
+
thinkingBudgetTokens: 16_000,
|
|
354
|
+
thinkingConfigurable: true,
|
|
355
|
+
supportedEffortLevels: ['low', 'medium', 'high', 'xhigh'],
|
|
356
|
+
defaultEffortLevel: 'medium',
|
|
357
|
+
supportsVision: true,
|
|
358
|
+
supportsPromptCaching: true,
|
|
359
|
+
supportsTools: true,
|
|
360
|
+
// Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
|
|
361
|
+
toolsRequireReasoningOff: true,
|
|
362
|
+
webSearchToolType: 'web_search',
|
|
363
|
+
codeExecutionToolType: 'code_interpreter',
|
|
364
|
+
// Repriced 2026-07-30 (20% cut from $2.50/$15).
|
|
365
|
+
inputPricePerMTok: 2,
|
|
366
|
+
outputPricePerMTok: 12,
|
|
367
|
+
// Cached input 0.1× input; write premium reported 1.25× (see section note).
|
|
368
|
+
cacheReadPricePerMTok: 0.2,
|
|
369
|
+
cacheWritePricePerMTok: 2.5,
|
|
370
|
+
// Not published — best-effort estimate.
|
|
371
|
+
knowledgeCutoff: '2026-03-01',
|
|
372
|
+
},
|
|
373
|
+
{
|
|
374
|
+
id: 'gpt-5.6-luna',
|
|
375
|
+
provider: 'openai',
|
|
376
|
+
label: 'GPT-5.6 Luna',
|
|
377
|
+
description: 'Fast & cheap OpenAI — light tasks & subagents',
|
|
378
|
+
contextWindow: 1_000_000,
|
|
379
|
+
maxOutputTokens: 128_000,
|
|
380
|
+
supportsThinking: true,
|
|
381
|
+
thinkingBudgetTokens: 8_000,
|
|
382
|
+
thinkingConfigurable: true,
|
|
383
|
+
supportedEffortLevels: ['low', 'medium', 'high', 'xhigh'],
|
|
384
|
+
defaultEffortLevel: 'medium',
|
|
385
|
+
supportsVision: true,
|
|
386
|
+
supportsPromptCaching: true,
|
|
387
|
+
supportsTools: true,
|
|
388
|
+
// Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
|
|
389
|
+
toolsRequireReasoningOff: true,
|
|
390
|
+
webSearchToolType: 'web_search',
|
|
391
|
+
codeExecutionToolType: 'code_interpreter',
|
|
392
|
+
// Repriced 2026-07-30 (80% cut from $1/$6).
|
|
393
|
+
inputPricePerMTok: 0.2,
|
|
394
|
+
outputPricePerMTok: 1.2,
|
|
395
|
+
// Cached input 0.1× input; write premium reported 1.25× (see section note).
|
|
396
|
+
cacheReadPricePerMTok: 0.02,
|
|
397
|
+
cacheWritePricePerMTok: 0.25,
|
|
398
|
+
// Not published — best-effort estimate.
|
|
399
|
+
knowledgeCutoff: '2026-03-01',
|
|
400
|
+
},
|
|
401
|
+
{
|
|
402
|
+
id: 'gpt-5.5',
|
|
403
|
+
provider: 'openai',
|
|
404
|
+
label: 'GPT-5.5',
|
|
405
|
+
description: 'OpenAI flagship — strong coding & tool use',
|
|
406
|
+
contextWindow: 1_050_000,
|
|
407
|
+
maxOutputTokens: 128_000,
|
|
408
|
+
supportsThinking: true,
|
|
409
|
+
thinkingBudgetTokens: 16_000,
|
|
410
|
+
thinkingConfigurable: true,
|
|
411
|
+
supportedEffortLevels: ['low', 'medium', 'high', 'xhigh'],
|
|
412
|
+
defaultEffortLevel: 'medium',
|
|
413
|
+
// OpenAI default + agentic-coding rec is medium (= M); 'none' is not
|
|
414
|
+
// offered (it disables reasoning outright).
|
|
415
|
+
supportsVision: true,
|
|
416
|
+
supportsPromptCaching: true,
|
|
417
|
+
supportsTools: true,
|
|
418
|
+
// Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
|
|
419
|
+
toolsRequireReasoningOff: true,
|
|
420
|
+
webSearchToolType: 'web_search',
|
|
421
|
+
codeExecutionToolType: 'code_interpreter',
|
|
422
|
+
inputPricePerMTok: 5,
|
|
423
|
+
outputPricePerMTok: 30,
|
|
424
|
+
// OpenAI auto-caches at no write premium; cached input billed at 0.1× input.
|
|
425
|
+
cacheReadPricePerMTok: 0.5,
|
|
426
|
+
cacheWritePricePerMTok: 5,
|
|
427
|
+
knowledgeCutoff: '2025-12-01',
|
|
428
|
+
// Superseded by gpt-5.6-sol (same price); still listed as current by
|
|
429
|
+
// OpenAI. Selectable under "Older models".
|
|
430
|
+
deprecatedAt: '2026-07-09',
|
|
431
|
+
},
|
|
432
|
+
{
|
|
433
|
+
id: 'gpt-5.4',
|
|
434
|
+
provider: 'openai',
|
|
435
|
+
label: 'GPT-5.4',
|
|
436
|
+
description: 'Affordable OpenAI frontier — strong coding & tool use',
|
|
437
|
+
contextWindow: 1_050_000,
|
|
438
|
+
maxOutputTokens: 128_000,
|
|
439
|
+
supportsThinking: true,
|
|
440
|
+
thinkingBudgetTokens: 16_000,
|
|
441
|
+
thinkingConfigurable: true,
|
|
442
|
+
supportedEffortLevels: ['low', 'medium', 'high', 'xhigh'],
|
|
443
|
+
defaultEffortLevel: 'medium',
|
|
444
|
+
supportsVision: true,
|
|
445
|
+
supportsPromptCaching: true,
|
|
446
|
+
supportsTools: true,
|
|
447
|
+
// Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
|
|
448
|
+
toolsRequireReasoningOff: true,
|
|
449
|
+
webSearchToolType: 'web_search',
|
|
450
|
+
codeExecutionToolType: 'code_interpreter',
|
|
451
|
+
inputPricePerMTok: 2.5,
|
|
452
|
+
outputPricePerMTok: 15,
|
|
453
|
+
// OpenAI auto-caches at no write premium; cached input billed at 0.1× input.
|
|
454
|
+
cacheReadPricePerMTok: 0.25,
|
|
455
|
+
cacheWritePricePerMTok: 2.5,
|
|
456
|
+
knowledgeCutoff: '2025-08-31',
|
|
457
|
+
// OpenAI still lists gpt-5.4 as current, but gpt-5.6-terra covers this
|
|
458
|
+
// tier at the same price — moved to "Older models" (deprecatedAt is OUR
|
|
459
|
+
// picker taxonomy, not OpenAI's deprecations page).
|
|
460
|
+
deprecatedAt: '2026-07-28',
|
|
461
|
+
},
|
|
462
|
+
{
|
|
463
|
+
id: 'gpt-5.4-mini',
|
|
464
|
+
provider: 'openai',
|
|
465
|
+
label: 'GPT-5.4 mini',
|
|
466
|
+
description: 'Cheap & fast OpenAI — light coding tasks & subagents',
|
|
467
|
+
contextWindow: 400_000,
|
|
468
|
+
maxOutputTokens: 128_000,
|
|
469
|
+
supportsThinking: true,
|
|
470
|
+
thinkingBudgetTokens: 8_000,
|
|
471
|
+
thinkingConfigurable: true,
|
|
472
|
+
supportedEffortLevels: ['low', 'medium', 'high', 'xhigh'],
|
|
473
|
+
defaultEffortLevel: 'medium',
|
|
474
|
+
supportsVision: true,
|
|
475
|
+
supportsPromptCaching: true,
|
|
476
|
+
supportsTools: true,
|
|
477
|
+
// Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
|
|
478
|
+
toolsRequireReasoningOff: true,
|
|
479
|
+
webSearchToolType: 'web_search',
|
|
480
|
+
codeExecutionToolType: 'code_interpreter',
|
|
481
|
+
inputPricePerMTok: 0.75,
|
|
482
|
+
outputPricePerMTok: 4.5,
|
|
483
|
+
// OpenAI auto-caches at no write premium; cached input billed at 0.1× input.
|
|
484
|
+
cacheReadPricePerMTok: 0.075,
|
|
485
|
+
cacheWritePricePerMTok: 0.75,
|
|
486
|
+
knowledgeCutoff: '2025-08-31',
|
|
487
|
+
// OpenAI still lists gpt-5.4-mini as current, but like gpt-5.4 above it's
|
|
488
|
+
// superseded in our lineup (cheap/fast tier is better served by the newer
|
|
489
|
+
// models) — moved to "Older models" (deprecatedAt is OUR picker taxonomy,
|
|
490
|
+
// not OpenAI's deprecations page).
|
|
491
|
+
deprecatedAt: '2026-08-01',
|
|
492
|
+
},
|
|
493
|
+
// ---------------------------------------------------------------------------
|
|
494
|
+
// Google
|
|
495
|
+
// Verified: https://ai.google.dev/gemini-api/docs/models
|
|
496
|
+
// https://ai.google.dev/gemini-api/docs/pricing
|
|
497
|
+
// https://ai.google.dev/gemini-api/docs/thinking
|
|
498
|
+
// Reasoning control is `thinking_level` (minimal|low|medium|high — cannot be
|
|
499
|
+
// fully off on 3.x; combining with legacy thinking_budget returns 400).
|
|
500
|
+
// NOTE: the previous catalog carried a fictional "gemini-3.1-pro" GA id — no
|
|
501
|
+
// such model exists; the pro tier is (still) gemini-3.1-pro-preview. Safe to
|
|
502
|
+
// replace outright: the google bond has never been implemented/wired, so no
|
|
503
|
+
// historical usage can reference the old ids.
|
|
504
|
+
// ---------------------------------------------------------------------------
|
|
505
|
+
{
|
|
506
|
+
id: 'gemini-3.6-flash',
|
|
507
|
+
provider: 'google',
|
|
508
|
+
label: 'Gemini 3.6 Flash',
|
|
509
|
+
description: 'Google agentic flagship — frontier intelligence + grounding',
|
|
510
|
+
// Window/output not on the pricing page — carried over from 3.5-flash;
|
|
511
|
+
// re-verify against /docs/models.
|
|
512
|
+
contextWindow: 1_048_576,
|
|
513
|
+
maxOutputTokens: 65_536,
|
|
514
|
+
supportsThinking: true,
|
|
515
|
+
thinkingBudgetTokens: 10_000,
|
|
516
|
+
thinkingConfigurable: true,
|
|
517
|
+
// thinking_level assumed unchanged from 3.5-flash (low|medium|high);
|
|
518
|
+
// re-verify against /docs/thinking.
|
|
519
|
+
supportedEffortLevels: ['low', 'medium', 'high'],
|
|
520
|
+
defaultEffortLevel: 'medium',
|
|
521
|
+
supportsVision: true,
|
|
522
|
+
supportsPromptCaching: true,
|
|
523
|
+
supportsTools: true,
|
|
524
|
+
webSearchToolType: 'google_search',
|
|
525
|
+
codeExecutionToolType: 'code_execution',
|
|
526
|
+
webFetchToolType: 'url_context',
|
|
527
|
+
// GA 2026-07-21 — same input price as 3.5-flash, CHEAPER output ($7.50 vs $9).
|
|
528
|
+
inputPricePerMTok: 1.5,
|
|
529
|
+
outputPricePerMTok: 7.5,
|
|
530
|
+
// Gemini context cache: read $0.15/M (0.1× input), no write premium
|
|
531
|
+
// (storage billed separately per hour — not modeled).
|
|
532
|
+
cacheReadPricePerMTok: 0.15,
|
|
533
|
+
cacheWritePricePerMTok: 1.5,
|
|
534
|
+
// Not published — best-effort estimate.
|
|
535
|
+
knowledgeCutoff: '2026-01-01',
|
|
536
|
+
},
|
|
537
|
+
{
|
|
538
|
+
id: 'gemini-3.5-flash',
|
|
539
|
+
provider: 'google',
|
|
540
|
+
label: 'Gemini 3.5 Flash',
|
|
541
|
+
description: 'Previous Google agentic flagship — fast, 1M context',
|
|
542
|
+
contextWindow: 1_048_576,
|
|
543
|
+
maxOutputTokens: 65_536,
|
|
544
|
+
supportsThinking: true,
|
|
545
|
+
thinkingBudgetTokens: 10_000,
|
|
546
|
+
thinkingConfigurable: true,
|
|
547
|
+
supportedEffortLevels: ['low', 'medium', 'high'],
|
|
548
|
+
defaultEffortLevel: 'medium',
|
|
549
|
+
// thinking_level: default medium (= M); Google recommends high (= L) for
|
|
550
|
+
// advanced coding/multi-step planning. 'minimal' exists below low; no
|
|
551
|
+
// fourth upward tier, so XL is not offered.
|
|
552
|
+
supportsVision: true,
|
|
553
|
+
supportsPromptCaching: true,
|
|
554
|
+
supportsTools: true,
|
|
555
|
+
webSearchToolType: 'google_search',
|
|
556
|
+
codeExecutionToolType: 'code_execution',
|
|
557
|
+
webFetchToolType: 'url_context',
|
|
558
|
+
inputPricePerMTok: 1.5,
|
|
559
|
+
outputPricePerMTok: 9,
|
|
560
|
+
// Gemini context cache: read $0.15/M (0.1× input), no write premium
|
|
561
|
+
// (storage billed separately per hour — not modeled).
|
|
562
|
+
cacheReadPricePerMTok: 0.15,
|
|
563
|
+
cacheWritePricePerMTok: 1.5,
|
|
564
|
+
knowledgeCutoff: '2025-01-01',
|
|
565
|
+
// Superseded by gemini-3.6-flash (2026-07-21); still served upstream.
|
|
566
|
+
// Selectable under "Older models".
|
|
567
|
+
deprecatedAt: '2026-07-21',
|
|
568
|
+
},
|
|
569
|
+
{
|
|
570
|
+
id: 'gemini-3.1-pro-preview',
|
|
571
|
+
provider: 'google',
|
|
572
|
+
label: 'Gemini 3.1 Pro',
|
|
573
|
+
description: 'Google pro tier — deep reasoning (preview id; no GA id exists)',
|
|
574
|
+
contextWindow: 1_048_576,
|
|
575
|
+
maxOutputTokens: 65_536,
|
|
576
|
+
supportsThinking: true,
|
|
577
|
+
thinkingBudgetTokens: 10_000,
|
|
578
|
+
thinkingConfigurable: true,
|
|
579
|
+
supportedEffortLevels: ['low', 'medium', 'high'],
|
|
580
|
+
defaultEffortLevel: 'medium',
|
|
581
|
+
// thinking_level: low|medium|high only (no minimal). Google's API default
|
|
582
|
+
// for this model is high (= L); M offers the cost step-down.
|
|
583
|
+
supportsVision: true,
|
|
584
|
+
supportsPromptCaching: true,
|
|
585
|
+
supportsTools: true,
|
|
586
|
+
webSearchToolType: 'google_search',
|
|
587
|
+
codeExecutionToolType: 'code_execution',
|
|
588
|
+
webFetchToolType: 'url_context',
|
|
589
|
+
// ≤200K-token prompts; >200K bills $4/$18 (tiering not modeled).
|
|
590
|
+
inputPricePerMTok: 2,
|
|
591
|
+
outputPricePerMTok: 12,
|
|
592
|
+
// Gemini context cache: read $0.20/M (≤200K), no write premium (hourly
|
|
593
|
+
// storage not modeled).
|
|
594
|
+
cacheReadPricePerMTok: 0.2,
|
|
595
|
+
cacheWritePricePerMTok: 2,
|
|
596
|
+
knowledgeCutoff: '2025-01-01',
|
|
597
|
+
},
|
|
598
|
+
// ---------------------------------------------------------------------------
|
|
599
|
+
// xAI (Grok)
|
|
600
|
+
// Verified: https://docs.x.ai/developers/models + /developers/grok-4-5
|
|
601
|
+
// (2026-07-28)
|
|
602
|
+
// grok-4.5 (2026-07-08) is the flagship: 500K ctx, reasoning_effort
|
|
603
|
+
// low|medium|high default high, image input. grok-4.3 stays served as the
|
|
604
|
+
// value tier with the BIGGER 1M window (reasoning_effort none|low|medium|
|
|
605
|
+
// high, default low). Max output tokens still not documented for any model.
|
|
606
|
+
// ---------------------------------------------------------------------------
|
|
607
|
+
{
|
|
608
|
+
id: 'grok-4.5',
|
|
609
|
+
provider: 'xai',
|
|
610
|
+
label: 'Grok 4.5',
|
|
611
|
+
description: 'xAI frontier — coding, agentic tasks & knowledge work',
|
|
612
|
+
contextWindow: 500_000,
|
|
613
|
+
// Max output not documented by xAI — conservative cap.
|
|
614
|
+
maxOutputTokens: 128_000,
|
|
615
|
+
supportsThinking: true,
|
|
616
|
+
thinkingBudgetTokens: 16_000,
|
|
617
|
+
thinkingConfigurable: true,
|
|
618
|
+
// reasoning_effort low|medium|high, default high (docs.x.ai/developers/
|
|
619
|
+
// grok-4-5) — no 'none' tier on 4.5, unlike 4.3.
|
|
620
|
+
supportedEffortLevels: ['low', 'medium', 'high'],
|
|
621
|
+
defaultEffortLevel: 'high',
|
|
622
|
+
// Multimodal input (text + images), text out.
|
|
623
|
+
supportsVision: true,
|
|
624
|
+
supportsPromptCaching: true,
|
|
625
|
+
supportsTools: true,
|
|
626
|
+
// ≤200K-token prompts; ≥200K bills 2× ($4/$0.60/$12 — tiering not
|
|
627
|
+
// modeled, same as the Gemini 3.1 Pro >200K tier).
|
|
628
|
+
inputPricePerMTok: 2,
|
|
629
|
+
outputPricePerMTok: 6,
|
|
630
|
+
// xAI cached input billed at a flat $0.30/M, no write premium.
|
|
631
|
+
cacheReadPricePerMTok: 0.3,
|
|
632
|
+
cacheWritePricePerMTok: 2,
|
|
633
|
+
// Official (docs.x.ai): 2026-02-01.
|
|
634
|
+
knowledgeCutoff: '2026-02-01',
|
|
635
|
+
},
|
|
636
|
+
{
|
|
637
|
+
id: 'grok-4.3',
|
|
638
|
+
provider: 'xai',
|
|
639
|
+
label: 'Grok 4.3',
|
|
640
|
+
description: 'xAI value tier — fast reasoning, bigger 1M context',
|
|
641
|
+
contextWindow: 1_000_000,
|
|
642
|
+
maxOutputTokens: 128_000,
|
|
643
|
+
supportsThinking: true,
|
|
644
|
+
thinkingBudgetTokens: 16_000,
|
|
645
|
+
thinkingConfigurable: true,
|
|
646
|
+
supportedEffortLevels: ['none', 'low', 'medium', 'high'],
|
|
647
|
+
defaultEffortLevel: 'low',
|
|
648
|
+
// xAI's default (and its own retirement-routing choice for agentic
|
|
649
|
+
// workloads) is low (= M); none (= S) disables reasoning for a true fast
|
|
650
|
+
// mode; step up for hard debugging/architecture turns.
|
|
651
|
+
supportsVision: true,
|
|
652
|
+
supportsPromptCaching: true,
|
|
653
|
+
supportsTools: true,
|
|
654
|
+
inputPricePerMTok: 1.25,
|
|
655
|
+
outputPricePerMTok: 2.5,
|
|
656
|
+
// xAI cached input billed at a flat $0.20/M, no write premium.
|
|
657
|
+
cacheReadPricePerMTok: 0.2,
|
|
658
|
+
cacheWritePricePerMTok: 1.25,
|
|
659
|
+
knowledgeCutoff: '2025-12-01',
|
|
660
|
+
// Superseded by grok-4.5 as the xAI pick (4.3 keeps the bigger 1M window
|
|
661
|
+
// — the reason it stays selectable under "Older models").
|
|
662
|
+
deprecatedAt: '2026-07-28',
|
|
663
|
+
},
|
|
664
|
+
{
|
|
665
|
+
id: 'grok-build-0.1',
|
|
666
|
+
provider: 'xai',
|
|
667
|
+
label: 'Grok Build',
|
|
668
|
+
description: 'Agentic coding specialist — fast & cheap (public beta)',
|
|
669
|
+
contextWindow: 256_000,
|
|
670
|
+
// Max output not documented by xAI — conservative cap.
|
|
671
|
+
maxOutputTokens: 64_000,
|
|
672
|
+
supportsThinking: true,
|
|
673
|
+
thinkingBudgetTokens: 8_000,
|
|
674
|
+
// Reasoning is always on and NOT configurable (reasoning_effort is not
|
|
675
|
+
// honored on grok-build) — successor to grok-code-fast-1 (that slug now
|
|
676
|
+
// auto-routes here).
|
|
677
|
+
thinkingConfigurable: false,
|
|
678
|
+
supportsVision: true,
|
|
679
|
+
supportsPromptCaching: true,
|
|
680
|
+
supportsTools: true,
|
|
681
|
+
inputPricePerMTok: 1,
|
|
682
|
+
outputPricePerMTok: 2,
|
|
683
|
+
// xAI cached input billed at a flat $0.20/M, no write premium.
|
|
684
|
+
cacheReadPricePerMTok: 0.2,
|
|
685
|
+
cacheWritePricePerMTok: 1,
|
|
686
|
+
// Not published by xAI — best-effort estimate (grok-4-generation base).
|
|
687
|
+
knowledgeCutoff: '2025-06-01',
|
|
688
|
+
// Niche coding beta; grok-4.5 is the xAI pick — kept out of the main list.
|
|
689
|
+
deprecatedAt: '2026-07-28',
|
|
690
|
+
},
|
|
691
|
+
{
|
|
692
|
+
id: 'grok-4.20-multi-agent-beta-0309',
|
|
693
|
+
provider: 'xai',
|
|
694
|
+
label: 'Grok 4.20',
|
|
695
|
+
description: 'Older xAI flagship — multi-agent',
|
|
696
|
+
// Current xAI docs list 1M (the old 2M figure is stale). This beta slug is
|
|
697
|
+
// an alias of the canonical grok-4.20-multi-agent-0309 — kept under the
|
|
698
|
+
// original id so saved selections + historical usage stay priceable.
|
|
699
|
+
contextWindow: 1_000_000,
|
|
700
|
+
maxOutputTokens: 128_000,
|
|
701
|
+
supportsThinking: true,
|
|
702
|
+
thinkingBudgetTokens: 16_000,
|
|
703
|
+
// On the multi-agent model reasoning_effort controls AGENT COUNT, not
|
|
704
|
+
// reasoning depth — do not drive it from the effort setting.
|
|
705
|
+
thinkingConfigurable: false,
|
|
706
|
+
supportsVision: true,
|
|
707
|
+
supportsPromptCaching: true,
|
|
708
|
+
supportsTools: true,
|
|
709
|
+
// Repriced by xAI when grok-4.3 launched (was $2/$6).
|
|
710
|
+
inputPricePerMTok: 1.25,
|
|
711
|
+
outputPricePerMTok: 2.5,
|
|
712
|
+
// xAI cached input billed at a flat $0.20/M, no write premium.
|
|
713
|
+
cacheReadPricePerMTok: 0.2,
|
|
714
|
+
cacheWritePricePerMTok: 1.25,
|
|
715
|
+
knowledgeCutoff: '2024-11-01',
|
|
716
|
+
deprecatedAt: '2026-04-30',
|
|
717
|
+
// DISABLED, not deleted: xAI rejects this model on the chat-completions
|
|
718
|
+
// endpoint outright — "Multi Agent requests are not allowed on chat
|
|
719
|
+
// completions" (400 on every call, verified live 2026-07-30) — so offering
|
|
720
|
+
// it in the picker only hands users a model that cannot answer. Their API
|
|
721
|
+
// also lists the canonical slug as `grok-4.20-multi-agent-0309` (no
|
|
722
|
+
// `beta-`), with `grok-4.20-0309-reasoning` / `-non-reasoning` as the
|
|
723
|
+
// chat-completions-capable variants if this family is wanted back.
|
|
724
|
+
// Stays in the catalogue so saved selections and past usage remain priceable.
|
|
725
|
+
disabled: true,
|
|
726
|
+
},
|
|
727
|
+
{
|
|
728
|
+
id: 'grok-code-fast-1',
|
|
729
|
+
provider: 'xai',
|
|
730
|
+
label: 'Grok Code',
|
|
731
|
+
description: 'Code specialist — fast & cheap',
|
|
732
|
+
contextWindow: 256_000,
|
|
733
|
+
maxOutputTokens: 64_000,
|
|
734
|
+
supportsThinking: true,
|
|
735
|
+
thinkingBudgetTokens: 8_000,
|
|
736
|
+
thinkingConfigurable: false,
|
|
737
|
+
supportsVision: false,
|
|
738
|
+
supportsPromptCaching: true,
|
|
739
|
+
supportsTools: true,
|
|
740
|
+
inputPricePerMTok: 0.2,
|
|
741
|
+
outputPricePerMTok: 1.5,
|
|
742
|
+
// xAI cached input billed at a discount (≈0.25× input), no write premium.
|
|
743
|
+
cacheReadPricePerMTok: 0.05,
|
|
744
|
+
cacheWritePricePerMTok: 0.2,
|
|
745
|
+
knowledgeCutoff: '2024-11-01',
|
|
746
|
+
// Deprecated by xAI 2026-05-15 (retires 2026-08-15; the slug auto-routes to
|
|
747
|
+
// grok-build-0.1 until then) — disabled: removed from selection + the
|
|
748
|
+
// public listing, but getModel() still prices any historical usage. NEVER
|
|
749
|
+
// delete.
|
|
750
|
+
disabled: true,
|
|
751
|
+
},
|
|
752
|
+
// ---------------------------------------------------------------------------
|
|
753
|
+
// DeepSeek
|
|
754
|
+
// Verified: https://api-docs.deepseek.com/quick_start/pricing
|
|
755
|
+
// https://api-docs.deepseek.com/guides/thinking_mode
|
|
756
|
+
// https://api-docs.deepseek.com/updates/ (2026-07-31)
|
|
757
|
+
// 2026-07-31: DeepSeek-V4-Flash OFFICIAL API launched in public beta — the
|
|
758
|
+
// SAME `deepseek-v4-flash` id now serves the re-post-trained 0731 build
|
|
759
|
+
// (same architecture/size; much stronger agent benchmarks — beats
|
|
760
|
+
// V4-Pro-Preview on Terminal Bench 2.1 / DeepSWE). No pricing/limit/
|
|
761
|
+
// capability changes. V4-Pro official release "coming soon" — re-verify
|
|
762
|
+
// pricing THEN (the announced peak-hour 2× was tied to the V4 official
|
|
763
|
+
// rollout and is still not on the rate card).
|
|
764
|
+
// OpenAI/Anthropic-compatible API; text/code only (no vision); 1M context,
|
|
765
|
+
// 384K max output, automatic context (prompt) caching with ABSOLUTE cache-hit
|
|
766
|
+
// prices (~1/50–1/120 of miss — not the old 0.1× rule). Launch discount made
|
|
767
|
+
// PERMANENT 2026-05-23 (Pro $1.74/$3.48 → $0.435/$0.87). Peak-hour 2×
|
|
768
|
+
// pricing announced for the mid-Jul 2026 "V4 official" release — re-verify
|
|
769
|
+
// then. Thinking now defaults ENABLED upstream and supports tool calling
|
|
770
|
+
// (reasoning_effort: high|max), BUT tool loops in thinking mode must replay
|
|
771
|
+
// assistant reasoning_content on every subsequent request (400 on omission).
|
|
772
|
+
// The bond explicitly sends thinking:{type:"disabled"} — Synthase runs
|
|
773
|
+
// DeepSeek as a non-thinking executor (Sonnet plans; DeepSeek executes).
|
|
774
|
+
// ---------------------------------------------------------------------------
|
|
775
|
+
{
|
|
776
|
+
id: 'deepseek-v4-pro',
|
|
777
|
+
provider: 'deepseek',
|
|
778
|
+
label: 'DeepSeek V4 Pro',
|
|
779
|
+
description: 'Frontier-class — rivals top models at low cost',
|
|
780
|
+
contextWindow: 1_000_000,
|
|
781
|
+
maxOutputTokens: 384_000,
|
|
782
|
+
// Run non-thinking (see section note): the executor tool loop would have to
|
|
783
|
+
// replay reasoning_content across every turn in thinking mode.
|
|
784
|
+
supportsThinking: false,
|
|
785
|
+
thinkingBudgetTokens: 0,
|
|
786
|
+
thinkingConfigurable: false,
|
|
787
|
+
supportsVision: false,
|
|
788
|
+
supportsPromptCaching: true,
|
|
789
|
+
supportsTools: true,
|
|
790
|
+
inputPricePerMTok: 0.435,
|
|
791
|
+
outputPricePerMTok: 0.87,
|
|
792
|
+
// DeepSeek automatic context cache: absolute cache-hit price ($/M).
|
|
793
|
+
cacheReadPricePerMTok: 0.003625,
|
|
794
|
+
cacheWritePricePerMTok: 0.435,
|
|
795
|
+
// Native-China DEFAULT (owner decision 2026-08-01): the US re-host
|
|
796
|
+
// (DeepInfra) bills ~3× list and ~28× cache reads, and agentic input is
|
|
797
|
+
// ~94% cache hits, so US processing ran ~5.7× native on real traffic.
|
|
798
|
+
// Users opt into US per model via the picker's region control.
|
|
799
|
+
regions: ['cn', 'us'],
|
|
800
|
+
// The free tier PLANS with this model on the cheap native host (it is the
|
|
801
|
+
// molecule-dev FREE_TIER_MODELS.plan), so CN is free-tier selectable; the
|
|
802
|
+
// ~3× US re-host stays paid-only (free users switch to Flash for US).
|
|
803
|
+
freeTierRegions: ['cn'],
|
|
804
|
+
// US = DeepInfra, verified 2026-08-01 via api.deepinfra.com/models/
|
|
805
|
+
// deepseek-ai/DeepSeek-V4-Pro. No cache-write premium (omitted → region
|
|
806
|
+
// input rate).
|
|
807
|
+
regionPricing: {
|
|
808
|
+
us: { inputPricePerMTok: 1.3, outputPricePerMTok: 2.6, cacheReadPricePerMTok: 0.1 },
|
|
809
|
+
},
|
|
810
|
+
// The announced peak-hour 2× surcharge (Beijing business hours) is STILL
|
|
811
|
+
// NOT ACTIVE as of 2026-07-28 — the official rate card lists a single flat
|
|
812
|
+
// rate, and no switch-over date is published. The pre-wired windows were
|
|
813
|
+
// REMOVED: they had been over-billing every peak-window turn 2× for weeks
|
|
814
|
+
// (this is the free-tier default model, so that directly shrank free
|
|
815
|
+
// users' allowances). Re-add via `peakPricing` the day DeepSeek's rate
|
|
816
|
+
// card actually shows the surcharge.
|
|
817
|
+
// Not published by DeepSeek — best-effort estimate.
|
|
818
|
+
knowledgeCutoff: '2025-07-01',
|
|
819
|
+
},
|
|
820
|
+
{
|
|
821
|
+
id: 'deepseek-v4-flash',
|
|
822
|
+
provider: 'deepseek',
|
|
823
|
+
label: 'DeepSeek V4 Flash',
|
|
824
|
+
description: 'Ultra-cheap & fast — economical agentic coding',
|
|
825
|
+
contextWindow: 1_000_000,
|
|
826
|
+
maxOutputTokens: 384_000,
|
|
827
|
+
// Non-thinking (see section note) — also the right tradeoff for a cheap,
|
|
828
|
+
// fast tool-calling executor.
|
|
829
|
+
supportsThinking: false,
|
|
830
|
+
thinkingBudgetTokens: 0,
|
|
831
|
+
thinkingConfigurable: false,
|
|
832
|
+
supportsVision: false,
|
|
833
|
+
supportsPromptCaching: true,
|
|
834
|
+
supportsTools: true,
|
|
835
|
+
// Free-tier default: cheapest model + a fast, non-thinking tool-calling
|
|
836
|
+
// executor — the model the IDE picks when none is chosen. Exactly one model
|
|
837
|
+
// in this catalog may carry freeTier (enforced by lookup.test.ts).
|
|
838
|
+
freeTier: true,
|
|
839
|
+
inputPricePerMTok: 0.14,
|
|
840
|
+
outputPricePerMTok: 0.28,
|
|
841
|
+
// DeepSeek automatic context cache: absolute cache-hit price ($/M).
|
|
842
|
+
cacheReadPricePerMTok: 0.0028,
|
|
843
|
+
cacheWritePricePerMTok: 0.14,
|
|
844
|
+
// Native-China default, matching deepseek-v4-pro (see its note) — even
|
|
845
|
+
// though Flash's US list price is BELOW native, its cache reads are 6.4×,
|
|
846
|
+
// and the plan/execute pair defaults to one region deliberately.
|
|
847
|
+
regions: ['cn', 'us'],
|
|
848
|
+
// US = DeepInfra, verified 2026-08-01 (api.deepinfra.com/models/…V4-Flash).
|
|
849
|
+
regionPricing: {
|
|
850
|
+
us: { inputPricePerMTok: 0.09, outputPricePerMTok: 0.18, cacheReadPricePerMTok: 0.018 },
|
|
851
|
+
},
|
|
852
|
+
// Peak-hour surcharge NOT active (see deepseek-v4-pro) — windows removed.
|
|
853
|
+
// Not published by DeepSeek — best-effort estimate.
|
|
854
|
+
knowledgeCutoff: '2025-07-01',
|
|
855
|
+
},
|
|
856
|
+
// ---------------------------------------------------------------------------
|
|
857
|
+
// Moonshot (Kimi)
|
|
858
|
+
// Verified: https://platform.kimi.ai/docs/models +
|
|
859
|
+
// /docs/guide/use-kimi-k2-thinking-model (2026-07-28).
|
|
860
|
+
// The moonshot bond now supports PRESERVED THINKING (ChatMessage.reasoning →
|
|
861
|
+
// reasoning_content replay through tool loops), which unblocked the two
|
|
862
|
+
// previously-excluded models: kimi-k3 (2026-07-16 flagship — 2.8T MoE, 1M
|
|
863
|
+
// ctx, forced thinking, reasoning_effort low|high|max default max upstream)
|
|
864
|
+
// and kimi-k2.7-code (coding flagship — forced thinking, no depth knob).
|
|
865
|
+
// kimi-k2.x thinking stays on/off only; the bond disables it for those by
|
|
866
|
+
// default (KIMI_REASONING_EFFORT env tunes it).
|
|
867
|
+
// ---------------------------------------------------------------------------
|
|
868
|
+
{
|
|
869
|
+
id: 'kimi-k3',
|
|
870
|
+
provider: 'moonshot',
|
|
871
|
+
label: 'Kimi K3',
|
|
872
|
+
description: 'Moonshot flagship — 2.8T open weights, 1M context, multimodal',
|
|
873
|
+
contextWindow: 1_000_000,
|
|
874
|
+
// Max output not documented — conservative cap (matches the K2.x family).
|
|
875
|
+
maxOutputTokens: 65_535,
|
|
876
|
+
supportsThinking: true,
|
|
877
|
+
thinkingBudgetTokens: 8_000,
|
|
878
|
+
// Thinking is FORCED ON; depth rides reasoning_effort (low|high|max). The
|
|
879
|
+
// upstream default is max — we default to high so agentic loops aren't
|
|
880
|
+
// pinned to the slowest/most expensive tier unless the user asks for it.
|
|
881
|
+
thinkingConfigurable: true,
|
|
882
|
+
supportedEffortLevels: ['low', 'high', 'max'],
|
|
883
|
+
defaultEffortLevel: 'high',
|
|
884
|
+
// Text + image + video input.
|
|
885
|
+
supportsVision: true,
|
|
886
|
+
supportsPromptCaching: true,
|
|
887
|
+
supportsTools: true,
|
|
888
|
+
inputPricePerMTok: 3,
|
|
889
|
+
outputPricePerMTok: 15,
|
|
890
|
+
// Automatic context cache: absolute cache-hit price ($0.30/M = 0.1× input).
|
|
891
|
+
cacheReadPricePerMTok: 0.3,
|
|
892
|
+
cacheWritePricePerMTok: 3,
|
|
893
|
+
// No US re-host exists (not on DeepInfra) — pinned to native China.
|
|
894
|
+
regions: ['cn'],
|
|
895
|
+
// Not published — best-effort estimate.
|
|
896
|
+
knowledgeCutoff: '2026-01-01',
|
|
897
|
+
},
|
|
898
|
+
{
|
|
899
|
+
id: 'kimi-k2.7-code',
|
|
900
|
+
provider: 'moonshot',
|
|
901
|
+
label: 'Kimi K2.7 Code',
|
|
902
|
+
description: 'Moonshot coding specialist — token-efficient agentic coding',
|
|
903
|
+
contextWindow: 262_144,
|
|
904
|
+
maxOutputTokens: 65_535,
|
|
905
|
+
supportsThinking: true,
|
|
906
|
+
thinkingBudgetTokens: 8_000,
|
|
907
|
+
// Thinking forced on, NO depth knob (reasoning_effort unsupported here) —
|
|
908
|
+
// the bond sends neither param and replays reasoning_content.
|
|
909
|
+
thinkingConfigurable: false,
|
|
910
|
+
// Coding model — vision not documented; conservative.
|
|
911
|
+
supportsVision: false,
|
|
912
|
+
supportsPromptCaching: true,
|
|
913
|
+
supportsTools: true,
|
|
914
|
+
inputPricePerMTok: 0.95,
|
|
915
|
+
outputPricePerMTok: 4,
|
|
916
|
+
// Automatic context cache: absolute cache-hit price ($0.19/M = 0.2× input).
|
|
917
|
+
cacheReadPricePerMTok: 0.19,
|
|
918
|
+
cacheWritePricePerMTok: 0.95,
|
|
919
|
+
// US default (DeepInfra bills below native here). Verified 2026-08-01.
|
|
920
|
+
regions: ['us', 'cn'],
|
|
921
|
+
regionPricing: {
|
|
922
|
+
us: { inputPricePerMTok: 0.74, outputPricePerMTok: 3.5, cacheReadPricePerMTok: 0.15 },
|
|
923
|
+
},
|
|
924
|
+
// Not published — best-effort estimate.
|
|
925
|
+
knowledgeCutoff: '2025-10-01',
|
|
926
|
+
// kimi-k3 is the Moonshot pick; the coding specialist stays selectable
|
|
927
|
+
// under "Older models" for anyone who wants the cheaper tier.
|
|
928
|
+
deprecatedAt: '2026-07-28',
|
|
929
|
+
},
|
|
930
|
+
{
|
|
931
|
+
id: 'kimi-k2.6',
|
|
932
|
+
provider: 'moonshot',
|
|
933
|
+
label: 'Kimi K2.6',
|
|
934
|
+
description: 'Multimodal agent — strong sequential tool use',
|
|
935
|
+
contextWindow: 262_144,
|
|
936
|
+
// Output shares the 256K context window (max_tokens default 32K upstream);
|
|
937
|
+
// practical cap.
|
|
938
|
+
maxOutputTokens: 65_535,
|
|
939
|
+
supportsThinking: true,
|
|
940
|
+
thinkingBudgetTokens: 8_000,
|
|
941
|
+
// Thinking is on/off only (default on upstream; the bond disables it by
|
|
942
|
+
// default for the executor loop — tune via KIMI_REASONING_EFFORT).
|
|
943
|
+
thinkingConfigurable: false,
|
|
944
|
+
supportsVision: true,
|
|
945
|
+
supportsPromptCaching: true,
|
|
946
|
+
supportsTools: true,
|
|
947
|
+
// Native platform.kimi.ai pricing (the old $0.68/$3.41 was OpenRouter's
|
|
948
|
+
// blended third-party rate).
|
|
949
|
+
inputPricePerMTok: 0.95,
|
|
950
|
+
outputPricePerMTok: 4,
|
|
951
|
+
cacheReadPricePerMTok: 0.16,
|
|
952
|
+
cacheWritePricePerMTok: 0.95,
|
|
953
|
+
// US default (DeepInfra bills below native here). Verified 2026-08-01.
|
|
954
|
+
regions: ['us', 'cn'],
|
|
955
|
+
regionPricing: {
|
|
956
|
+
us: { inputPricePerMTok: 0.75, outputPricePerMTok: 3.5, cacheReadPricePerMTok: 0.15 },
|
|
957
|
+
},
|
|
958
|
+
knowledgeCutoff: '2025-04-01',
|
|
959
|
+
// Superseded by kimi-k3; moved to "Older models".
|
|
960
|
+
deprecatedAt: '2026-07-28',
|
|
961
|
+
},
|
|
962
|
+
{
|
|
963
|
+
id: 'kimi-k2.5',
|
|
964
|
+
provider: 'moonshot',
|
|
965
|
+
label: 'Kimi K2.5',
|
|
966
|
+
description: 'Older Kimi — strong sequential tool use',
|
|
967
|
+
contextWindow: 262_144,
|
|
968
|
+
maxOutputTokens: 65_535,
|
|
969
|
+
supportsThinking: true,
|
|
970
|
+
thinkingBudgetTokens: 8_000,
|
|
971
|
+
// Thinking on/off only; no preserved-thinking support upstream.
|
|
972
|
+
thinkingConfigurable: false,
|
|
973
|
+
supportsVision: true,
|
|
974
|
+
supportsPromptCaching: true,
|
|
975
|
+
supportsTools: true,
|
|
976
|
+
// Native platform.kimi.ai pricing.
|
|
977
|
+
inputPricePerMTok: 0.6,
|
|
978
|
+
outputPricePerMTok: 3,
|
|
979
|
+
cacheReadPricePerMTok: 0.1,
|
|
980
|
+
cacheWritePricePerMTok: 0.6,
|
|
981
|
+
// US default (DeepInfra bills below native here). Verified 2026-08-01.
|
|
982
|
+
regions: ['us', 'cn'],
|
|
983
|
+
regionPricing: {
|
|
984
|
+
us: { inputPricePerMTok: 0.45, outputPricePerMTok: 2.25, cacheReadPricePerMTok: 0.07 },
|
|
985
|
+
},
|
|
986
|
+
knowledgeCutoff: '2024-04-01',
|
|
987
|
+
// Superseded by kimi-k2.6 (still served upstream, no announced retirement);
|
|
988
|
+
// kept selectable (Older models) + priceable.
|
|
989
|
+
deprecatedAt: '2026-04-01',
|
|
990
|
+
},
|
|
991
|
+
// ---------------------------------------------------------------------------
|
|
992
|
+
// MiniMax
|
|
993
|
+
// Verified: https://platform.minimax.io/docs/guides/models-intro
|
|
994
|
+
// https://platform.minimax.io/docs/guides/pricing-paygo.md
|
|
995
|
+
// minimax-m3 (2026-06-01) is the flagship: 1M context (≥512K guaranteed),
|
|
996
|
+
// multimodal (text+image+video in), thinking adaptive|disabled (no budget /
|
|
997
|
+
// effort levels). Run non-thinking like the DeepSeek executors — with
|
|
998
|
+
// thinking on, reasoning must be carried through tool loops. m2.7 thinking
|
|
999
|
+
// CANNOT be disabled upstream.
|
|
1000
|
+
// ---------------------------------------------------------------------------
|
|
1001
|
+
{
|
|
1002
|
+
id: 'minimax-m3',
|
|
1003
|
+
provider: 'minimax',
|
|
1004
|
+
label: 'MiniMax M3',
|
|
1005
|
+
description: 'Agentic flagship — 1M context, multimodal, great value',
|
|
1006
|
+
contextWindow: 1_048_576,
|
|
1007
|
+
// Upstream recommends 128K max_tokens (hard max 512K).
|
|
1008
|
+
maxOutputTokens: 131_072,
|
|
1009
|
+
// Run non-thinking (the bond sends thinking:{type:"disabled"} for M3);
|
|
1010
|
+
// native control is adaptive|disabled only — no depth levels.
|
|
1011
|
+
supportsThinking: false,
|
|
1012
|
+
thinkingBudgetTokens: 0,
|
|
1013
|
+
thinkingConfigurable: false,
|
|
1014
|
+
supportsVision: true,
|
|
1015
|
+
supportsPromptCaching: true,
|
|
1016
|
+
supportsTools: true,
|
|
1017
|
+
// ≤512K-token prompts; >512K bills 2× (tiering not modeled).
|
|
1018
|
+
inputPricePerMTok: 0.3,
|
|
1019
|
+
outputPricePerMTok: 1.2,
|
|
1020
|
+
cacheReadPricePerMTok: 0.06,
|
|
1021
|
+
// M3 cache-write price not published — M2.7's rate (≥ input as required).
|
|
1022
|
+
cacheWritePricePerMTok: 0.375,
|
|
1023
|
+
// US default. DeepInfra list matches native; only the cache write differs
|
|
1024
|
+
// (no premium → region input rate). Verified 2026-08-01.
|
|
1025
|
+
regions: ['us', 'cn'],
|
|
1026
|
+
regionPricing: {
|
|
1027
|
+
us: { inputPricePerMTok: 0.3, outputPricePerMTok: 1.2, cacheReadPricePerMTok: 0.06 },
|
|
1028
|
+
},
|
|
1029
|
+
// From the official HF chat template ("Knowledge cutoff: January 2026").
|
|
1030
|
+
knowledgeCutoff: '2026-01-01',
|
|
1031
|
+
},
|
|
1032
|
+
{
|
|
1033
|
+
id: 'minimax-m2.7',
|
|
1034
|
+
provider: 'minimax',
|
|
1035
|
+
label: 'MiniMax M2.7',
|
|
1036
|
+
description: 'Agentic productivity — strong value for coding',
|
|
1037
|
+
contextWindow: 204_800,
|
|
1038
|
+
maxOutputTokens: 196_608,
|
|
1039
|
+
supportsThinking: true,
|
|
1040
|
+
thinkingBudgetTokens: 8_000,
|
|
1041
|
+
// Thinking is ALWAYS ON for M2.x (cannot be disabled) — no depth control.
|
|
1042
|
+
thinkingConfigurable: false,
|
|
1043
|
+
supportsVision: false,
|
|
1044
|
+
supportsPromptCaching: true,
|
|
1045
|
+
supportsTools: true,
|
|
1046
|
+
// Native platform repriced (was $0.25/$1.00).
|
|
1047
|
+
inputPricePerMTok: 0.3,
|
|
1048
|
+
outputPricePerMTok: 1.2,
|
|
1049
|
+
cacheReadPricePerMTok: 0.06,
|
|
1050
|
+
cacheWritePricePerMTok: 0.375,
|
|
1051
|
+
// US default (DeepInfra still bills the pre-reprice rate). Verified
|
|
1052
|
+
// 2026-08-01.
|
|
1053
|
+
regions: ['us', 'cn'],
|
|
1054
|
+
regionPricing: {
|
|
1055
|
+
us: { inputPricePerMTok: 0.25, outputPricePerMTok: 1, cacheReadPricePerMTok: 0.05 },
|
|
1056
|
+
},
|
|
1057
|
+
knowledgeCutoff: '2025-09-01',
|
|
1058
|
+
// Superseded by minimax-m3 (same price, 1M ctx, multimodal); moved to
|
|
1059
|
+
// "Older models".
|
|
1060
|
+
deprecatedAt: '2026-07-28',
|
|
1061
|
+
},
|
|
1062
|
+
{
|
|
1063
|
+
id: 'minimax-m2.5',
|
|
1064
|
+
provider: 'minimax',
|
|
1065
|
+
label: 'MiniMax M2.5',
|
|
1066
|
+
description: 'Older MiniMax — strong value for coding',
|
|
1067
|
+
contextWindow: 196_608,
|
|
1068
|
+
maxOutputTokens: 196_608,
|
|
1069
|
+
supportsThinking: true,
|
|
1070
|
+
thinkingBudgetTokens: 8_000,
|
|
1071
|
+
// Thinking always on for M2.x — no depth control.
|
|
1072
|
+
thinkingConfigurable: false,
|
|
1073
|
+
supportsVision: false,
|
|
1074
|
+
supportsPromptCaching: true,
|
|
1075
|
+
supportsTools: true,
|
|
1076
|
+
inputPricePerMTok: 0.3,
|
|
1077
|
+
outputPricePerMTok: 1.2,
|
|
1078
|
+
cacheReadPricePerMTok: 0.03,
|
|
1079
|
+
cacheWritePricePerMTok: 0.375,
|
|
1080
|
+
// No US re-host exists (not on DeepInfra) — pinned to native China.
|
|
1081
|
+
regions: ['cn'],
|
|
1082
|
+
knowledgeCutoff: '2025-01-01',
|
|
1083
|
+
// Superseded by minimax-m3 (legacy upstream, still served); kept selectable
|
|
1084
|
+
// (Older models) + priceable.
|
|
1085
|
+
deprecatedAt: '2026-03-18',
|
|
1086
|
+
},
|
|
1087
|
+
// ---------------------------------------------------------------------------
|
|
1088
|
+
// Alibaba (Qwen)
|
|
1089
|
+
// Verified: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
|
|
1090
|
+
// https://www.alibabacloud.com/help/en/model-studio/qwen-coder
|
|
1091
|
+
// https://openrouter.ai/qwen/qwen3.7-max
|
|
1092
|
+
// qwen3.7-max (2026-05-21) is the agentic flagship — Alibaba's own Qwen-Coder
|
|
1093
|
+
// docs now recommend the general-purpose models over Qwen-Coder. Its thinking
|
|
1094
|
+
// uses enable_thinking (default ON for the 3.7 series) + thinking_budget
|
|
1095
|
+
// (token cap) — a real budget param, so effort scales the budget. The
|
|
1096
|
+
// qwen3-coder models are NON-thinking (previous catalog entry was wrong).
|
|
1097
|
+
// Prices are DashScope international list rates (the bond calls DashScope,
|
|
1098
|
+
// not OpenRouter; a 50%-off promo currently applies — billed at list).
|
|
1099
|
+
// ---------------------------------------------------------------------------
|
|
1100
|
+
{
|
|
1101
|
+
id: 'qwen3.7-max',
|
|
1102
|
+
provider: 'alibaba',
|
|
1103
|
+
label: 'Qwen3.7 Max',
|
|
1104
|
+
description: 'Alibaba agentic flagship — 1M context, hybrid thinking',
|
|
1105
|
+
contextWindow: 1_000_000,
|
|
1106
|
+
maxOutputTokens: 65_536,
|
|
1107
|
+
supportsThinking: true,
|
|
1108
|
+
thinkingBudgetTokens: 8_000,
|
|
1109
|
+
// enable_thinking + thinking_budget: a controllable token budget → effort
|
|
1110
|
+
// scales the budget (no native level names).
|
|
1111
|
+
thinkingConfigurable: true,
|
|
1112
|
+
supportedEffortLevels: ['4K', '8K', '16K', '32K'],
|
|
1113
|
+
defaultEffortLevel: '8K',
|
|
1114
|
+
effortBudgetTokens: { '4K': 4000, '8K': 8000, '16K': 16000, '32K': 32000 },
|
|
1115
|
+
supportsVision: false,
|
|
1116
|
+
supportsPromptCaching: true,
|
|
1117
|
+
supportsTools: true,
|
|
1118
|
+
inputPricePerMTok: 2.5,
|
|
1119
|
+
outputPricePerMTok: 7.5,
|
|
1120
|
+
// Implicit context cache: read ≈0.2× input, no write premium.
|
|
1121
|
+
cacheReadPricePerMTok: 0.5,
|
|
1122
|
+
cacheWritePricePerMTok: 2.5,
|
|
1123
|
+
// US default. DeepInfra bills identical rates (no regionPricing needed).
|
|
1124
|
+
// Verified 2026-08-01.
|
|
1125
|
+
regions: ['us', 'cn'],
|
|
1126
|
+
// Not published by Alibaba — best-effort estimate.
|
|
1127
|
+
knowledgeCutoff: '2026-01-01',
|
|
1128
|
+
},
|
|
1129
|
+
{
|
|
1130
|
+
id: 'qwen3-coder-plus',
|
|
1131
|
+
provider: 'alibaba',
|
|
1132
|
+
label: 'Qwen3 Coder Plus',
|
|
1133
|
+
description: 'Coding specialist — 1M context',
|
|
1134
|
+
contextWindow: 1_000_000,
|
|
1135
|
+
maxOutputTokens: 65_536,
|
|
1136
|
+
// Qwen3-Coder models support ONLY non-thinking mode (no thinking control
|
|
1137
|
+
// at all) — the previous "configurable thinking" entry was wrong.
|
|
1138
|
+
supportsThinking: false,
|
|
1139
|
+
thinkingBudgetTokens: 0,
|
|
1140
|
+
thinkingConfigurable: false,
|
|
1141
|
+
supportsVision: false,
|
|
1142
|
+
supportsPromptCaching: true,
|
|
1143
|
+
supportsTools: true,
|
|
1144
|
+
// DashScope international list rate, flat across input tiers (the old
|
|
1145
|
+
// $0.65/$3.25 was OpenRouter's promo rate).
|
|
1146
|
+
inputPricePerMTok: 1,
|
|
1147
|
+
outputPricePerMTok: 5,
|
|
1148
|
+
// Implicit context cache: read 0.2× input, no write premium.
|
|
1149
|
+
cacheReadPricePerMTok: 0.2,
|
|
1150
|
+
cacheWritePricePerMTok: 1,
|
|
1151
|
+
// US default (DeepInfra bills well below native). Verified 2026-08-01.
|
|
1152
|
+
regions: ['us', 'cn'],
|
|
1153
|
+
regionPricing: {
|
|
1154
|
+
us: { inputPricePerMTok: 0.3, outputPricePerMTok: 1, cacheReadPricePerMTok: 0.1 },
|
|
1155
|
+
},
|
|
1156
|
+
knowledgeCutoff: '2025-06-01',
|
|
1157
|
+
// Alibaba itself recommends the general-purpose models over Qwen-Coder;
|
|
1158
|
+
// qwen3.7-max is the pick — moved to "Older models".
|
|
1159
|
+
deprecatedAt: '2026-07-28',
|
|
1160
|
+
},
|
|
1161
|
+
// ---------------------------------------------------------------------------
|
|
1162
|
+
// Zhipu (GLM)
|
|
1163
|
+
// Verified: https://docs.z.ai/guides/overview/pricing
|
|
1164
|
+
// https://docs.z.ai/api-reference/llm/chat-completion
|
|
1165
|
+
// glm-5.2 (standalone API since 2026-06-16) is the flagship. It is the ONLY
|
|
1166
|
+
// GLM model with reasoning_effort (values minimal|none|low|medium|high|
|
|
1167
|
+
// xhigh|max; low/medium coerce to high, xhigh coerces to max — effective
|
|
1168
|
+
// levels are high|max plus minimal/none = skip thinking; default max).
|
|
1169
|
+
// glm-5 has thinking on/off only and was REPRICED (was $0.72/$2.30).
|
|
1170
|
+
// ---------------------------------------------------------------------------
|
|
1171
|
+
{
|
|
1172
|
+
id: 'glm-5.2',
|
|
1173
|
+
provider: 'zhipu',
|
|
1174
|
+
label: 'GLM-5.2',
|
|
1175
|
+
description: 'Open-source SOTA agentic — 1M context',
|
|
1176
|
+
contextWindow: 1_048_576,
|
|
1177
|
+
maxOutputTokens: 131_072,
|
|
1178
|
+
supportsThinking: true,
|
|
1179
|
+
thinkingBudgetTokens: 8_000,
|
|
1180
|
+
thinkingConfigurable: true,
|
|
1181
|
+
supportedEffortLevels: ['minimal', 'high', 'max'],
|
|
1182
|
+
defaultEffortLevel: 'high',
|
|
1183
|
+
// Z.ai's default is max (= L, positioned for long-horizon coding); M =
|
|
1184
|
+
// high is the balanced tier; minimal (= S) skips thinking for fast turns.
|
|
1185
|
+
supportsVision: false,
|
|
1186
|
+
supportsPromptCaching: true,
|
|
1187
|
+
supportsTools: true,
|
|
1188
|
+
webSearchToolType: 'web_search',
|
|
1189
|
+
inputPricePerMTok: 1.4,
|
|
1190
|
+
outputPricePerMTok: 4.4,
|
|
1191
|
+
// GLM context cache: read ≈0.19× input, no write premium.
|
|
1192
|
+
cacheReadPricePerMTok: 0.26,
|
|
1193
|
+
cacheWritePricePerMTok: 1.4,
|
|
1194
|
+
// US default (DeepInfra bills ~half native). Verified 2026-08-01.
|
|
1195
|
+
regions: ['us', 'cn'],
|
|
1196
|
+
regionPricing: {
|
|
1197
|
+
us: { inputPricePerMTok: 0.75, outputPricePerMTok: 2.4, cacheReadPricePerMTok: 0.14 },
|
|
1198
|
+
},
|
|
1199
|
+
// Not published by Z.ai — best-effort estimate.
|
|
1200
|
+
knowledgeCutoff: '2025-06-01',
|
|
1201
|
+
},
|
|
1202
|
+
{
|
|
1203
|
+
id: 'glm-5',
|
|
1204
|
+
provider: 'zhipu',
|
|
1205
|
+
label: 'GLM-5',
|
|
1206
|
+
description: '77.8% SWE-bench — open-source agentic, 200K context',
|
|
1207
|
+
contextWindow: 202_752,
|
|
1208
|
+
maxOutputTokens: 131_072,
|
|
1209
|
+
supportsThinking: true,
|
|
1210
|
+
thinkingBudgetTokens: 8_000,
|
|
1211
|
+
// Thinking on/off only (reasoning_effort is glm-5.2-only) — no depth
|
|
1212
|
+
// control; defaults to enabled upstream.
|
|
1213
|
+
thinkingConfigurable: false,
|
|
1214
|
+
supportsVision: false,
|
|
1215
|
+
supportsPromptCaching: true,
|
|
1216
|
+
supportsTools: true,
|
|
1217
|
+
webSearchToolType: 'web_search',
|
|
1218
|
+
// Repriced by Z.ai around the GLM-5.1 launch (was $0.72/$2.30).
|
|
1219
|
+
inputPricePerMTok: 1,
|
|
1220
|
+
outputPricePerMTok: 3.2,
|
|
1221
|
+
// GLM context cache: read 0.2× input, no write premium.
|
|
1222
|
+
cacheReadPricePerMTok: 0.2,
|
|
1223
|
+
cacheWritePricePerMTok: 1,
|
|
1224
|
+
// US default (DeepInfra bills below native). Verified 2026-08-01.
|
|
1225
|
+
regions: ['us', 'cn'],
|
|
1226
|
+
regionPricing: {
|
|
1227
|
+
us: { inputPricePerMTok: 0.6, outputPricePerMTok: 2.08, cacheReadPricePerMTok: 0.12 },
|
|
1228
|
+
},
|
|
1229
|
+
knowledgeCutoff: '2025-01-01',
|
|
1230
|
+
// Superseded by glm-5.2; moved to "Older models".
|
|
1231
|
+
deprecatedAt: '2026-07-28',
|
|
1232
|
+
},
|
|
1233
|
+
];
|