@molecule/api-resource-ai-models 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/models.js ADDED
@@ -0,0 +1,1233 @@
1
+ /**
2
+ * Available AI models.
3
+ *
4
+ * This is the single source of truth for which models Synthase can use.
5
+ * Server-side code (chat handler, compaction) consumes the full definitions
6
+ * directly; clients receive the `PublicModel` projection from `GET /ai/models`.
7
+ *
8
+ * @module
9
+ */
10
+ /**
11
+ * All available AI models, grouped by provider, ordered from most to least capable.
12
+ *
13
+ * To add or remove a model, edit this array. Both the server-side validation
14
+ * and the public discovery endpoint will update automatically.
15
+ *
16
+ * Effort is each model's OWN native value — there is no abstract scale (see
17
+ * {@link ModelDefinition.supportedEffortLevels}):
18
+ * - A model driven by a provider-native effort/level param lists its provider
19
+ * values verbatim in `supportedEffortLevels` (ascending), with
20
+ * `defaultEffortLevel` = the provider's default/recommended level for agentic
21
+ * coding. NO `effortBudgetTokens`.
22
+ * - A model with a controllable token budget but no native level names (e.g.
23
+ * Claude Haiku 4.5's `budget_tokens`, Qwen's `thinking_budget`) lists
24
+ * scaled-budget labels (`['4K', '8K', '16K', '32K']`) with `effortBudgetTokens`
25
+ * mapping each label to the token budget it sends.
26
+ * - A model whose reasoning is fixed (always-on or on/off only, no depth
27
+ * control) carries `thinkingConfigurable: false` and OMITS both fields —
28
+ * there is nothing to tune.
29
+ *
30
+ * Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
31
+ * 2026-07-30 GPT-5.6 repricing — cross-check prices against models.dev with
32
+ * `npm run check:model-freshness` from the workspace root):
33
+ * - Anthropic: https://platform.claude.com/docs/en/about-claude/models/overview
34
+ * + /docs/en/build-with-claude/effort (fable-5 / opus-5 / sonnet-5 current;
35
+ * opus-4-8 superseded by opus-5 at identical pricing but still served — it is
36
+ * the recommended refusal-fallback model; effort ladder on all three current
37
+ * models is low|medium|high|xhigh|max; budget_tokens 400s on 4.7+)
38
+ * - OpenAI: https://developers.openai.com/api/docs/pricing (GPT-5.6 family GA
39
+ * 2026-07-09; REPRICED 2026-07-30: -luna cut 80% to $0.20/$1.20, -terra cut
40
+ * 20% to $2/$12, -sol unchanged $5/$30; cache read 0.1× input; gpt-5.5/
41
+ * gpt-5.4 still listed as current; long-context 2× variants exist upstream —
42
+ * not modeled, same as the Gemini/Grok tiers; Sol "Fast mode" 2.5× speed at
43
+ * 2× price announced 2026-07-30 — not yet modeled)
44
+ * - Google: https://ai.google.dev/gemini-api/docs/pricing (gemini-3.6-flash GA
45
+ * 2026-07-21 $1.50/$7.50 supersedes 3.5-flash as the agentic flagship;
46
+ * gemini-3.1-pro-preview still the pro tier — "3.5 Pro" has NOT shipped as
47
+ * of 2026-07-28 despite the coming-soon badge; do not add until it has an id)
48
+ * - xAI: https://docs.x.ai/developers/models + /developers/grok-4-5
49
+ * (grok-4.5 flagship 2026-07-08: $2/$6, 500K ctx, ≥200K prompts bill 2× —
50
+ * not modeled; reasoning_effort low|medium|high default high, image input;
51
+ * grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
52
+ * grok-code-fast-1 no longer listed — retires 2026-08-15)
53
+ * - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (unchanged V4
54
+ * Pro/Flash pricing; legacy deepseek-chat/-reasoner ids fully retired
55
+ * 2026-07-24 — never in this catalog; the announced peak-hour 2× surcharge is
56
+ * still NOT active as of 2026-07-28, see the entries)
57
+ * - Moonshot: https://platform.kimi.ai/docs/models (kimi-k3 flagship 2026-07-16
58
+ * — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
59
+ * reasoning_content that must be replayed through tool loops, the same
60
+ * constraint that keeps kimi-k2.7-code out; add BOTH once the moonshot bond
61
+ * supports preserved thinking + reasoning_effort low|high|max. kimi-k2.6
62
+ * remains the newest model the bond can run correctly.)
63
+ * - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
64
+ * minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
65
+ * - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
66
+ * (unchanged; qwen3.8-max-preview, 2026-07-19, is Token-Plan-subscriber-only
67
+ * — not on the pay-as-you-go API, so it cannot be added yet; qwen3.7-max
68
+ * currently runs a 50%-off promo — still billed here at list, $2.50/$7.50)
69
+ * - Zhipu: https://docs.z.ai/guides/overview/pricing (unchanged; glm-5.2 is
70
+ * the newest — "GLM-5.3/5.5" rumors have no released ids as of 2026-07-28)
71
+ *
72
+ * Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
73
+ * where the provider doesn't publish one; the provider sources above verify
74
+ * id / pricing / context window.
75
+ */
76
+ export const MODELS = [
77
+ // ---------------------------------------------------------------------------
78
+ // Anthropic
79
+ // Verified: https://platform.claude.com/docs/en/about-claude/models/overview
80
+ // https://platform.claude.com/docs/en/build-with-claude/effort
81
+ // Effort is output_config.effort with adaptive thinking on 4.6+ models —
82
+ // budget_tokens is REJECTED (400) on fable-5 / opus-4-8 / sonnet-5 and
83
+ // deprecated on the 4.6 family. Only Haiku 4.5 still uses budget_tokens.
84
+ // ---------------------------------------------------------------------------
85
+ {
86
+ id: 'claude-fable-5',
87
+ provider: 'anthropic',
88
+ label: 'Claude Fable 5',
89
+ description: 'Most capable Anthropic — frontier reasoning & long-horizon agents',
90
+ contextWindow: 1_000_000,
91
+ maxOutputTokens: 128_000,
92
+ supportsThinking: true,
93
+ thinkingBudgetTokens: 16_000,
94
+ thinkingConfigurable: true,
95
+ supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'],
96
+ defaultEffortLevel: 'high',
97
+ // Thinking is ALWAYS ON (adaptive); effort is the only depth control.
98
+ // Full five-level ladder (medium was missing here — Fable supports low
99
+ // through max). Default/recommended is high; xhigh for the most
100
+ // capability-sensitive work; low still performs well on routine tasks.
101
+ supportsVision: true,
102
+ supportsPromptCaching: true,
103
+ supportsTools: true,
104
+ webSearchToolType: 'web_search_20260209',
105
+ codeExecutionToolType: 'code_execution_20250825',
106
+ webFetchToolType: 'web_fetch_20260209',
107
+ inputPricePerMTok: 10,
108
+ outputPricePerMTok: 50,
109
+ // Anthropic 5-minute prompt cache: read 0.1× input, write 1.25× input.
110
+ cacheReadPricePerMTok: 1,
111
+ cacheWritePricePerMTok: 12.5,
112
+ knowledgeCutoff: '2026-01-01',
113
+ },
114
+ {
115
+ id: 'claude-opus-5',
116
+ provider: 'anthropic',
117
+ label: 'Claude Opus 5',
118
+ description: 'Anthropic Opus flagship — step-change agentic coding at 4.8 pricing',
119
+ // Fast mode (research preview, Claude API only): same model at up to 2.5×
120
+ // output speed, $10/$50 per MTok (2× standard; platform.claude.com fast-mode
121
+ // docs, verified 2026-07-31). Cache rates follow Anthropic's standard
122
+ // ratios (read 0.1× input, write 1.25× input) applied to the fast input
123
+ // rate. Separate upstream rate limits from standard Opus.
124
+ fastPricing: {
125
+ inputPricePerMTok: 10,
126
+ outputPricePerMTok: 50,
127
+ cacheReadPricePerMTok: 1,
128
+ cacheWritePricePerMTok: 12.5,
129
+ },
130
+ contextWindow: 1_000_000,
131
+ maxOutputTokens: 128_000,
132
+ supportsThinking: true,
133
+ thinkingBudgetTokens: 16_000,
134
+ thinkingConfigurable: true,
135
+ supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'],
136
+ defaultEffortLevel: 'high',
137
+ // Thinking on by default (adaptive); disabling is only valid at effort
138
+ // high or below (xhigh/max + disabled → 400). Full five-level ladder;
139
+ // Anthropic's rec: xhigh for coding/agentic, but low/medium punch far
140
+ // above their weight on this model. 512-token prompt-cache minimum.
141
+ supportsVision: true,
142
+ supportsPromptCaching: true,
143
+ supportsTools: true,
144
+ webSearchToolType: 'web_search_20260209',
145
+ codeExecutionToolType: 'code_execution_20250825',
146
+ webFetchToolType: 'web_fetch_20260209',
147
+ // Drop-in successor to Opus 4.8 at identical pricing.
148
+ inputPricePerMTok: 5,
149
+ outputPricePerMTok: 25,
150
+ // Anthropic 5-minute prompt cache: read 0.1× input, write 1.25× input.
151
+ cacheReadPricePerMTok: 0.5,
152
+ cacheWritePricePerMTok: 6.25,
153
+ // Not published at verification time — best-effort estimate (≥ Opus 4.8's).
154
+ knowledgeCutoff: '2026-01-01',
155
+ },
156
+ {
157
+ id: 'claude-opus-4-8',
158
+ provider: 'anthropic',
159
+ label: 'Claude Opus 4.8',
160
+ description: 'Previous Opus — deep reasoning; the opus-5 refusal fallback',
161
+ contextWindow: 1_000_000,
162
+ maxOutputTokens: 128_000,
163
+ supportsThinking: true,
164
+ thinkingBudgetTokens: 16_000,
165
+ thinkingConfigurable: true,
166
+ supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'],
167
+ defaultEffortLevel: 'high',
168
+ // Full five-level ladder (medium was missing here). Anthropic's rec for
169
+ // coding/agentic on Opus 4.8 is xhigh; the API default is high.
170
+ supportsVision: true,
171
+ supportsPromptCaching: true,
172
+ supportsTools: true,
173
+ webSearchToolType: 'web_search_20260209',
174
+ codeExecutionToolType: 'code_execution_20250825',
175
+ webFetchToolType: 'web_fetch_20260209',
176
+ inputPricePerMTok: 5,
177
+ outputPricePerMTok: 25,
178
+ // Anthropic 5-minute prompt cache: read 0.1× input, write 1.25× input.
179
+ cacheReadPricePerMTok: 0.5,
180
+ cacheWritePricePerMTok: 6.25,
181
+ knowledgeCutoff: '2026-01-01',
182
+ // Superseded by claude-opus-5 (same price); still served upstream and the
183
+ // recommended refusal-fallback target. Selectable under "Older models".
184
+ deprecatedAt: '2026-07-28',
185
+ },
186
+ {
187
+ id: 'claude-sonnet-5',
188
+ provider: 'anthropic',
189
+ label: 'Claude Sonnet 5',
190
+ description: 'Fast & capable — near-Opus coding at Sonnet cost',
191
+ contextWindow: 1_000_000,
192
+ maxOutputTokens: 128_000,
193
+ supportsThinking: true,
194
+ thinkingBudgetTokens: 16_000,
195
+ thinkingConfigurable: true,
196
+ supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'],
197
+ defaultEffortLevel: 'high',
198
+ // Adaptive thinking on by default; full five-level ladder (medium was
199
+ // missing here). Default/rec is high, xhigh for the hardest coding/agentic
200
+ // tasks, medium ≈ Sonnet 4.6 at high.
201
+ supportsVision: true,
202
+ supportsPromptCaching: true,
203
+ supportsTools: true,
204
+ webSearchToolType: 'web_search_20260209',
205
+ codeExecutionToolType: 'code_execution_20250825',
206
+ webFetchToolType: 'web_fetch_20260209',
207
+ // Standard pricing. Intro pricing ($2/$10) applies through 2026-08-31 —
208
+ // billed here at standard so metering never under-charges; revisit after.
209
+ inputPricePerMTok: 3,
210
+ outputPricePerMTok: 15,
211
+ // Anthropic 5-minute prompt cache: read 0.1× input, write 1.25× input.
212
+ cacheReadPricePerMTok: 0.3,
213
+ cacheWritePricePerMTok: 3.75,
214
+ knowledgeCutoff: '2026-01-01',
215
+ },
216
+ {
217
+ id: 'claude-opus-4-6',
218
+ provider: 'anthropic',
219
+ label: 'Claude Opus 4.6',
220
+ description: 'Previous-generation Opus — deep reasoning & complex tasks',
221
+ contextWindow: 1_000_000,
222
+ maxOutputTokens: 128_000,
223
+ supportsThinking: true,
224
+ thinkingBudgetTokens: 16_000,
225
+ thinkingConfigurable: true,
226
+ supportedEffortLevels: ['low', 'medium', 'high', 'max'],
227
+ defaultEffortLevel: 'medium',
228
+ // No xhigh on the 4.6 family; budget_tokens still accepted but deprecated —
229
+ // adaptive + effort is the recommended control.
230
+ supportsVision: true,
231
+ supportsPromptCaching: true,
232
+ supportsTools: true,
233
+ webSearchToolType: 'web_search_20260209',
234
+ codeExecutionToolType: 'code_execution_20250825',
235
+ webFetchToolType: 'web_fetch_20260209',
236
+ inputPricePerMTok: 5,
237
+ outputPricePerMTok: 25,
238
+ // Anthropic 5-minute prompt cache: read 0.1× input, write 1.25× input.
239
+ cacheReadPricePerMTok: 0.5,
240
+ cacheWritePricePerMTok: 6.25,
241
+ knowledgeCutoff: '2025-05-01',
242
+ // Superseded by claude-opus-4-8; kept selectable (Older models) + priceable.
243
+ deprecatedAt: '2026-06-16',
244
+ },
245
+ {
246
+ id: 'claude-sonnet-4-6',
247
+ provider: 'anthropic',
248
+ label: 'Claude Sonnet 4.6',
249
+ description: 'Previous-generation Sonnet — fast & balanced',
250
+ contextWindow: 1_000_000,
251
+ // Was wrongly 64K — Sonnet 4.6 supports 128K output (models overview).
252
+ maxOutputTokens: 128_000,
253
+ supportsThinking: true,
254
+ thinkingBudgetTokens: 10_000,
255
+ thinkingConfigurable: true,
256
+ supportedEffortLevels: ['low', 'medium', 'high', 'max'],
257
+ defaultEffortLevel: 'medium',
258
+ // No xhigh on the 4.6 family. Anthropic's recommended default for agentic
259
+ // coding on Sonnet 4.6 is medium (= M); high is its API default (= L).
260
+ supportsVision: true,
261
+ supportsPromptCaching: true,
262
+ supportsTools: true,
263
+ webSearchToolType: 'web_search_20260209',
264
+ codeExecutionToolType: 'code_execution_20250825',
265
+ webFetchToolType: 'web_fetch_20260209',
266
+ inputPricePerMTok: 3,
267
+ outputPricePerMTok: 15,
268
+ // Anthropic 5-minute prompt cache: read 0.1× input, write 1.25× input.
269
+ cacheReadPricePerMTok: 0.3,
270
+ cacheWritePricePerMTok: 3.75,
271
+ knowledgeCutoff: '2025-08-01',
272
+ // Superseded by claude-sonnet-5; kept selectable (Older models) + priceable.
273
+ deprecatedAt: '2026-07-07',
274
+ },
275
+ {
276
+ id: 'claude-haiku-4-5-20251001',
277
+ provider: 'anthropic',
278
+ label: 'Claude Haiku 4.5',
279
+ description: 'Fastest Anthropic — quick tasks & iteration',
280
+ contextWindow: 200_000,
281
+ maxOutputTokens: 64_000,
282
+ supportsThinking: true,
283
+ thinkingBudgetTokens: 8_000,
284
+ thinkingConfigurable: true,
285
+ // Haiku 4.5 does NOT support output_config.effort (400) or adaptive
286
+ // thinking — it stays on the budget_tokens path, so its levels are
287
+ // budget labels (effortBudgetTokens), not a native effort param.
288
+ supportedEffortLevels: ['4K', '8K', '16K', '32K'],
289
+ defaultEffortLevel: '8K',
290
+ effortBudgetTokens: { '4K': 4000, '8K': 8000, '16K': 16000, '32K': 32000 },
291
+ supportsVision: true,
292
+ supportsPromptCaching: true,
293
+ supportsTools: true,
294
+ webSearchToolType: 'web_search_20250305',
295
+ webFetchToolType: 'web_fetch_20250910',
296
+ inputPricePerMTok: 1,
297
+ outputPricePerMTok: 5,
298
+ // Anthropic 5-minute prompt cache: read 0.1× input, write 1.25× input.
299
+ cacheReadPricePerMTok: 0.1,
300
+ cacheWritePricePerMTok: 1.25,
301
+ knowledgeCutoff: '2025-02-01',
302
+ },
303
+ // ---------------------------------------------------------------------------
304
+ // OpenAI
305
+ // Verified: https://developers.openai.com/api/docs/pricing (2026-07-28)
306
+ // GPT-5.6 family GA 2026-07-09: gpt-5.6 is an alias for gpt-5.6-sol; Sol is
307
+ // the frontier tier, Terra the balanced tier, Luna the cheap/fast tier.
308
+ // Cached input is billed at 0.1× input. The official pricing page shows no
309
+ // cache-WRITE premium, but third-party trackers report 1.25× with a 30-min
310
+ // cache life for the 5.6 family — billed here at 1.25× (conservative;
311
+ // re-verify against the official docs). reasoning_effort values for 5.6 are
312
+ // not yet on the docs page — carried over from gpt-5.5 (low|medium|high|
313
+ // xhigh, default medium); re-verify. Long-context 2× price variants exist
314
+ // upstream — not modeled (same as the Gemini/Grok >200K tiers).
315
+ // ---------------------------------------------------------------------------
316
+ {
317
+ id: 'gpt-5.6-sol',
318
+ provider: 'openai',
319
+ label: 'GPT-5.6 Sol',
320
+ description: 'OpenAI frontier — the hardest coding & reasoning work',
321
+ // Reported as 1.05M; floored to 1M until the docs state it (understating
322
+ // only makes compaction slightly earlier — overstating risks overflow).
323
+ contextWindow: 1_000_000,
324
+ maxOutputTokens: 128_000,
325
+ supportsThinking: true,
326
+ thinkingBudgetTokens: 16_000,
327
+ thinkingConfigurable: true,
328
+ supportedEffortLevels: ['low', 'medium', 'high', 'xhigh'],
329
+ defaultEffortLevel: 'medium',
330
+ supportsVision: true,
331
+ supportsPromptCaching: true,
332
+ supportsTools: true,
333
+ // Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
334
+ toolsRequireReasoningOff: true,
335
+ webSearchToolType: 'web_search',
336
+ codeExecutionToolType: 'code_interpreter',
337
+ inputPricePerMTok: 5,
338
+ outputPricePerMTok: 30,
339
+ // Cached input 0.1× input; write premium reported 1.25× (see section note).
340
+ cacheReadPricePerMTok: 0.5,
341
+ cacheWritePricePerMTok: 6.25,
342
+ // Not published — best-effort estimate.
343
+ knowledgeCutoff: '2026-03-01',
344
+ },
345
+ {
346
+ id: 'gpt-5.6-terra',
347
+ provider: 'openai',
348
+ label: 'GPT-5.6 Terra',
349
+ description: 'Balanced OpenAI — everyday coding & tool use',
350
+ contextWindow: 1_000_000,
351
+ maxOutputTokens: 128_000,
352
+ supportsThinking: true,
353
+ thinkingBudgetTokens: 16_000,
354
+ thinkingConfigurable: true,
355
+ supportedEffortLevels: ['low', 'medium', 'high', 'xhigh'],
356
+ defaultEffortLevel: 'medium',
357
+ supportsVision: true,
358
+ supportsPromptCaching: true,
359
+ supportsTools: true,
360
+ // Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
361
+ toolsRequireReasoningOff: true,
362
+ webSearchToolType: 'web_search',
363
+ codeExecutionToolType: 'code_interpreter',
364
+ // Repriced 2026-07-30 (20% cut from $2.50/$15).
365
+ inputPricePerMTok: 2,
366
+ outputPricePerMTok: 12,
367
+ // Cached input 0.1× input; write premium reported 1.25× (see section note).
368
+ cacheReadPricePerMTok: 0.2,
369
+ cacheWritePricePerMTok: 2.5,
370
+ // Not published — best-effort estimate.
371
+ knowledgeCutoff: '2026-03-01',
372
+ },
373
+ {
374
+ id: 'gpt-5.6-luna',
375
+ provider: 'openai',
376
+ label: 'GPT-5.6 Luna',
377
+ description: 'Fast & cheap OpenAI — light tasks & subagents',
378
+ contextWindow: 1_000_000,
379
+ maxOutputTokens: 128_000,
380
+ supportsThinking: true,
381
+ thinkingBudgetTokens: 8_000,
382
+ thinkingConfigurable: true,
383
+ supportedEffortLevels: ['low', 'medium', 'high', 'xhigh'],
384
+ defaultEffortLevel: 'medium',
385
+ supportsVision: true,
386
+ supportsPromptCaching: true,
387
+ supportsTools: true,
388
+ // Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
389
+ toolsRequireReasoningOff: true,
390
+ webSearchToolType: 'web_search',
391
+ codeExecutionToolType: 'code_interpreter',
392
+ // Repriced 2026-07-30 (80% cut from $1/$6).
393
+ inputPricePerMTok: 0.2,
394
+ outputPricePerMTok: 1.2,
395
+ // Cached input 0.1× input; write premium reported 1.25× (see section note).
396
+ cacheReadPricePerMTok: 0.02,
397
+ cacheWritePricePerMTok: 0.25,
398
+ // Not published — best-effort estimate.
399
+ knowledgeCutoff: '2026-03-01',
400
+ },
401
+ {
402
+ id: 'gpt-5.5',
403
+ provider: 'openai',
404
+ label: 'GPT-5.5',
405
+ description: 'OpenAI flagship — strong coding & tool use',
406
+ contextWindow: 1_050_000,
407
+ maxOutputTokens: 128_000,
408
+ supportsThinking: true,
409
+ thinkingBudgetTokens: 16_000,
410
+ thinkingConfigurable: true,
411
+ supportedEffortLevels: ['low', 'medium', 'high', 'xhigh'],
412
+ defaultEffortLevel: 'medium',
413
+ // OpenAI default + agentic-coding rec is medium (= M); 'none' is not
414
+ // offered (it disables reasoning outright).
415
+ supportsVision: true,
416
+ supportsPromptCaching: true,
417
+ supportsTools: true,
418
+ // Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
419
+ toolsRequireReasoningOff: true,
420
+ webSearchToolType: 'web_search',
421
+ codeExecutionToolType: 'code_interpreter',
422
+ inputPricePerMTok: 5,
423
+ outputPricePerMTok: 30,
424
+ // OpenAI auto-caches at no write premium; cached input billed at 0.1× input.
425
+ cacheReadPricePerMTok: 0.5,
426
+ cacheWritePricePerMTok: 5,
427
+ knowledgeCutoff: '2025-12-01',
428
+ // Superseded by gpt-5.6-sol (same price); still listed as current by
429
+ // OpenAI. Selectable under "Older models".
430
+ deprecatedAt: '2026-07-09',
431
+ },
432
+ {
433
+ id: 'gpt-5.4',
434
+ provider: 'openai',
435
+ label: 'GPT-5.4',
436
+ description: 'Affordable OpenAI frontier — strong coding & tool use',
437
+ contextWindow: 1_050_000,
438
+ maxOutputTokens: 128_000,
439
+ supportsThinking: true,
440
+ thinkingBudgetTokens: 16_000,
441
+ thinkingConfigurable: true,
442
+ supportedEffortLevels: ['low', 'medium', 'high', 'xhigh'],
443
+ defaultEffortLevel: 'medium',
444
+ supportsVision: true,
445
+ supportsPromptCaching: true,
446
+ supportsTools: true,
447
+ // Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
448
+ toolsRequireReasoningOff: true,
449
+ webSearchToolType: 'web_search',
450
+ codeExecutionToolType: 'code_interpreter',
451
+ inputPricePerMTok: 2.5,
452
+ outputPricePerMTok: 15,
453
+ // OpenAI auto-caches at no write premium; cached input billed at 0.1× input.
454
+ cacheReadPricePerMTok: 0.25,
455
+ cacheWritePricePerMTok: 2.5,
456
+ knowledgeCutoff: '2025-08-31',
457
+ // OpenAI still lists gpt-5.4 as current, but gpt-5.6-terra covers this
458
+ // tier at the same price — moved to "Older models" (deprecatedAt is OUR
459
+ // picker taxonomy, not OpenAI's deprecations page).
460
+ deprecatedAt: '2026-07-28',
461
+ },
462
+ {
463
+ id: 'gpt-5.4-mini',
464
+ provider: 'openai',
465
+ label: 'GPT-5.4 mini',
466
+ description: 'Cheap & fast OpenAI — light coding tasks & subagents',
467
+ contextWindow: 400_000,
468
+ maxOutputTokens: 128_000,
469
+ supportsThinking: true,
470
+ thinkingBudgetTokens: 8_000,
471
+ thinkingConfigurable: true,
472
+ supportedEffortLevels: ['low', 'medium', 'high', 'xhigh'],
473
+ defaultEffortLevel: 'medium',
474
+ supportsVision: true,
475
+ supportsPromptCaching: true,
476
+ supportsTools: true,
477
+ // Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
478
+ toolsRequireReasoningOff: true,
479
+ webSearchToolType: 'web_search',
480
+ codeExecutionToolType: 'code_interpreter',
481
+ inputPricePerMTok: 0.75,
482
+ outputPricePerMTok: 4.5,
483
+ // OpenAI auto-caches at no write premium; cached input billed at 0.1× input.
484
+ cacheReadPricePerMTok: 0.075,
485
+ cacheWritePricePerMTok: 0.75,
486
+ knowledgeCutoff: '2025-08-31',
487
+ // OpenAI still lists gpt-5.4-mini as current, but like gpt-5.4 above it's
488
+ // superseded in our lineup (cheap/fast tier is better served by the newer
489
+ // models) — moved to "Older models" (deprecatedAt is OUR picker taxonomy,
490
+ // not OpenAI's deprecations page).
491
+ deprecatedAt: '2026-08-01',
492
+ },
493
+ // ---------------------------------------------------------------------------
494
+ // Google
495
+ // Verified: https://ai.google.dev/gemini-api/docs/models
496
+ // https://ai.google.dev/gemini-api/docs/pricing
497
+ // https://ai.google.dev/gemini-api/docs/thinking
498
+ // Reasoning control is `thinking_level` (minimal|low|medium|high — cannot be
499
+ // fully off on 3.x; combining with legacy thinking_budget returns 400).
500
+ // NOTE: the previous catalog carried a fictional "gemini-3.1-pro" GA id — no
501
+ // such model exists; the pro tier is (still) gemini-3.1-pro-preview. Safe to
502
+ // replace outright: the google bond has never been implemented/wired, so no
503
+ // historical usage can reference the old ids.
504
+ // ---------------------------------------------------------------------------
505
+ {
506
+ id: 'gemini-3.6-flash',
507
+ provider: 'google',
508
+ label: 'Gemini 3.6 Flash',
509
+ description: 'Google agentic flagship — frontier intelligence + grounding',
510
+ // Window/output not on the pricing page — carried over from 3.5-flash;
511
+ // re-verify against /docs/models.
512
+ contextWindow: 1_048_576,
513
+ maxOutputTokens: 65_536,
514
+ supportsThinking: true,
515
+ thinkingBudgetTokens: 10_000,
516
+ thinkingConfigurable: true,
517
+ // thinking_level assumed unchanged from 3.5-flash (low|medium|high);
518
+ // re-verify against /docs/thinking.
519
+ supportedEffortLevels: ['low', 'medium', 'high'],
520
+ defaultEffortLevel: 'medium',
521
+ supportsVision: true,
522
+ supportsPromptCaching: true,
523
+ supportsTools: true,
524
+ webSearchToolType: 'google_search',
525
+ codeExecutionToolType: 'code_execution',
526
+ webFetchToolType: 'url_context',
527
+ // GA 2026-07-21 — same input price as 3.5-flash, CHEAPER output ($7.50 vs $9).
528
+ inputPricePerMTok: 1.5,
529
+ outputPricePerMTok: 7.5,
530
+ // Gemini context cache: read $0.15/M (0.1× input), no write premium
531
+ // (storage billed separately per hour — not modeled).
532
+ cacheReadPricePerMTok: 0.15,
533
+ cacheWritePricePerMTok: 1.5,
534
+ // Not published — best-effort estimate.
535
+ knowledgeCutoff: '2026-01-01',
536
+ },
537
+ {
538
+ id: 'gemini-3.5-flash',
539
+ provider: 'google',
540
+ label: 'Gemini 3.5 Flash',
541
+ description: 'Previous Google agentic flagship — fast, 1M context',
542
+ contextWindow: 1_048_576,
543
+ maxOutputTokens: 65_536,
544
+ supportsThinking: true,
545
+ thinkingBudgetTokens: 10_000,
546
+ thinkingConfigurable: true,
547
+ supportedEffortLevels: ['low', 'medium', 'high'],
548
+ defaultEffortLevel: 'medium',
549
+ // thinking_level: default medium (= M); Google recommends high (= L) for
550
+ // advanced coding/multi-step planning. 'minimal' exists below low; no
551
+ // fourth upward tier, so XL is not offered.
552
+ supportsVision: true,
553
+ supportsPromptCaching: true,
554
+ supportsTools: true,
555
+ webSearchToolType: 'google_search',
556
+ codeExecutionToolType: 'code_execution',
557
+ webFetchToolType: 'url_context',
558
+ inputPricePerMTok: 1.5,
559
+ outputPricePerMTok: 9,
560
+ // Gemini context cache: read $0.15/M (0.1× input), no write premium
561
+ // (storage billed separately per hour — not modeled).
562
+ cacheReadPricePerMTok: 0.15,
563
+ cacheWritePricePerMTok: 1.5,
564
+ knowledgeCutoff: '2025-01-01',
565
+ // Superseded by gemini-3.6-flash (2026-07-21); still served upstream.
566
+ // Selectable under "Older models".
567
+ deprecatedAt: '2026-07-21',
568
+ },
569
+ {
570
+ id: 'gemini-3.1-pro-preview',
571
+ provider: 'google',
572
+ label: 'Gemini 3.1 Pro',
573
+ description: 'Google pro tier — deep reasoning (preview id; no GA id exists)',
574
+ contextWindow: 1_048_576,
575
+ maxOutputTokens: 65_536,
576
+ supportsThinking: true,
577
+ thinkingBudgetTokens: 10_000,
578
+ thinkingConfigurable: true,
579
+ supportedEffortLevels: ['low', 'medium', 'high'],
580
+ defaultEffortLevel: 'medium',
581
+ // thinking_level: low|medium|high only (no minimal). Google's API default
582
+ // for this model is high (= L); M offers the cost step-down.
583
+ supportsVision: true,
584
+ supportsPromptCaching: true,
585
+ supportsTools: true,
586
+ webSearchToolType: 'google_search',
587
+ codeExecutionToolType: 'code_execution',
588
+ webFetchToolType: 'url_context',
589
+ // ≤200K-token prompts; >200K bills $4/$18 (tiering not modeled).
590
+ inputPricePerMTok: 2,
591
+ outputPricePerMTok: 12,
592
+ // Gemini context cache: read $0.20/M (≤200K), no write premium (hourly
593
+ // storage not modeled).
594
+ cacheReadPricePerMTok: 0.2,
595
+ cacheWritePricePerMTok: 2,
596
+ knowledgeCutoff: '2025-01-01',
597
+ },
598
+ // ---------------------------------------------------------------------------
599
+ // xAI (Grok)
600
+ // Verified: https://docs.x.ai/developers/models + /developers/grok-4-5
601
+ // (2026-07-28)
602
+ // grok-4.5 (2026-07-08) is the flagship: 500K ctx, reasoning_effort
603
+ // low|medium|high default high, image input. grok-4.3 stays served as the
604
+ // value tier with the BIGGER 1M window (reasoning_effort none|low|medium|
605
+ // high, default low). Max output tokens still not documented for any model.
606
+ // ---------------------------------------------------------------------------
607
+ {
608
+ id: 'grok-4.5',
609
+ provider: 'xai',
610
+ label: 'Grok 4.5',
611
+ description: 'xAI frontier — coding, agentic tasks & knowledge work',
612
+ contextWindow: 500_000,
613
+ // Max output not documented by xAI — conservative cap.
614
+ maxOutputTokens: 128_000,
615
+ supportsThinking: true,
616
+ thinkingBudgetTokens: 16_000,
617
+ thinkingConfigurable: true,
618
+ // reasoning_effort low|medium|high, default high (docs.x.ai/developers/
619
+ // grok-4-5) — no 'none' tier on 4.5, unlike 4.3.
620
+ supportedEffortLevels: ['low', 'medium', 'high'],
621
+ defaultEffortLevel: 'high',
622
+ // Multimodal input (text + images), text out.
623
+ supportsVision: true,
624
+ supportsPromptCaching: true,
625
+ supportsTools: true,
626
+ // ≤200K-token prompts; ≥200K bills 2× ($4/$0.60/$12 — tiering not
627
+ // modeled, same as the Gemini 3.1 Pro >200K tier).
628
+ inputPricePerMTok: 2,
629
+ outputPricePerMTok: 6,
630
+ // xAI cached input billed at a flat $0.30/M, no write premium.
631
+ cacheReadPricePerMTok: 0.3,
632
+ cacheWritePricePerMTok: 2,
633
+ // Official (docs.x.ai): 2026-02-01.
634
+ knowledgeCutoff: '2026-02-01',
635
+ },
636
+ {
637
+ id: 'grok-4.3',
638
+ provider: 'xai',
639
+ label: 'Grok 4.3',
640
+ description: 'xAI value tier — fast reasoning, bigger 1M context',
641
+ contextWindow: 1_000_000,
642
+ maxOutputTokens: 128_000,
643
+ supportsThinking: true,
644
+ thinkingBudgetTokens: 16_000,
645
+ thinkingConfigurable: true,
646
+ supportedEffortLevels: ['none', 'low', 'medium', 'high'],
647
+ defaultEffortLevel: 'low',
648
+ // xAI's default (and its own retirement-routing choice for agentic
649
+ // workloads) is low (= M); none (= S) disables reasoning for a true fast
650
+ // mode; step up for hard debugging/architecture turns.
651
+ supportsVision: true,
652
+ supportsPromptCaching: true,
653
+ supportsTools: true,
654
+ inputPricePerMTok: 1.25,
655
+ outputPricePerMTok: 2.5,
656
+ // xAI cached input billed at a flat $0.20/M, no write premium.
657
+ cacheReadPricePerMTok: 0.2,
658
+ cacheWritePricePerMTok: 1.25,
659
+ knowledgeCutoff: '2025-12-01',
660
+ // Superseded by grok-4.5 as the xAI pick (4.3 keeps the bigger 1M window
661
+ // — the reason it stays selectable under "Older models").
662
+ deprecatedAt: '2026-07-28',
663
+ },
664
+ {
665
+ id: 'grok-build-0.1',
666
+ provider: 'xai',
667
+ label: 'Grok Build',
668
+ description: 'Agentic coding specialist — fast & cheap (public beta)',
669
+ contextWindow: 256_000,
670
+ // Max output not documented by xAI — conservative cap.
671
+ maxOutputTokens: 64_000,
672
+ supportsThinking: true,
673
+ thinkingBudgetTokens: 8_000,
674
+ // Reasoning is always on and NOT configurable (reasoning_effort is not
675
+ // honored on grok-build) — successor to grok-code-fast-1 (that slug now
676
+ // auto-routes here).
677
+ thinkingConfigurable: false,
678
+ supportsVision: true,
679
+ supportsPromptCaching: true,
680
+ supportsTools: true,
681
+ inputPricePerMTok: 1,
682
+ outputPricePerMTok: 2,
683
+ // xAI cached input billed at a flat $0.20/M, no write premium.
684
+ cacheReadPricePerMTok: 0.2,
685
+ cacheWritePricePerMTok: 1,
686
+ // Not published by xAI — best-effort estimate (grok-4-generation base).
687
+ knowledgeCutoff: '2025-06-01',
688
+ // Niche coding beta; grok-4.5 is the xAI pick — kept out of the main list.
689
+ deprecatedAt: '2026-07-28',
690
+ },
691
+ {
692
+ id: 'grok-4.20-multi-agent-beta-0309',
693
+ provider: 'xai',
694
+ label: 'Grok 4.20',
695
+ description: 'Older xAI flagship — multi-agent',
696
+ // Current xAI docs list 1M (the old 2M figure is stale). This beta slug is
697
+ // an alias of the canonical grok-4.20-multi-agent-0309 — kept under the
698
+ // original id so saved selections + historical usage stay priceable.
699
+ contextWindow: 1_000_000,
700
+ maxOutputTokens: 128_000,
701
+ supportsThinking: true,
702
+ thinkingBudgetTokens: 16_000,
703
+ // On the multi-agent model reasoning_effort controls AGENT COUNT, not
704
+ // reasoning depth — do not drive it from the effort setting.
705
+ thinkingConfigurable: false,
706
+ supportsVision: true,
707
+ supportsPromptCaching: true,
708
+ supportsTools: true,
709
+ // Repriced by xAI when grok-4.3 launched (was $2/$6).
710
+ inputPricePerMTok: 1.25,
711
+ outputPricePerMTok: 2.5,
712
+ // xAI cached input billed at a flat $0.20/M, no write premium.
713
+ cacheReadPricePerMTok: 0.2,
714
+ cacheWritePricePerMTok: 1.25,
715
+ knowledgeCutoff: '2024-11-01',
716
+ deprecatedAt: '2026-04-30',
717
+ // DISABLED, not deleted: xAI rejects this model on the chat-completions
718
+ // endpoint outright — "Multi Agent requests are not allowed on chat
719
+ // completions" (400 on every call, verified live 2026-07-30) — so offering
720
+ // it in the picker only hands users a model that cannot answer. Their API
721
+ // also lists the canonical slug as `grok-4.20-multi-agent-0309` (no
722
+ // `beta-`), with `grok-4.20-0309-reasoning` / `-non-reasoning` as the
723
+ // chat-completions-capable variants if this family is wanted back.
724
+ // Stays in the catalogue so saved selections and past usage remain priceable.
725
+ disabled: true,
726
+ },
727
+ {
728
+ id: 'grok-code-fast-1',
729
+ provider: 'xai',
730
+ label: 'Grok Code',
731
+ description: 'Code specialist — fast & cheap',
732
+ contextWindow: 256_000,
733
+ maxOutputTokens: 64_000,
734
+ supportsThinking: true,
735
+ thinkingBudgetTokens: 8_000,
736
+ thinkingConfigurable: false,
737
+ supportsVision: false,
738
+ supportsPromptCaching: true,
739
+ supportsTools: true,
740
+ inputPricePerMTok: 0.2,
741
+ outputPricePerMTok: 1.5,
742
+ // xAI cached input billed at a discount (≈0.25× input), no write premium.
743
+ cacheReadPricePerMTok: 0.05,
744
+ cacheWritePricePerMTok: 0.2,
745
+ knowledgeCutoff: '2024-11-01',
746
+ // Deprecated by xAI 2026-05-15 (retires 2026-08-15; the slug auto-routes to
747
+ // grok-build-0.1 until then) — disabled: removed from selection + the
748
+ // public listing, but getModel() still prices any historical usage. NEVER
749
+ // delete.
750
+ disabled: true,
751
+ },
752
+ // ---------------------------------------------------------------------------
753
+ // DeepSeek
754
+ // Verified: https://api-docs.deepseek.com/quick_start/pricing
755
+ // https://api-docs.deepseek.com/guides/thinking_mode
756
+ // https://api-docs.deepseek.com/updates/ (2026-07-31)
757
+ // 2026-07-31: DeepSeek-V4-Flash OFFICIAL API launched in public beta — the
758
+ // SAME `deepseek-v4-flash` id now serves the re-post-trained 0731 build
759
+ // (same architecture/size; much stronger agent benchmarks — beats
760
+ // V4-Pro-Preview on Terminal Bench 2.1 / DeepSWE). No pricing/limit/
761
+ // capability changes. V4-Pro official release "coming soon" — re-verify
762
+ // pricing THEN (the announced peak-hour 2× was tied to the V4 official
763
+ // rollout and is still not on the rate card).
764
+ // OpenAI/Anthropic-compatible API; text/code only (no vision); 1M context,
765
+ // 384K max output, automatic context (prompt) caching with ABSOLUTE cache-hit
766
+ // prices (~1/50–1/120 of miss — not the old 0.1× rule). Launch discount made
767
+ // PERMANENT 2026-05-23 (Pro $1.74/$3.48 → $0.435/$0.87). Peak-hour 2×
768
+ // pricing announced for the mid-Jul 2026 "V4 official" release — re-verify
769
+ // then. Thinking now defaults ENABLED upstream and supports tool calling
770
+ // (reasoning_effort: high|max), BUT tool loops in thinking mode must replay
771
+ // assistant reasoning_content on every subsequent request (400 on omission).
772
+ // The bond explicitly sends thinking:{type:"disabled"} — Synthase runs
773
+ // DeepSeek as a non-thinking executor (Sonnet plans; DeepSeek executes).
774
+ // ---------------------------------------------------------------------------
775
+ {
776
+ id: 'deepseek-v4-pro',
777
+ provider: 'deepseek',
778
+ label: 'DeepSeek V4 Pro',
779
+ description: 'Frontier-class — rivals top models at low cost',
780
+ contextWindow: 1_000_000,
781
+ maxOutputTokens: 384_000,
782
+ // Run non-thinking (see section note): the executor tool loop would have to
783
+ // replay reasoning_content across every turn in thinking mode.
784
+ supportsThinking: false,
785
+ thinkingBudgetTokens: 0,
786
+ thinkingConfigurable: false,
787
+ supportsVision: false,
788
+ supportsPromptCaching: true,
789
+ supportsTools: true,
790
+ inputPricePerMTok: 0.435,
791
+ outputPricePerMTok: 0.87,
792
+ // DeepSeek automatic context cache: absolute cache-hit price ($/M).
793
+ cacheReadPricePerMTok: 0.003625,
794
+ cacheWritePricePerMTok: 0.435,
795
+ // Native-China DEFAULT (owner decision 2026-08-01): the US re-host
796
+ // (DeepInfra) bills ~3× list and ~28× cache reads, and agentic input is
797
+ // ~94% cache hits, so US processing ran ~5.7× native on real traffic.
798
+ // Users opt into US per model via the picker's region control.
799
+ regions: ['cn', 'us'],
800
+ // The free tier PLANS with this model on the cheap native host (it is the
801
+ // molecule-dev FREE_TIER_MODELS.plan), so CN is free-tier selectable; the
802
+ // ~3× US re-host stays paid-only (free users switch to Flash for US).
803
+ freeTierRegions: ['cn'],
804
+ // US = DeepInfra, verified 2026-08-01 via api.deepinfra.com/models/
805
+ // deepseek-ai/DeepSeek-V4-Pro. No cache-write premium (omitted → region
806
+ // input rate).
807
+ regionPricing: {
808
+ us: { inputPricePerMTok: 1.3, outputPricePerMTok: 2.6, cacheReadPricePerMTok: 0.1 },
809
+ },
810
+ // The announced peak-hour 2× surcharge (Beijing business hours) is STILL
811
+ // NOT ACTIVE as of 2026-07-28 — the official rate card lists a single flat
812
+ // rate, and no switch-over date is published. The pre-wired windows were
813
+ // REMOVED: they had been over-billing every peak-window turn 2× for weeks
814
+ // (this is the free-tier default model, so that directly shrank free
815
+ // users' allowances). Re-add via `peakPricing` the day DeepSeek's rate
816
+ // card actually shows the surcharge.
817
+ // Not published by DeepSeek — best-effort estimate.
818
+ knowledgeCutoff: '2025-07-01',
819
+ },
820
+ {
821
+ id: 'deepseek-v4-flash',
822
+ provider: 'deepseek',
823
+ label: 'DeepSeek V4 Flash',
824
+ description: 'Ultra-cheap & fast — economical agentic coding',
825
+ contextWindow: 1_000_000,
826
+ maxOutputTokens: 384_000,
827
+ // Non-thinking (see section note) — also the right tradeoff for a cheap,
828
+ // fast tool-calling executor.
829
+ supportsThinking: false,
830
+ thinkingBudgetTokens: 0,
831
+ thinkingConfigurable: false,
832
+ supportsVision: false,
833
+ supportsPromptCaching: true,
834
+ supportsTools: true,
835
+ // Free-tier default: cheapest model + a fast, non-thinking tool-calling
836
+ // executor — the model the IDE picks when none is chosen. Exactly one model
837
+ // in this catalog may carry freeTier (enforced by lookup.test.ts).
838
+ freeTier: true,
839
+ inputPricePerMTok: 0.14,
840
+ outputPricePerMTok: 0.28,
841
+ // DeepSeek automatic context cache: absolute cache-hit price ($/M).
842
+ cacheReadPricePerMTok: 0.0028,
843
+ cacheWritePricePerMTok: 0.14,
844
+ // Native-China default, matching deepseek-v4-pro (see its note) — even
845
+ // though Flash's US list price is BELOW native, its cache reads are 6.4×,
846
+ // and the plan/execute pair defaults to one region deliberately.
847
+ regions: ['cn', 'us'],
848
+ // US = DeepInfra, verified 2026-08-01 (api.deepinfra.com/models/…V4-Flash).
849
+ regionPricing: {
850
+ us: { inputPricePerMTok: 0.09, outputPricePerMTok: 0.18, cacheReadPricePerMTok: 0.018 },
851
+ },
852
+ // Peak-hour surcharge NOT active (see deepseek-v4-pro) — windows removed.
853
+ // Not published by DeepSeek — best-effort estimate.
854
+ knowledgeCutoff: '2025-07-01',
855
+ },
856
+ // ---------------------------------------------------------------------------
857
+ // Moonshot (Kimi)
858
+ // Verified: https://platform.kimi.ai/docs/models +
859
+ // /docs/guide/use-kimi-k2-thinking-model (2026-07-28).
860
+ // The moonshot bond now supports PRESERVED THINKING (ChatMessage.reasoning →
861
+ // reasoning_content replay through tool loops), which unblocked the two
862
+ // previously-excluded models: kimi-k3 (2026-07-16 flagship — 2.8T MoE, 1M
863
+ // ctx, forced thinking, reasoning_effort low|high|max default max upstream)
864
+ // and kimi-k2.7-code (coding flagship — forced thinking, no depth knob).
865
+ // kimi-k2.x thinking stays on/off only; the bond disables it for those by
866
+ // default (KIMI_REASONING_EFFORT env tunes it).
867
+ // ---------------------------------------------------------------------------
868
+ {
869
+ id: 'kimi-k3',
870
+ provider: 'moonshot',
871
+ label: 'Kimi K3',
872
+ description: 'Moonshot flagship — 2.8T open weights, 1M context, multimodal',
873
+ contextWindow: 1_000_000,
874
+ // Max output not documented — conservative cap (matches the K2.x family).
875
+ maxOutputTokens: 65_535,
876
+ supportsThinking: true,
877
+ thinkingBudgetTokens: 8_000,
878
+ // Thinking is FORCED ON; depth rides reasoning_effort (low|high|max). The
879
+ // upstream default is max — we default to high so agentic loops aren't
880
+ // pinned to the slowest/most expensive tier unless the user asks for it.
881
+ thinkingConfigurable: true,
882
+ supportedEffortLevels: ['low', 'high', 'max'],
883
+ defaultEffortLevel: 'high',
884
+ // Text + image + video input.
885
+ supportsVision: true,
886
+ supportsPromptCaching: true,
887
+ supportsTools: true,
888
+ inputPricePerMTok: 3,
889
+ outputPricePerMTok: 15,
890
+ // Automatic context cache: absolute cache-hit price ($0.30/M = 0.1× input).
891
+ cacheReadPricePerMTok: 0.3,
892
+ cacheWritePricePerMTok: 3,
893
+ // No US re-host exists (not on DeepInfra) — pinned to native China.
894
+ regions: ['cn'],
895
+ // Not published — best-effort estimate.
896
+ knowledgeCutoff: '2026-01-01',
897
+ },
898
+ {
899
+ id: 'kimi-k2.7-code',
900
+ provider: 'moonshot',
901
+ label: 'Kimi K2.7 Code',
902
+ description: 'Moonshot coding specialist — token-efficient agentic coding',
903
+ contextWindow: 262_144,
904
+ maxOutputTokens: 65_535,
905
+ supportsThinking: true,
906
+ thinkingBudgetTokens: 8_000,
907
+ // Thinking forced on, NO depth knob (reasoning_effort unsupported here) —
908
+ // the bond sends neither param and replays reasoning_content.
909
+ thinkingConfigurable: false,
910
+ // Coding model — vision not documented; conservative.
911
+ supportsVision: false,
912
+ supportsPromptCaching: true,
913
+ supportsTools: true,
914
+ inputPricePerMTok: 0.95,
915
+ outputPricePerMTok: 4,
916
+ // Automatic context cache: absolute cache-hit price ($0.19/M = 0.2× input).
917
+ cacheReadPricePerMTok: 0.19,
918
+ cacheWritePricePerMTok: 0.95,
919
+ // US default (DeepInfra bills below native here). Verified 2026-08-01.
920
+ regions: ['us', 'cn'],
921
+ regionPricing: {
922
+ us: { inputPricePerMTok: 0.74, outputPricePerMTok: 3.5, cacheReadPricePerMTok: 0.15 },
923
+ },
924
+ // Not published — best-effort estimate.
925
+ knowledgeCutoff: '2025-10-01',
926
+ // kimi-k3 is the Moonshot pick; the coding specialist stays selectable
927
+ // under "Older models" for anyone who wants the cheaper tier.
928
+ deprecatedAt: '2026-07-28',
929
+ },
930
+ {
931
+ id: 'kimi-k2.6',
932
+ provider: 'moonshot',
933
+ label: 'Kimi K2.6',
934
+ description: 'Multimodal agent — strong sequential tool use',
935
+ contextWindow: 262_144,
936
+ // Output shares the 256K context window (max_tokens default 32K upstream);
937
+ // practical cap.
938
+ maxOutputTokens: 65_535,
939
+ supportsThinking: true,
940
+ thinkingBudgetTokens: 8_000,
941
+ // Thinking is on/off only (default on upstream; the bond disables it by
942
+ // default for the executor loop — tune via KIMI_REASONING_EFFORT).
943
+ thinkingConfigurable: false,
944
+ supportsVision: true,
945
+ supportsPromptCaching: true,
946
+ supportsTools: true,
947
+ // Native platform.kimi.ai pricing (the old $0.68/$3.41 was OpenRouter's
948
+ // blended third-party rate).
949
+ inputPricePerMTok: 0.95,
950
+ outputPricePerMTok: 4,
951
+ cacheReadPricePerMTok: 0.16,
952
+ cacheWritePricePerMTok: 0.95,
953
+ // US default (DeepInfra bills below native here). Verified 2026-08-01.
954
+ regions: ['us', 'cn'],
955
+ regionPricing: {
956
+ us: { inputPricePerMTok: 0.75, outputPricePerMTok: 3.5, cacheReadPricePerMTok: 0.15 },
957
+ },
958
+ knowledgeCutoff: '2025-04-01',
959
+ // Superseded by kimi-k3; moved to "Older models".
960
+ deprecatedAt: '2026-07-28',
961
+ },
962
+ {
963
+ id: 'kimi-k2.5',
964
+ provider: 'moonshot',
965
+ label: 'Kimi K2.5',
966
+ description: 'Older Kimi — strong sequential tool use',
967
+ contextWindow: 262_144,
968
+ maxOutputTokens: 65_535,
969
+ supportsThinking: true,
970
+ thinkingBudgetTokens: 8_000,
971
+ // Thinking on/off only; no preserved-thinking support upstream.
972
+ thinkingConfigurable: false,
973
+ supportsVision: true,
974
+ supportsPromptCaching: true,
975
+ supportsTools: true,
976
+ // Native platform.kimi.ai pricing.
977
+ inputPricePerMTok: 0.6,
978
+ outputPricePerMTok: 3,
979
+ cacheReadPricePerMTok: 0.1,
980
+ cacheWritePricePerMTok: 0.6,
981
+ // US default (DeepInfra bills below native here). Verified 2026-08-01.
982
+ regions: ['us', 'cn'],
983
+ regionPricing: {
984
+ us: { inputPricePerMTok: 0.45, outputPricePerMTok: 2.25, cacheReadPricePerMTok: 0.07 },
985
+ },
986
+ knowledgeCutoff: '2024-04-01',
987
+ // Superseded by kimi-k2.6 (still served upstream, no announced retirement);
988
+ // kept selectable (Older models) + priceable.
989
+ deprecatedAt: '2026-04-01',
990
+ },
991
+ // ---------------------------------------------------------------------------
992
+ // MiniMax
993
+ // Verified: https://platform.minimax.io/docs/guides/models-intro
994
+ // https://platform.minimax.io/docs/guides/pricing-paygo.md
995
+ // minimax-m3 (2026-06-01) is the flagship: 1M context (≥512K guaranteed),
996
+ // multimodal (text+image+video in), thinking adaptive|disabled (no budget /
997
+ // effort levels). Run non-thinking like the DeepSeek executors — with
998
+ // thinking on, reasoning must be carried through tool loops. m2.7 thinking
999
+ // CANNOT be disabled upstream.
1000
+ // ---------------------------------------------------------------------------
1001
+ {
1002
+ id: 'minimax-m3',
1003
+ provider: 'minimax',
1004
+ label: 'MiniMax M3',
1005
+ description: 'Agentic flagship — 1M context, multimodal, great value',
1006
+ contextWindow: 1_048_576,
1007
+ // Upstream recommends 128K max_tokens (hard max 512K).
1008
+ maxOutputTokens: 131_072,
1009
+ // Run non-thinking (the bond sends thinking:{type:"disabled"} for M3);
1010
+ // native control is adaptive|disabled only — no depth levels.
1011
+ supportsThinking: false,
1012
+ thinkingBudgetTokens: 0,
1013
+ thinkingConfigurable: false,
1014
+ supportsVision: true,
1015
+ supportsPromptCaching: true,
1016
+ supportsTools: true,
1017
+ // ≤512K-token prompts; >512K bills 2× (tiering not modeled).
1018
+ inputPricePerMTok: 0.3,
1019
+ outputPricePerMTok: 1.2,
1020
+ cacheReadPricePerMTok: 0.06,
1021
+ // M3 cache-write price not published — M2.7's rate (≥ input as required).
1022
+ cacheWritePricePerMTok: 0.375,
1023
+ // US default. DeepInfra list matches native; only the cache write differs
1024
+ // (no premium → region input rate). Verified 2026-08-01.
1025
+ regions: ['us', 'cn'],
1026
+ regionPricing: {
1027
+ us: { inputPricePerMTok: 0.3, outputPricePerMTok: 1.2, cacheReadPricePerMTok: 0.06 },
1028
+ },
1029
+ // From the official HF chat template ("Knowledge cutoff: January 2026").
1030
+ knowledgeCutoff: '2026-01-01',
1031
+ },
1032
+ {
1033
+ id: 'minimax-m2.7',
1034
+ provider: 'minimax',
1035
+ label: 'MiniMax M2.7',
1036
+ description: 'Agentic productivity — strong value for coding',
1037
+ contextWindow: 204_800,
1038
+ maxOutputTokens: 196_608,
1039
+ supportsThinking: true,
1040
+ thinkingBudgetTokens: 8_000,
1041
+ // Thinking is ALWAYS ON for M2.x (cannot be disabled) — no depth control.
1042
+ thinkingConfigurable: false,
1043
+ supportsVision: false,
1044
+ supportsPromptCaching: true,
1045
+ supportsTools: true,
1046
+ // Native platform repriced (was $0.25/$1.00).
1047
+ inputPricePerMTok: 0.3,
1048
+ outputPricePerMTok: 1.2,
1049
+ cacheReadPricePerMTok: 0.06,
1050
+ cacheWritePricePerMTok: 0.375,
1051
+ // US default (DeepInfra still bills the pre-reprice rate). Verified
1052
+ // 2026-08-01.
1053
+ regions: ['us', 'cn'],
1054
+ regionPricing: {
1055
+ us: { inputPricePerMTok: 0.25, outputPricePerMTok: 1, cacheReadPricePerMTok: 0.05 },
1056
+ },
1057
+ knowledgeCutoff: '2025-09-01',
1058
+ // Superseded by minimax-m3 (same price, 1M ctx, multimodal); moved to
1059
+ // "Older models".
1060
+ deprecatedAt: '2026-07-28',
1061
+ },
1062
+ {
1063
+ id: 'minimax-m2.5',
1064
+ provider: 'minimax',
1065
+ label: 'MiniMax M2.5',
1066
+ description: 'Older MiniMax — strong value for coding',
1067
+ contextWindow: 196_608,
1068
+ maxOutputTokens: 196_608,
1069
+ supportsThinking: true,
1070
+ thinkingBudgetTokens: 8_000,
1071
+ // Thinking always on for M2.x — no depth control.
1072
+ thinkingConfigurable: false,
1073
+ supportsVision: false,
1074
+ supportsPromptCaching: true,
1075
+ supportsTools: true,
1076
+ inputPricePerMTok: 0.3,
1077
+ outputPricePerMTok: 1.2,
1078
+ cacheReadPricePerMTok: 0.03,
1079
+ cacheWritePricePerMTok: 0.375,
1080
+ // No US re-host exists (not on DeepInfra) — pinned to native China.
1081
+ regions: ['cn'],
1082
+ knowledgeCutoff: '2025-01-01',
1083
+ // Superseded by minimax-m3 (legacy upstream, still served); kept selectable
1084
+ // (Older models) + priceable.
1085
+ deprecatedAt: '2026-03-18',
1086
+ },
1087
+ // ---------------------------------------------------------------------------
1088
+ // Alibaba (Qwen)
1089
+ // Verified: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
1090
+ // https://www.alibabacloud.com/help/en/model-studio/qwen-coder
1091
+ // https://openrouter.ai/qwen/qwen3.7-max
1092
+ // qwen3.7-max (2026-05-21) is the agentic flagship — Alibaba's own Qwen-Coder
1093
+ // docs now recommend the general-purpose models over Qwen-Coder. Its thinking
1094
+ // uses enable_thinking (default ON for the 3.7 series) + thinking_budget
1095
+ // (token cap) — a real budget param, so effort scales the budget. The
1096
+ // qwen3-coder models are NON-thinking (previous catalog entry was wrong).
1097
+ // Prices are DashScope international list rates (the bond calls DashScope,
1098
+ // not OpenRouter; a 50%-off promo currently applies — billed at list).
1099
+ // ---------------------------------------------------------------------------
1100
+ {
1101
+ id: 'qwen3.7-max',
1102
+ provider: 'alibaba',
1103
+ label: 'Qwen3.7 Max',
1104
+ description: 'Alibaba agentic flagship — 1M context, hybrid thinking',
1105
+ contextWindow: 1_000_000,
1106
+ maxOutputTokens: 65_536,
1107
+ supportsThinking: true,
1108
+ thinkingBudgetTokens: 8_000,
1109
+ // enable_thinking + thinking_budget: a controllable token budget → effort
1110
+ // scales the budget (no native level names).
1111
+ thinkingConfigurable: true,
1112
+ supportedEffortLevels: ['4K', '8K', '16K', '32K'],
1113
+ defaultEffortLevel: '8K',
1114
+ effortBudgetTokens: { '4K': 4000, '8K': 8000, '16K': 16000, '32K': 32000 },
1115
+ supportsVision: false,
1116
+ supportsPromptCaching: true,
1117
+ supportsTools: true,
1118
+ inputPricePerMTok: 2.5,
1119
+ outputPricePerMTok: 7.5,
1120
+ // Implicit context cache: read ≈0.2× input, no write premium.
1121
+ cacheReadPricePerMTok: 0.5,
1122
+ cacheWritePricePerMTok: 2.5,
1123
+ // US default. DeepInfra bills identical rates (no regionPricing needed).
1124
+ // Verified 2026-08-01.
1125
+ regions: ['us', 'cn'],
1126
+ // Not published by Alibaba — best-effort estimate.
1127
+ knowledgeCutoff: '2026-01-01',
1128
+ },
1129
+ {
1130
+ id: 'qwen3-coder-plus',
1131
+ provider: 'alibaba',
1132
+ label: 'Qwen3 Coder Plus',
1133
+ description: 'Coding specialist — 1M context',
1134
+ contextWindow: 1_000_000,
1135
+ maxOutputTokens: 65_536,
1136
+ // Qwen3-Coder models support ONLY non-thinking mode (no thinking control
1137
+ // at all) — the previous "configurable thinking" entry was wrong.
1138
+ supportsThinking: false,
1139
+ thinkingBudgetTokens: 0,
1140
+ thinkingConfigurable: false,
1141
+ supportsVision: false,
1142
+ supportsPromptCaching: true,
1143
+ supportsTools: true,
1144
+ // DashScope international list rate, flat across input tiers (the old
1145
+ // $0.65/$3.25 was OpenRouter's promo rate).
1146
+ inputPricePerMTok: 1,
1147
+ outputPricePerMTok: 5,
1148
+ // Implicit context cache: read 0.2× input, no write premium.
1149
+ cacheReadPricePerMTok: 0.2,
1150
+ cacheWritePricePerMTok: 1,
1151
+ // US default (DeepInfra bills well below native). Verified 2026-08-01.
1152
+ regions: ['us', 'cn'],
1153
+ regionPricing: {
1154
+ us: { inputPricePerMTok: 0.3, outputPricePerMTok: 1, cacheReadPricePerMTok: 0.1 },
1155
+ },
1156
+ knowledgeCutoff: '2025-06-01',
1157
+ // Alibaba itself recommends the general-purpose models over Qwen-Coder;
1158
+ // qwen3.7-max is the pick — moved to "Older models".
1159
+ deprecatedAt: '2026-07-28',
1160
+ },
1161
+ // ---------------------------------------------------------------------------
1162
+ // Zhipu (GLM)
1163
+ // Verified: https://docs.z.ai/guides/overview/pricing
1164
+ // https://docs.z.ai/api-reference/llm/chat-completion
1165
+ // glm-5.2 (standalone API since 2026-06-16) is the flagship. It is the ONLY
1166
+ // GLM model with reasoning_effort (values minimal|none|low|medium|high|
1167
+ // xhigh|max; low/medium coerce to high, xhigh coerces to max — effective
1168
+ // levels are high|max plus minimal/none = skip thinking; default max).
1169
+ // glm-5 has thinking on/off only and was REPRICED (was $0.72/$2.30).
1170
+ // ---------------------------------------------------------------------------
1171
+ {
1172
+ id: 'glm-5.2',
1173
+ provider: 'zhipu',
1174
+ label: 'GLM-5.2',
1175
+ description: 'Open-source SOTA agentic — 1M context',
1176
+ contextWindow: 1_048_576,
1177
+ maxOutputTokens: 131_072,
1178
+ supportsThinking: true,
1179
+ thinkingBudgetTokens: 8_000,
1180
+ thinkingConfigurable: true,
1181
+ supportedEffortLevels: ['minimal', 'high', 'max'],
1182
+ defaultEffortLevel: 'high',
1183
+ // Z.ai's default is max (= L, positioned for long-horizon coding); M =
1184
+ // high is the balanced tier; minimal (= S) skips thinking for fast turns.
1185
+ supportsVision: false,
1186
+ supportsPromptCaching: true,
1187
+ supportsTools: true,
1188
+ webSearchToolType: 'web_search',
1189
+ inputPricePerMTok: 1.4,
1190
+ outputPricePerMTok: 4.4,
1191
+ // GLM context cache: read ≈0.19× input, no write premium.
1192
+ cacheReadPricePerMTok: 0.26,
1193
+ cacheWritePricePerMTok: 1.4,
1194
+ // US default (DeepInfra bills ~half native). Verified 2026-08-01.
1195
+ regions: ['us', 'cn'],
1196
+ regionPricing: {
1197
+ us: { inputPricePerMTok: 0.75, outputPricePerMTok: 2.4, cacheReadPricePerMTok: 0.14 },
1198
+ },
1199
+ // Not published by Z.ai — best-effort estimate.
1200
+ knowledgeCutoff: '2025-06-01',
1201
+ },
1202
+ {
1203
+ id: 'glm-5',
1204
+ provider: 'zhipu',
1205
+ label: 'GLM-5',
1206
+ description: '77.8% SWE-bench — open-source agentic, 200K context',
1207
+ contextWindow: 202_752,
1208
+ maxOutputTokens: 131_072,
1209
+ supportsThinking: true,
1210
+ thinkingBudgetTokens: 8_000,
1211
+ // Thinking on/off only (reasoning_effort is glm-5.2-only) — no depth
1212
+ // control; defaults to enabled upstream.
1213
+ thinkingConfigurable: false,
1214
+ supportsVision: false,
1215
+ supportsPromptCaching: true,
1216
+ supportsTools: true,
1217
+ webSearchToolType: 'web_search',
1218
+ // Repriced by Z.ai around the GLM-5.1 launch (was $0.72/$2.30).
1219
+ inputPricePerMTok: 1,
1220
+ outputPricePerMTok: 3.2,
1221
+ // GLM context cache: read 0.2× input, no write premium.
1222
+ cacheReadPricePerMTok: 0.2,
1223
+ cacheWritePricePerMTok: 1,
1224
+ // US default (DeepInfra bills below native). Verified 2026-08-01.
1225
+ regions: ['us', 'cn'],
1226
+ regionPricing: {
1227
+ us: { inputPricePerMTok: 0.6, outputPricePerMTok: 2.08, cacheReadPricePerMTok: 0.12 },
1228
+ },
1229
+ knowledgeCutoff: '2025-01-01',
1230
+ // Superseded by glm-5.2; moved to "Older models".
1231
+ deprecatedAt: '2026-07-28',
1232
+ },
1233
+ ];