@molecule/api-resource-ai-models 1.0.1 → 1.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -4
- package/dist/models.d.ts +15 -3
- package/dist/models.d.ts.map +1 -1
- package/dist/models.js +85 -3
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -3,7 +3,7 @@ AUTO-GENERATED — DO NOT EDIT THIS FILE.
|
|
|
3
3
|
Generated by `mlcl sync-docs` from the package's src/index.ts JSDoc + mlcl/registry.json.
|
|
4
4
|
Edits here are overwritten on the next commit (molecule's pre-commit hook regenerates).
|
|
5
5
|
To change this document, edit the module-level JSDoc in src/index.ts.
|
|
6
|
-
Generated: 2026-08-
|
|
6
|
+
Generated: 2026-08-06T19:42:52.531Z
|
|
7
7
|
-->
|
|
8
8
|
|
|
9
9
|
# @molecule/api-resource-ai-models
|
|
@@ -563,6 +563,9 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
563
563
|
opus-4-8 superseded by opus-5 at identical pricing but still served — it is
|
|
564
564
|
the recommended refusal-fallback model; effort ladder on all three current
|
|
565
565
|
models is low|medium|high|xhigh|max; budget_tokens 400s on 4.7+)
|
|
566
|
+
(re-verified 2026-08-06: added opus-4-7 — legacy but Active, $5/$25,
|
|
567
|
+
cache $0.50/$6.25, 1M ctx / 128K out per the overview + pricing pages;
|
|
568
|
+
models.dev first listed it 2026-08-06)
|
|
566
569
|
- OpenAI: https://developers.openai.com/api/docs/pricing (GPT-5.6 family GA
|
|
567
570
|
2026-07-09; REPRICED 2026-07-30: -luna cut 80% to $0.20/$1.20, -terra cut
|
|
568
571
|
20% to $2/$12, -sol unchanged $5/$30; cache read 0.1× input; gpt-5.5/
|
|
@@ -591,9 +594,18 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
591
594
|
- MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
|
|
592
595
|
minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
|
|
593
596
|
- Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
|
|
594
|
-
(
|
|
595
|
-
|
|
596
|
-
|
|
597
|
+
(qwen3.8-max GA'd 2026-08-03 on the pay-as-you-go international API and is
|
|
598
|
+
IN the catalog — verified 2026-08-04: flat $2/$6 per MTok on
|
|
599
|
+
help.aliyun.com/en/model-studio/model-pricing (Singapore International
|
|
600
|
+
CNY 14.988/44.965 at the same fixed conversion that maps qwen3.7-max's
|
|
601
|
+
CNY 18.736/56.207 to its $2.50/$7.50 list), 1M ctx, hybrid thinking,
|
|
602
|
+
tools. Cache rates come from the ZH context-cache doc
|
|
603
|
+
(help.aliyun.com/zh/model-studio/context-cache), which lists qwen3.8-max
|
|
604
|
+
as supported in every region under the unconditional standard table
|
|
605
|
+
(implicit: hit 20% of input, creation 100%; explicit: hit 10%, creation
|
|
606
|
+
125%) — the EN edition of that doc simply lags (zero qwen3.8 mentions),
|
|
607
|
+
which an earlier pass misread as "excepted/console-only". qwen3.7-max
|
|
608
|
+
still runs its 50%-off promo — billed here at list, $2.50/$7.50)
|
|
597
609
|
- Zhipu: https://docs.z.ai/guides/overview/pricing (unchanged; glm-5.2 is
|
|
598
610
|
the newest — "GLM-5.3/5.5" rumors have no released ids as of 2026-07-28)
|
|
599
611
|
|
package/dist/models.d.ts
CHANGED
|
@@ -36,6 +36,9 @@ import type { ModelDefinition } from './types.js';
|
|
|
36
36
|
* opus-4-8 superseded by opus-5 at identical pricing but still served — it is
|
|
37
37
|
* the recommended refusal-fallback model; effort ladder on all three current
|
|
38
38
|
* models is low|medium|high|xhigh|max; budget_tokens 400s on 4.7+)
|
|
39
|
+
* (re-verified 2026-08-06: added opus-4-7 — legacy but Active, $5/$25,
|
|
40
|
+
* cache $0.50/$6.25, 1M ctx / 128K out per the overview + pricing pages;
|
|
41
|
+
* models.dev first listed it 2026-08-06)
|
|
39
42
|
* - OpenAI: https://developers.openai.com/api/docs/pricing (GPT-5.6 family GA
|
|
40
43
|
* 2026-07-09; REPRICED 2026-07-30: -luna cut 80% to $0.20/$1.20, -terra cut
|
|
41
44
|
* 20% to $2/$12, -sol unchanged $5/$30; cache read 0.1× input; gpt-5.5/
|
|
@@ -64,9 +67,18 @@ import type { ModelDefinition } from './types.js';
|
|
|
64
67
|
* - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
|
|
65
68
|
* minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
|
|
66
69
|
* - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
|
|
67
|
-
* (
|
|
68
|
-
*
|
|
69
|
-
*
|
|
70
|
+
* (qwen3.8-max GA'd 2026-08-03 on the pay-as-you-go international API and is
|
|
71
|
+
* IN the catalog — verified 2026-08-04: flat $2/$6 per MTok on
|
|
72
|
+
* help.aliyun.com/en/model-studio/model-pricing (Singapore International
|
|
73
|
+
* CNY 14.988/44.965 at the same fixed conversion that maps qwen3.7-max's
|
|
74
|
+
* CNY 18.736/56.207 to its $2.50/$7.50 list), 1M ctx, hybrid thinking,
|
|
75
|
+
* tools. Cache rates come from the ZH context-cache doc
|
|
76
|
+
* (help.aliyun.com/zh/model-studio/context-cache), which lists qwen3.8-max
|
|
77
|
+
* as supported in every region under the unconditional standard table
|
|
78
|
+
* (implicit: hit 20% of input, creation 100%; explicit: hit 10%, creation
|
|
79
|
+
* 125%) — the EN edition of that doc simply lags (zero qwen3.8 mentions),
|
|
80
|
+
* which an earlier pass misread as "excepted/console-only". qwen3.7-max
|
|
81
|
+
* still runs its 50%-off promo — billed here at list, $2.50/$7.50)
|
|
70
82
|
* - Zhipu: https://docs.z.ai/guides/overview/pricing (unchanged; glm-5.2 is
|
|
71
83
|
* the newest — "GLM-5.3/5.5" rumors have no released ids as of 2026-07-28)
|
|
72
84
|
*
|
package/dist/models.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD
|
|
1
|
+
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA6EG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAmtCnC,CAAA"}
|
package/dist/models.js
CHANGED
|
@@ -35,6 +35,9 @@
|
|
|
35
35
|
* opus-4-8 superseded by opus-5 at identical pricing but still served — it is
|
|
36
36
|
* the recommended refusal-fallback model; effort ladder on all three current
|
|
37
37
|
* models is low|medium|high|xhigh|max; budget_tokens 400s on 4.7+)
|
|
38
|
+
* (re-verified 2026-08-06: added opus-4-7 — legacy but Active, $5/$25,
|
|
39
|
+
* cache $0.50/$6.25, 1M ctx / 128K out per the overview + pricing pages;
|
|
40
|
+
* models.dev first listed it 2026-08-06)
|
|
38
41
|
* - OpenAI: https://developers.openai.com/api/docs/pricing (GPT-5.6 family GA
|
|
39
42
|
* 2026-07-09; REPRICED 2026-07-30: -luna cut 80% to $0.20/$1.20, -terra cut
|
|
40
43
|
* 20% to $2/$12, -sol unchanged $5/$30; cache read 0.1× input; gpt-5.5/
|
|
@@ -63,9 +66,18 @@
|
|
|
63
66
|
* - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
|
|
64
67
|
* minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
|
|
65
68
|
* - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
|
|
66
|
-
* (
|
|
67
|
-
*
|
|
68
|
-
*
|
|
69
|
+
* (qwen3.8-max GA'd 2026-08-03 on the pay-as-you-go international API and is
|
|
70
|
+
* IN the catalog — verified 2026-08-04: flat $2/$6 per MTok on
|
|
71
|
+
* help.aliyun.com/en/model-studio/model-pricing (Singapore International
|
|
72
|
+
* CNY 14.988/44.965 at the same fixed conversion that maps qwen3.7-max's
|
|
73
|
+
* CNY 18.736/56.207 to its $2.50/$7.50 list), 1M ctx, hybrid thinking,
|
|
74
|
+
* tools. Cache rates come from the ZH context-cache doc
|
|
75
|
+
* (help.aliyun.com/zh/model-studio/context-cache), which lists qwen3.8-max
|
|
76
|
+
* as supported in every region under the unconditional standard table
|
|
77
|
+
* (implicit: hit 20% of input, creation 100%; explicit: hit 10%, creation
|
|
78
|
+
* 125%) — the EN edition of that doc simply lags (zero qwen3.8 mentions),
|
|
79
|
+
* which an earlier pass misread as "excepted/console-only". qwen3.7-max
|
|
80
|
+
* still runs its 50%-off promo — billed here at list, $2.50/$7.50)
|
|
69
81
|
* - Zhipu: https://docs.z.ai/guides/overview/pricing (unchanged; glm-5.2 is
|
|
70
82
|
* the newest — "GLM-5.3/5.5" rumors have no released ids as of 2026-07-28)
|
|
71
83
|
*
|
|
@@ -213,6 +225,40 @@ export const MODELS = [
|
|
|
213
225
|
cacheWritePricePerMTok: 3.75,
|
|
214
226
|
knowledgeCutoff: '2026-01-01',
|
|
215
227
|
},
|
|
228
|
+
{
|
|
229
|
+
id: 'claude-opus-4-7',
|
|
230
|
+
provider: 'anthropic',
|
|
231
|
+
label: 'Claude Opus 4.7',
|
|
232
|
+
description: 'Older Opus — long-horizon agentic work, knowledge work & vision',
|
|
233
|
+
contextWindow: 1_000_000,
|
|
234
|
+
maxOutputTokens: 128_000,
|
|
235
|
+
supportsThinking: true,
|
|
236
|
+
thinkingBudgetTokens: 16_000,
|
|
237
|
+
thinkingConfigurable: true,
|
|
238
|
+
supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'],
|
|
239
|
+
defaultEffortLevel: 'high',
|
|
240
|
+
// Adaptive thinking only — budget_tokens is REJECTED (400); xhigh effort
|
|
241
|
+
// debuted on this model. Unlike opus-5, omitting `thinking` runs WITHOUT
|
|
242
|
+
// thinking — set {type:"adaptive"} explicitly. First model on the new
|
|
243
|
+
// tokenizer (~30% more tokens than 4.6 for the same text).
|
|
244
|
+
supportsVision: true,
|
|
245
|
+
supportsPromptCaching: true,
|
|
246
|
+
supportsTools: true,
|
|
247
|
+
webSearchToolType: 'web_search_20260209',
|
|
248
|
+
codeExecutionToolType: 'code_execution_20250825',
|
|
249
|
+
webFetchToolType: 'web_fetch_20260209',
|
|
250
|
+
inputPricePerMTok: 5,
|
|
251
|
+
outputPricePerMTok: 25,
|
|
252
|
+
// Anthropic 5-minute prompt cache: read 0.1× input, write 1.25× input.
|
|
253
|
+
cacheReadPricePerMTok: 0.5,
|
|
254
|
+
cacheWritePricePerMTok: 6.25,
|
|
255
|
+
knowledgeCutoff: '2026-01-01',
|
|
256
|
+
// Superseded by claude-opus-4-8 (launched 2026-05-28) at identical pricing;
|
|
257
|
+
// still Active upstream (deprecations page 2026-08-06: retires no sooner
|
|
258
|
+
// than 2027-04-16). Selectable under "Older models". NO fast mode —
|
|
259
|
+
// speed:"fast" on 4.7 returns an error (pricing page, fast-mode section).
|
|
260
|
+
deprecatedAt: '2026-05-28',
|
|
261
|
+
},
|
|
216
262
|
{
|
|
217
263
|
id: 'claude-opus-4-6',
|
|
218
264
|
provider: 'anthropic',
|
|
@@ -1097,6 +1143,42 @@ export const MODELS = [
|
|
|
1097
1143
|
// Prices are DashScope international list rates (the bond calls DashScope,
|
|
1098
1144
|
// not OpenRouter; a 50%-off promo currently applies — billed at list).
|
|
1099
1145
|
// ---------------------------------------------------------------------------
|
|
1146
|
+
// qwen3.8-max (GA 2026-08-03) succeeds qwen3.7-max as the agentic flagship,
|
|
1147
|
+
// priced BELOW it at $2/$6 (intl CNY 14.988/44.965, same fixed conversion).
|
|
1148
|
+
// Same hybrid thinking mechanism as the 3.7 series (enable_thinking default
|
|
1149
|
+
// ON + thinking_budget; preserve_thinking supported). Context cache uses the
|
|
1150
|
+
// standard implicit rates — see the Sources block for the ZH-doc citation.
|
|
1151
|
+
{
|
|
1152
|
+
id: 'qwen3.8-max',
|
|
1153
|
+
provider: 'alibaba',
|
|
1154
|
+
label: 'Qwen3.8 Max',
|
|
1155
|
+
description: 'Alibaba agentic flagship — 1M context, hybrid thinking',
|
|
1156
|
+
contextWindow: 1_000_000,
|
|
1157
|
+
// Alibaba's public pages don't state a max-output figure; models.dev says
|
|
1158
|
+
// 131,072 — kept at the 3.7-max figure until the provider publishes one
|
|
1159
|
+
// (understating only shortens completions; overstating would error).
|
|
1160
|
+
maxOutputTokens: 65_536,
|
|
1161
|
+
supportsThinking: true,
|
|
1162
|
+
thinkingBudgetTokens: 8_000,
|
|
1163
|
+
thinkingConfigurable: true,
|
|
1164
|
+
supportedEffortLevels: ['4K', '8K', '16K', '32K'],
|
|
1165
|
+
defaultEffortLevel: '8K',
|
|
1166
|
+
effortBudgetTokens: { '4K': 4000, '8K': 8000, '16K': 16000, '32K': 32000 },
|
|
1167
|
+
// models.dev claims image+video input, but Alibaba's own model catalog
|
|
1168
|
+
// lists qwen3.8-max under text generation (VL remains a separate line) —
|
|
1169
|
+
// false until the provider's page says otherwise.
|
|
1170
|
+
supportsVision: false,
|
|
1171
|
+
supportsPromptCaching: true,
|
|
1172
|
+
supportsTools: true,
|
|
1173
|
+
inputPricePerMTok: 2,
|
|
1174
|
+
outputPricePerMTok: 6,
|
|
1175
|
+
// Implicit context cache: read = 20% of input, no write premium.
|
|
1176
|
+
cacheReadPricePerMTok: 0.4,
|
|
1177
|
+
cacheWritePricePerMTok: 2,
|
|
1178
|
+
regions: ['us', 'cn'],
|
|
1179
|
+
// Not published by Alibaba — best-effort estimate.
|
|
1180
|
+
knowledgeCutoff: '2026-04-01',
|
|
1181
|
+
},
|
|
1100
1182
|
{
|
|
1101
1183
|
id: 'qwen3.7-max',
|
|
1102
1184
|
provider: 'alibaba',
|
package/package.json
CHANGED