@molecule/api-resource-ai-models 1.0.1 → 1.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -3,7 +3,7 @@ AUTO-GENERATED — DO NOT EDIT THIS FILE.
3
3
  Generated by `mlcl sync-docs` from the package's src/index.ts JSDoc + mlcl/registry.json.
4
4
  Edits here are overwritten on the next commit (molecule's pre-commit hook regenerates).
5
5
  To change this document, edit the module-level JSDoc in src/index.ts.
6
- Generated: 2026-08-04T01:49:08.449Z
6
+ Generated: 2026-08-06T19:42:52.531Z
7
7
  -->
8
8
 
9
9
  # @molecule/api-resource-ai-models
@@ -563,6 +563,9 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
563
563
  opus-4-8 superseded by opus-5 at identical pricing but still served — it is
564
564
  the recommended refusal-fallback model; effort ladder on all three current
565
565
  models is low|medium|high|xhigh|max; budget_tokens 400s on 4.7+)
566
+ (re-verified 2026-08-06: added opus-4-7 — legacy but Active, $5/$25,
567
+ cache $0.50/$6.25, 1M ctx / 128K out per the overview + pricing pages;
568
+ models.dev first listed it 2026-08-06)
566
569
  - OpenAI: https://developers.openai.com/api/docs/pricing (GPT-5.6 family GA
567
570
  2026-07-09; REPRICED 2026-07-30: -luna cut 80% to $0.20/$1.20, -terra cut
568
571
  20% to $2/$12, -sol unchanged $5/$30; cache read 0.1× input; gpt-5.5/
@@ -591,9 +594,18 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
591
594
  - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
592
595
  minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
593
596
  - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
594
- (unchanged; qwen3.8-max-preview, 2026-07-19, is Token-Plan-subscriber-only
595
- — not on the pay-as-you-go API, so it cannot be added yet; qwen3.7-max
596
- currently runs a 50%-off promo — still billed here at list, $2.50/$7.50)
597
+ (qwen3.8-max GA'd 2026-08-03 on the pay-as-you-go international API and is
598
+ IN the catalog — verified 2026-08-04: flat $2/$6 per MTok on
599
+ help.aliyun.com/en/model-studio/model-pricing (Singapore International
600
+ CNY 14.988/44.965 at the same fixed conversion that maps qwen3.7-max's
601
+ CNY 18.736/56.207 to its $2.50/$7.50 list), 1M ctx, hybrid thinking,
602
+ tools. Cache rates come from the ZH context-cache doc
603
+ (help.aliyun.com/zh/model-studio/context-cache), which lists qwen3.8-max
604
+ as supported in every region under the unconditional standard table
605
+ (implicit: hit 20% of input, creation 100%; explicit: hit 10%, creation
606
+ 125%) — the EN edition of that doc simply lags (zero qwen3.8 mentions),
607
+ which an earlier pass misread as "excepted/console-only". qwen3.7-max
608
+ still runs its 50%-off promo — billed here at list, $2.50/$7.50)
597
609
  - Zhipu: https://docs.z.ai/guides/overview/pricing (unchanged; glm-5.2 is
598
610
  the newest — "GLM-5.3/5.5" rumors have no released ids as of 2026-07-28)
599
611
 
package/dist/models.d.ts CHANGED
@@ -36,6 +36,9 @@ import type { ModelDefinition } from './types.js';
36
36
  * opus-4-8 superseded by opus-5 at identical pricing but still served — it is
37
37
  * the recommended refusal-fallback model; effort ladder on all three current
38
38
  * models is low|medium|high|xhigh|max; budget_tokens 400s on 4.7+)
39
+ * (re-verified 2026-08-06: added opus-4-7 — legacy but Active, $5/$25,
40
+ * cache $0.50/$6.25, 1M ctx / 128K out per the overview + pricing pages;
41
+ * models.dev first listed it 2026-08-06)
39
42
  * - OpenAI: https://developers.openai.com/api/docs/pricing (GPT-5.6 family GA
40
43
  * 2026-07-09; REPRICED 2026-07-30: -luna cut 80% to $0.20/$1.20, -terra cut
41
44
  * 20% to $2/$12, -sol unchanged $5/$30; cache read 0.1× input; gpt-5.5/
@@ -64,9 +67,18 @@ import type { ModelDefinition } from './types.js';
64
67
  * - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
65
68
  * minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
66
69
  * - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
67
- * (unchanged; qwen3.8-max-preview, 2026-07-19, is Token-Plan-subscriber-only
68
- * — not on the pay-as-you-go API, so it cannot be added yet; qwen3.7-max
69
- * currently runs a 50%-off promo — still billed here at list, $2.50/$7.50)
70
+ * (qwen3.8-max GA'd 2026-08-03 on the pay-as-you-go international API and is
71
+ * IN the catalog — verified 2026-08-04: flat $2/$6 per MTok on
72
+ * help.aliyun.com/en/model-studio/model-pricing (Singapore International
73
+ * CNY 14.988/44.965 at the same fixed conversion that maps qwen3.7-max's
74
+ * CNY 18.736/56.207 to its $2.50/$7.50 list), 1M ctx, hybrid thinking,
75
+ * tools. Cache rates come from the ZH context-cache doc
76
+ * (help.aliyun.com/zh/model-studio/context-cache), which lists qwen3.8-max
77
+ * as supported in every region under the unconditional standard table
78
+ * (implicit: hit 20% of input, creation 100%; explicit: hit 10%, creation
79
+ * 125%) — the EN edition of that doc simply lags (zero qwen3.8 mentions),
80
+ * which an earlier pass misread as "excepted/console-only". qwen3.7-max
81
+ * still runs its 50%-off promo — billed here at list, $2.50/$7.50)
70
82
  * - Zhipu: https://docs.z.ai/guides/overview/pricing (unchanged; glm-5.2 is
71
83
  * the newest — "GLM-5.3/5.5" rumors have no released ids as of 2026-07-28)
72
84
  *
@@ -1 +1 @@
1
- {"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAiEG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EA6oCnC,CAAA"}
1
+ {"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA6EG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAmtCnC,CAAA"}
package/dist/models.js CHANGED
@@ -35,6 +35,9 @@
35
35
  * opus-4-8 superseded by opus-5 at identical pricing but still served — it is
36
36
  * the recommended refusal-fallback model; effort ladder on all three current
37
37
  * models is low|medium|high|xhigh|max; budget_tokens 400s on 4.7+)
38
+ * (re-verified 2026-08-06: added opus-4-7 — legacy but Active, $5/$25,
39
+ * cache $0.50/$6.25, 1M ctx / 128K out per the overview + pricing pages;
40
+ * models.dev first listed it 2026-08-06)
38
41
  * - OpenAI: https://developers.openai.com/api/docs/pricing (GPT-5.6 family GA
39
42
  * 2026-07-09; REPRICED 2026-07-30: -luna cut 80% to $0.20/$1.20, -terra cut
40
43
  * 20% to $2/$12, -sol unchanged $5/$30; cache read 0.1× input; gpt-5.5/
@@ -63,9 +66,18 @@
63
66
  * - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
64
67
  * minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
65
68
  * - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
66
- * (unchanged; qwen3.8-max-preview, 2026-07-19, is Token-Plan-subscriber-only
67
- * — not on the pay-as-you-go API, so it cannot be added yet; qwen3.7-max
68
- * currently runs a 50%-off promo — still billed here at list, $2.50/$7.50)
69
+ * (qwen3.8-max GA'd 2026-08-03 on the pay-as-you-go international API and is
70
+ * IN the catalog — verified 2026-08-04: flat $2/$6 per MTok on
71
+ * help.aliyun.com/en/model-studio/model-pricing (Singapore International
72
+ * CNY 14.988/44.965 at the same fixed conversion that maps qwen3.7-max's
73
+ * CNY 18.736/56.207 to its $2.50/$7.50 list), 1M ctx, hybrid thinking,
74
+ * tools. Cache rates come from the ZH context-cache doc
75
+ * (help.aliyun.com/zh/model-studio/context-cache), which lists qwen3.8-max
76
+ * as supported in every region under the unconditional standard table
77
+ * (implicit: hit 20% of input, creation 100%; explicit: hit 10%, creation
78
+ * 125%) — the EN edition of that doc simply lags (zero qwen3.8 mentions),
79
+ * which an earlier pass misread as "excepted/console-only". qwen3.7-max
80
+ * still runs its 50%-off promo — billed here at list, $2.50/$7.50)
69
81
  * - Zhipu: https://docs.z.ai/guides/overview/pricing (unchanged; glm-5.2 is
70
82
  * the newest — "GLM-5.3/5.5" rumors have no released ids as of 2026-07-28)
71
83
  *
@@ -213,6 +225,40 @@ export const MODELS = [
213
225
  cacheWritePricePerMTok: 3.75,
214
226
  knowledgeCutoff: '2026-01-01',
215
227
  },
228
+ {
229
+ id: 'claude-opus-4-7',
230
+ provider: 'anthropic',
231
+ label: 'Claude Opus 4.7',
232
+ description: 'Older Opus — long-horizon agentic work, knowledge work & vision',
233
+ contextWindow: 1_000_000,
234
+ maxOutputTokens: 128_000,
235
+ supportsThinking: true,
236
+ thinkingBudgetTokens: 16_000,
237
+ thinkingConfigurable: true,
238
+ supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'],
239
+ defaultEffortLevel: 'high',
240
+ // Adaptive thinking only — budget_tokens is REJECTED (400); xhigh effort
241
+ // debuted on this model. Unlike opus-5, omitting `thinking` runs WITHOUT
242
+ // thinking — set {type:"adaptive"} explicitly. First model on the new
243
+ // tokenizer (~30% more tokens than 4.6 for the same text).
244
+ supportsVision: true,
245
+ supportsPromptCaching: true,
246
+ supportsTools: true,
247
+ webSearchToolType: 'web_search_20260209',
248
+ codeExecutionToolType: 'code_execution_20250825',
249
+ webFetchToolType: 'web_fetch_20260209',
250
+ inputPricePerMTok: 5,
251
+ outputPricePerMTok: 25,
252
+ // Anthropic 5-minute prompt cache: read 0.1× input, write 1.25× input.
253
+ cacheReadPricePerMTok: 0.5,
254
+ cacheWritePricePerMTok: 6.25,
255
+ knowledgeCutoff: '2026-01-01',
256
+ // Superseded by claude-opus-4-8 (launched 2026-05-28) at identical pricing;
257
+ // still Active upstream (deprecations page 2026-08-06: retires no sooner
258
+ // than 2027-04-16). Selectable under "Older models". NO fast mode —
259
+ // speed:"fast" on 4.7 returns an error (pricing page, fast-mode section).
260
+ deprecatedAt: '2026-05-28',
261
+ },
216
262
  {
217
263
  id: 'claude-opus-4-6',
218
264
  provider: 'anthropic',
@@ -1097,6 +1143,42 @@ export const MODELS = [
1097
1143
  // Prices are DashScope international list rates (the bond calls DashScope,
1098
1144
  // not OpenRouter; a 50%-off promo currently applies — billed at list).
1099
1145
  // ---------------------------------------------------------------------------
1146
+ // qwen3.8-max (GA 2026-08-03) succeeds qwen3.7-max as the agentic flagship,
1147
+ // priced BELOW it at $2/$6 (intl CNY 14.988/44.965, same fixed conversion).
1148
+ // Same hybrid thinking mechanism as the 3.7 series (enable_thinking default
1149
+ // ON + thinking_budget; preserve_thinking supported). Context cache uses the
1150
+ // standard implicit rates — see the Sources block for the ZH-doc citation.
1151
+ {
1152
+ id: 'qwen3.8-max',
1153
+ provider: 'alibaba',
1154
+ label: 'Qwen3.8 Max',
1155
+ description: 'Alibaba agentic flagship — 1M context, hybrid thinking',
1156
+ contextWindow: 1_000_000,
1157
+ // Alibaba's public pages don't state a max-output figure; models.dev says
1158
+ // 131,072 — kept at the 3.7-max figure until the provider publishes one
1159
+ // (understating only shortens completions; overstating would error).
1160
+ maxOutputTokens: 65_536,
1161
+ supportsThinking: true,
1162
+ thinkingBudgetTokens: 8_000,
1163
+ thinkingConfigurable: true,
1164
+ supportedEffortLevels: ['4K', '8K', '16K', '32K'],
1165
+ defaultEffortLevel: '8K',
1166
+ effortBudgetTokens: { '4K': 4000, '8K': 8000, '16K': 16000, '32K': 32000 },
1167
+ // models.dev claims image+video input, but Alibaba's own model catalog
1168
+ // lists qwen3.8-max under text generation (VL remains a separate line) —
1169
+ // false until the provider's page says otherwise.
1170
+ supportsVision: false,
1171
+ supportsPromptCaching: true,
1172
+ supportsTools: true,
1173
+ inputPricePerMTok: 2,
1174
+ outputPricePerMTok: 6,
1175
+ // Implicit context cache: read = 20% of input, no write premium.
1176
+ cacheReadPricePerMTok: 0.4,
1177
+ cacheWritePricePerMTok: 2,
1178
+ regions: ['us', 'cn'],
1179
+ // Not published by Alibaba — best-effort estimate.
1180
+ knowledgeCutoff: '2026-04-01',
1181
+ },
1100
1182
  {
1101
1183
  id: 'qwen3.7-max',
1102
1184
  provider: 'alibaba',
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@molecule/api-resource-ai-models",
3
- "version": "1.0.1",
3
+ "version": "1.0.2",
4
4
  "description": "AI model catalog — server-side source of truth plus an authentication-gated discovery endpoint",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",