@molecule/api-resource-ai-models 1.2.7 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -3,7 +3,7 @@ AUTO-GENERATED — DO NOT EDIT THIS FILE.
3
3
  Generated by `mlcl sync-docs` from the package's src/index.ts JSDoc + mlcl/registry.json.
4
4
  Edits here are overwritten on the next commit (molecule's pre-commit hook regenerates).
5
5
  To change this document, edit the module-level JSDoc in src/index.ts.
6
- Generated: 2026-08-28T08:14:50.991Z
6
+ Generated: 2026-08-28T19:44:44.140Z
7
7
  -->
8
8
 
9
9
  # @molecule/api-resource-ai-models
@@ -851,6 +851,16 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
851
851
  $1.65/$4.951), so nothing would ever select it — and carrying both would
852
852
  put two selectable Alibaba flagships in one family. Revisit only if Alibaba
853
853
  publishes it as a distinct first-party DashScope model id.)
854
+ (re-verified 2026-08-28 on the per-model pages, which publish USD rates per
855
+ region directly — a better source than the CNY pricing table: qwen3.8-max's
856
+ Singapore implicit-cache rate is $0.25, NOT the 20%/$0.40 the ZH cache doc's
857
+ standard table implies. That doc names qwen3.8-max, qwen3.8-flash and
858
+ qwen3.8-2.4t-a95b as exceptions and defers to the console; the model pages
859
+ are the console's figures. qwen3.7-max's $0.50 (20%) is confirmed unchanged.
860
+ qwen3.8-flash (2026-08-26) ADDED: $0.15/$0.47, implicit cache $0.016, 1M ctx
861
+ / 131,072 out, Image+Text+Video input, tools, hybrid thinking ON by default
862
+ with thinking_budget. It is cn-region-only — api.deepinfra.com has no
863
+ Qwen/Qwen3.8-Flash — so ALIBABA_US_MODEL_MAP needs no entry.)
854
864
  - Zhipu: https://docs.z.ai/guides/overview/pricing + docs.z.ai/guides/llm/
855
865
  glm-5.3 (verified 2026-08-26: glm-5.3 shipped 2026-08-14 and IS in the
856
866
  catalog — $1.40/$4.40, cached $0.26, i.e. glm-5.2's card unchanged, 1M ctx,
package/dist/models.d.ts CHANGED
@@ -135,6 +135,16 @@ import type { ModelDefinition } from './types.js';
135
135
  * $1.65/$4.951), so nothing would ever select it — and carrying both would
136
136
  * put two selectable Alibaba flagships in one family. Revisit only if Alibaba
137
137
  * publishes it as a distinct first-party DashScope model id.)
138
+ * (re-verified 2026-08-28 on the per-model pages, which publish USD rates per
139
+ * region directly — a better source than the CNY pricing table: qwen3.8-max's
140
+ * Singapore implicit-cache rate is $0.25, NOT the 20%/$0.40 the ZH cache doc's
141
+ * standard table implies. That doc names qwen3.8-max, qwen3.8-flash and
142
+ * qwen3.8-2.4t-a95b as exceptions and defers to the console; the model pages
143
+ * are the console's figures. qwen3.7-max's $0.50 (20%) is confirmed unchanged.
144
+ * qwen3.8-flash (2026-08-26) ADDED: $0.15/$0.47, implicit cache $0.016, 1M ctx
145
+ * / 131,072 out, Image+Text+Video input, tools, hybrid thinking ON by default
146
+ * with thinking_budget. It is cn-region-only — api.deepinfra.com has no
147
+ * Qwen/Qwen3.8-Flash — so ALIBABA_US_MODEL_MAP needs no entry.)
138
148
  * - Zhipu: https://docs.z.ai/guides/overview/pricing + docs.z.ai/guides/llm/
139
149
  * glm-5.3 (verified 2026-08-26: glm-5.3 shipped 2026-08-14 and IS in the
140
150
  * catalog — $1.40/$4.40, cached $0.26, i.e. glm-5.2's card unchanged, 1M ctx,
@@ -1 +1 @@
1
- {"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAmJG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAsjDnC,CAAA"}
1
+ {"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA6JG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAymDnC,CAAA"}
package/dist/models.js CHANGED
@@ -134,6 +134,16 @@
134
134
  * $1.65/$4.951), so nothing would ever select it — and carrying both would
135
135
  * put two selectable Alibaba flagships in one family. Revisit only if Alibaba
136
136
  * publishes it as a distinct first-party DashScope model id.)
137
+ * (re-verified 2026-08-28 on the per-model pages, which publish USD rates per
138
+ * region directly — a better source than the CNY pricing table: qwen3.8-max's
139
+ * Singapore implicit-cache rate is $0.25, NOT the 20%/$0.40 the ZH cache doc's
140
+ * standard table implies. That doc names qwen3.8-max, qwen3.8-flash and
141
+ * qwen3.8-2.4t-a95b as exceptions and defers to the console; the model pages
142
+ * are the console's figures. qwen3.7-max's $0.50 (20%) is confirmed unchanged.
143
+ * qwen3.8-flash (2026-08-26) ADDED: $0.15/$0.47, implicit cache $0.016, 1M ctx
144
+ * / 131,072 out, Image+Text+Video input, tools, hybrid thinking ON by default
145
+ * with thinking_budget. It is cn-region-only — api.deepinfra.com has no
146
+ * Qwen/Qwen3.8-Flash — so ALIBABA_US_MODEL_MAP needs no entry.)
137
147
  * - Zhipu: https://docs.z.ai/guides/overview/pricing + docs.z.ai/guides/llm/
138
148
  * glm-5.3 (verified 2026-08-26: glm-5.3 shipped 2026-08-14 and IS in the
139
149
  * catalog — $1.40/$4.40, cached $0.26, i.e. glm-5.2's card unchanged, 1M ctx,
@@ -1469,16 +1479,24 @@ export const MODELS = [
1469
1479
  supportedEffortLevels: ['4K', '8K', '16K', '32K'],
1470
1480
  defaultEffortLevel: '8K',
1471
1481
  effortBudgetTokens: { '4K': 4000, '8K': 8000, '16K': 16000, '32K': 32000 },
1472
- // models.dev claims image+video input, but Alibaba's own model catalog
1473
- // lists qwen3.8-max under text generation (VL remains a separate line) —
1474
- // false until the provider's page says otherwise.
1482
+ // The provider's own model page DOES list Image/Text/Video input (checked
1483
+ // 2026-08-28, superseding the earlier "text generation only" reading of the
1484
+ // model catalog page). Kept false anyway: this model's DEFAULT region is the
1485
+ // DeepInfra US re-host, which serves it as `text-generation` — vision is not
1486
+ // offerable where it actually dispatches. Native-only qwen3.8-flash below
1487
+ // carries the flag.
1475
1488
  supportsVision: false,
1476
1489
  supportsPromptCaching: true,
1477
1490
  supportsTools: true,
1478
1491
  inputPricePerMTok: 2,
1479
1492
  outputPricePerMTok: 6,
1480
- // Implicit context cache: read = 20% of input, no write premium.
1481
- cacheReadPricePerMTok: 0.4,
1493
+ // Implicit context cache. The ZH cache doc's standard 20% does NOT apply
1494
+ // here: that doc names qwen3.8-max as an exception and points at the
1495
+ // console, and the model page publishes the real Singapore rate — $0.25,
1496
+ // i.e. 12.5% of input (verified 2026-08-28 on
1497
+ // alibabacloud.com/help/en/model-studio/qwen3-8-max). Implicit creation is
1498
+ // not billed beyond input, so write stays 1× input.
1499
+ cacheReadPricePerMTok: 0.25,
1482
1500
  cacheWritePricePerMTok: 2,
1483
1501
  regions: ['us', 'cn'],
1484
1502
  // US = DeepInfra (Qwen/Qwen3.8-Max), verified 2026-08-14 against
@@ -1492,6 +1510,49 @@ export const MODELS = [
1492
1510
  // Not published by Alibaba — best-effort estimate.
1493
1511
  knowledgeCutoff: '2026-04-01',
1494
1512
  },
1513
+ {
1514
+ id: 'qwen3.8-flash',
1515
+ provider: 'alibaba',
1516
+ label: 'Qwen3.8 Flash',
1517
+ description: 'Alibaba cheap tier — 1M context, multimodal, hybrid thinking',
1518
+ // Model page: context window 1,000,000; max input 991,808 (983,616 in
1519
+ // thinking mode); max chain-of-thought 262,144; max output 131,072.
1520
+ contextWindow: 1_000_000,
1521
+ maxOutputTokens: 131_072,
1522
+ // Same hybrid-thinking mechanism as qwen3.8-max — the deep-thinking doc
1523
+ // lists the "Qwen3.8 Flash series" as hybrid with thinking ON by default,
1524
+ // and thinking_budget applies — so effort scales the budget identically.
1525
+ supportsThinking: true,
1526
+ thinkingBudgetTokens: 8_000,
1527
+ thinkingConfigurable: true,
1528
+ supportedEffortLevels: ['4K', '8K', '16K', '32K'],
1529
+ defaultEffortLevel: '8K',
1530
+ effortBudgetTokens: { '4K': 4000, '8K': 8000, '16K': 16000, '32K': 32000 },
1531
+ // Model page lists Image / Text / Video input. Unlike qwen3.8-max this one
1532
+ // has no US re-host to lose it to — it only ever dispatches to the native
1533
+ // host, which serves the multimodal surface.
1534
+ supportsVision: true,
1535
+ supportsPromptCaching: true,
1536
+ supportsTools: true,
1537
+ // Singapore International list card, published in USD on the model page:
1538
+ // $0.15 / $0.47, implicit cache $0.016. (Beijing is cheaper at
1539
+ // $0.113/$0.382/$0.014; the bond calls the international endpoint.)
1540
+ inputPricePerMTok: 0.15,
1541
+ outputPricePerMTok: 0.47,
1542
+ // Implicit context cache. Like qwen3.8-max this model is an explicit
1543
+ // exception to the ZH doc's standard 20%; the published rate is $0.016
1544
+ // (~10.7% of input). Creation is not billed beyond input → write = 1×.
1545
+ cacheReadPricePerMTok: 0.016,
1546
+ cacheWritePricePerMTok: 0.15,
1547
+ // No US re-host exists — DeepInfra serves Qwen3.8-Max, Qwen3.8-27B and
1548
+ // Qwen3.8-2.4T-A95B but returns "model not found" for Qwen/Qwen3.8-Flash
1549
+ // (checked 2026-08-28), so this is pinned to the native host and bills the
1550
+ // card above. ALIBABA_US_MODEL_MAP therefore needs no entry.
1551
+ regions: ['cn'],
1552
+ // Not published by Alibaba — best-effort estimate, same 3.8 generation as
1553
+ // qwen3.8-max.
1554
+ knowledgeCutoff: '2026-04-01',
1555
+ },
1495
1556
  {
1496
1557
  id: 'qwen3.7-max',
1497
1558
  provider: 'alibaba',
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@molecule/api-resource-ai-models",
3
- "version": "1.2.7",
3
+ "version": "1.3.0",
4
4
  "description": "AI model catalog — server-side source of truth plus an authentication-gated discovery endpoint",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",