@molecule/api-resource-ai-models 1.2.7 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +11 -1
- package/dist/models.d.ts +10 -0
- package/dist/models.d.ts.map +1 -1
- package/dist/models.js +66 -5
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -3,7 +3,7 @@ AUTO-GENERATED — DO NOT EDIT THIS FILE.
|
|
|
3
3
|
Generated by `mlcl sync-docs` from the package's src/index.ts JSDoc + mlcl/registry.json.
|
|
4
4
|
Edits here are overwritten on the next commit (molecule's pre-commit hook regenerates).
|
|
5
5
|
To change this document, edit the module-level JSDoc in src/index.ts.
|
|
6
|
-
Generated: 2026-08-
|
|
6
|
+
Generated: 2026-08-28T19:44:44.140Z
|
|
7
7
|
-->
|
|
8
8
|
|
|
9
9
|
# @molecule/api-resource-ai-models
|
|
@@ -851,6 +851,16 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
851
851
|
$1.65/$4.951), so nothing would ever select it — and carrying both would
|
|
852
852
|
put two selectable Alibaba flagships in one family. Revisit only if Alibaba
|
|
853
853
|
publishes it as a distinct first-party DashScope model id.)
|
|
854
|
+
(re-verified 2026-08-28 on the per-model pages, which publish USD rates per
|
|
855
|
+
region directly — a better source than the CNY pricing table: qwen3.8-max's
|
|
856
|
+
Singapore implicit-cache rate is $0.25, NOT the 20%/$0.40 the ZH cache doc's
|
|
857
|
+
standard table implies. That doc names qwen3.8-max, qwen3.8-flash and
|
|
858
|
+
qwen3.8-2.4t-a95b as exceptions and defers to the console; the model pages
|
|
859
|
+
are the console's figures. qwen3.7-max's $0.50 (20%) is confirmed unchanged.
|
|
860
|
+
qwen3.8-flash (2026-08-26) ADDED: $0.15/$0.47, implicit cache $0.016, 1M ctx
|
|
861
|
+
/ 131,072 out, Image+Text+Video input, tools, hybrid thinking ON by default
|
|
862
|
+
with thinking_budget. It is cn-region-only — api.deepinfra.com has no
|
|
863
|
+
Qwen/Qwen3.8-Flash — so ALIBABA_US_MODEL_MAP needs no entry.)
|
|
854
864
|
- Zhipu: https://docs.z.ai/guides/overview/pricing + docs.z.ai/guides/llm/
|
|
855
865
|
glm-5.3 (verified 2026-08-26: glm-5.3 shipped 2026-08-14 and IS in the
|
|
856
866
|
catalog — $1.40/$4.40, cached $0.26, i.e. glm-5.2's card unchanged, 1M ctx,
|
package/dist/models.d.ts
CHANGED
|
@@ -135,6 +135,16 @@ import type { ModelDefinition } from './types.js';
|
|
|
135
135
|
* $1.65/$4.951), so nothing would ever select it — and carrying both would
|
|
136
136
|
* put two selectable Alibaba flagships in one family. Revisit only if Alibaba
|
|
137
137
|
* publishes it as a distinct first-party DashScope model id.)
|
|
138
|
+
* (re-verified 2026-08-28 on the per-model pages, which publish USD rates per
|
|
139
|
+
* region directly — a better source than the CNY pricing table: qwen3.8-max's
|
|
140
|
+
* Singapore implicit-cache rate is $0.25, NOT the 20%/$0.40 the ZH cache doc's
|
|
141
|
+
* standard table implies. That doc names qwen3.8-max, qwen3.8-flash and
|
|
142
|
+
* qwen3.8-2.4t-a95b as exceptions and defers to the console; the model pages
|
|
143
|
+
* are the console's figures. qwen3.7-max's $0.50 (20%) is confirmed unchanged.
|
|
144
|
+
* qwen3.8-flash (2026-08-26) ADDED: $0.15/$0.47, implicit cache $0.016, 1M ctx
|
|
145
|
+
* / 131,072 out, Image+Text+Video input, tools, hybrid thinking ON by default
|
|
146
|
+
* with thinking_budget. It is cn-region-only — api.deepinfra.com has no
|
|
147
|
+
* Qwen/Qwen3.8-Flash — so ALIBABA_US_MODEL_MAP needs no entry.)
|
|
138
148
|
* - Zhipu: https://docs.z.ai/guides/overview/pricing + docs.z.ai/guides/llm/
|
|
139
149
|
* glm-5.3 (verified 2026-08-26: glm-5.3 shipped 2026-08-14 and IS in the
|
|
140
150
|
* catalog — $1.40/$4.40, cached $0.26, i.e. glm-5.2's card unchanged, 1M ctx,
|
package/dist/models.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD
|
|
1
|
+
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA6JG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAymDnC,CAAA"}
|
package/dist/models.js
CHANGED
|
@@ -134,6 +134,16 @@
|
|
|
134
134
|
* $1.65/$4.951), so nothing would ever select it — and carrying both would
|
|
135
135
|
* put two selectable Alibaba flagships in one family. Revisit only if Alibaba
|
|
136
136
|
* publishes it as a distinct first-party DashScope model id.)
|
|
137
|
+
* (re-verified 2026-08-28 on the per-model pages, which publish USD rates per
|
|
138
|
+
* region directly — a better source than the CNY pricing table: qwen3.8-max's
|
|
139
|
+
* Singapore implicit-cache rate is $0.25, NOT the 20%/$0.40 the ZH cache doc's
|
|
140
|
+
* standard table implies. That doc names qwen3.8-max, qwen3.8-flash and
|
|
141
|
+
* qwen3.8-2.4t-a95b as exceptions and defers to the console; the model pages
|
|
142
|
+
* are the console's figures. qwen3.7-max's $0.50 (20%) is confirmed unchanged.
|
|
143
|
+
* qwen3.8-flash (2026-08-26) ADDED: $0.15/$0.47, implicit cache $0.016, 1M ctx
|
|
144
|
+
* / 131,072 out, Image+Text+Video input, tools, hybrid thinking ON by default
|
|
145
|
+
* with thinking_budget. It is cn-region-only — api.deepinfra.com has no
|
|
146
|
+
* Qwen/Qwen3.8-Flash — so ALIBABA_US_MODEL_MAP needs no entry.)
|
|
137
147
|
* - Zhipu: https://docs.z.ai/guides/overview/pricing + docs.z.ai/guides/llm/
|
|
138
148
|
* glm-5.3 (verified 2026-08-26: glm-5.3 shipped 2026-08-14 and IS in the
|
|
139
149
|
* catalog — $1.40/$4.40, cached $0.26, i.e. glm-5.2's card unchanged, 1M ctx,
|
|
@@ -1469,16 +1479,24 @@ export const MODELS = [
|
|
|
1469
1479
|
supportedEffortLevels: ['4K', '8K', '16K', '32K'],
|
|
1470
1480
|
defaultEffortLevel: '8K',
|
|
1471
1481
|
effortBudgetTokens: { '4K': 4000, '8K': 8000, '16K': 16000, '32K': 32000 },
|
|
1472
|
-
//
|
|
1473
|
-
//
|
|
1474
|
-
// false
|
|
1482
|
+
// The provider's own model page DOES list Image/Text/Video input (checked
|
|
1483
|
+
// 2026-08-28, superseding the earlier "text generation only" reading of the
|
|
1484
|
+
// model catalog page). Kept false anyway: this model's DEFAULT region is the
|
|
1485
|
+
// DeepInfra US re-host, which serves it as `text-generation` — vision is not
|
|
1486
|
+
// offerable where it actually dispatches. Native-only qwen3.8-flash below
|
|
1487
|
+
// carries the flag.
|
|
1475
1488
|
supportsVision: false,
|
|
1476
1489
|
supportsPromptCaching: true,
|
|
1477
1490
|
supportsTools: true,
|
|
1478
1491
|
inputPricePerMTok: 2,
|
|
1479
1492
|
outputPricePerMTok: 6,
|
|
1480
|
-
// Implicit context cache
|
|
1481
|
-
|
|
1493
|
+
// Implicit context cache. The ZH cache doc's standard 20% does NOT apply
|
|
1494
|
+
// here: that doc names qwen3.8-max as an exception and points at the
|
|
1495
|
+
// console, and the model page publishes the real Singapore rate — $0.25,
|
|
1496
|
+
// i.e. 12.5% of input (verified 2026-08-28 on
|
|
1497
|
+
// alibabacloud.com/help/en/model-studio/qwen3-8-max). Implicit creation is
|
|
1498
|
+
// not billed beyond input, so write stays 1× input.
|
|
1499
|
+
cacheReadPricePerMTok: 0.25,
|
|
1482
1500
|
cacheWritePricePerMTok: 2,
|
|
1483
1501
|
regions: ['us', 'cn'],
|
|
1484
1502
|
// US = DeepInfra (Qwen/Qwen3.8-Max), verified 2026-08-14 against
|
|
@@ -1492,6 +1510,49 @@ export const MODELS = [
|
|
|
1492
1510
|
// Not published by Alibaba — best-effort estimate.
|
|
1493
1511
|
knowledgeCutoff: '2026-04-01',
|
|
1494
1512
|
},
|
|
1513
|
+
{
|
|
1514
|
+
id: 'qwen3.8-flash',
|
|
1515
|
+
provider: 'alibaba',
|
|
1516
|
+
label: 'Qwen3.8 Flash',
|
|
1517
|
+
description: 'Alibaba cheap tier — 1M context, multimodal, hybrid thinking',
|
|
1518
|
+
// Model page: context window 1,000,000; max input 991,808 (983,616 in
|
|
1519
|
+
// thinking mode); max chain-of-thought 262,144; max output 131,072.
|
|
1520
|
+
contextWindow: 1_000_000,
|
|
1521
|
+
maxOutputTokens: 131_072,
|
|
1522
|
+
// Same hybrid-thinking mechanism as qwen3.8-max — the deep-thinking doc
|
|
1523
|
+
// lists the "Qwen3.8 Flash series" as hybrid with thinking ON by default,
|
|
1524
|
+
// and thinking_budget applies — so effort scales the budget identically.
|
|
1525
|
+
supportsThinking: true,
|
|
1526
|
+
thinkingBudgetTokens: 8_000,
|
|
1527
|
+
thinkingConfigurable: true,
|
|
1528
|
+
supportedEffortLevels: ['4K', '8K', '16K', '32K'],
|
|
1529
|
+
defaultEffortLevel: '8K',
|
|
1530
|
+
effortBudgetTokens: { '4K': 4000, '8K': 8000, '16K': 16000, '32K': 32000 },
|
|
1531
|
+
// Model page lists Image / Text / Video input. Unlike qwen3.8-max this one
|
|
1532
|
+
// has no US re-host to lose it to — it only ever dispatches to the native
|
|
1533
|
+
// host, which serves the multimodal surface.
|
|
1534
|
+
supportsVision: true,
|
|
1535
|
+
supportsPromptCaching: true,
|
|
1536
|
+
supportsTools: true,
|
|
1537
|
+
// Singapore International list card, published in USD on the model page:
|
|
1538
|
+
// $0.15 / $0.47, implicit cache $0.016. (Beijing is cheaper at
|
|
1539
|
+
// $0.113/$0.382/$0.014; the bond calls the international endpoint.)
|
|
1540
|
+
inputPricePerMTok: 0.15,
|
|
1541
|
+
outputPricePerMTok: 0.47,
|
|
1542
|
+
// Implicit context cache. Like qwen3.8-max this model is an explicit
|
|
1543
|
+
// exception to the ZH doc's standard 20%; the published rate is $0.016
|
|
1544
|
+
// (~10.7% of input). Creation is not billed beyond input → write = 1×.
|
|
1545
|
+
cacheReadPricePerMTok: 0.016,
|
|
1546
|
+
cacheWritePricePerMTok: 0.15,
|
|
1547
|
+
// No US re-host exists — DeepInfra serves Qwen3.8-Max, Qwen3.8-27B and
|
|
1548
|
+
// Qwen3.8-2.4T-A95B but returns "model not found" for Qwen/Qwen3.8-Flash
|
|
1549
|
+
// (checked 2026-08-28), so this is pinned to the native host and bills the
|
|
1550
|
+
// card above. ALIBABA_US_MODEL_MAP therefore needs no entry.
|
|
1551
|
+
regions: ['cn'],
|
|
1552
|
+
// Not published by Alibaba — best-effort estimate, same 3.8 generation as
|
|
1553
|
+
// qwen3.8-max.
|
|
1554
|
+
knowledgeCutoff: '2026-04-01',
|
|
1555
|
+
},
|
|
1495
1556
|
{
|
|
1496
1557
|
id: 'qwen3.7-max',
|
|
1497
1558
|
provider: 'alibaba',
|
package/package.json
CHANGED