@molecule/api-resource-ai-models 1.2.6 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +21 -1
- package/dist/models.d.ts +20 -0
- package/dist/models.d.ts.map +1 -1
- package/dist/models.js +156 -18
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -3,7 +3,7 @@ AUTO-GENERATED — DO NOT EDIT THIS FILE.
|
|
|
3
3
|
Generated by `mlcl sync-docs` from the package's src/index.ts JSDoc + mlcl/registry.json.
|
|
4
4
|
Edits here are overwritten on the next commit (molecule's pre-commit hook regenerates).
|
|
5
5
|
To change this document, edit the module-level JSDoc in src/index.ts.
|
|
6
|
-
Generated: 2026-08-
|
|
6
|
+
Generated: 2026-08-28T19:44:44.140Z
|
|
7
7
|
-->
|
|
8
8
|
|
|
9
9
|
# @molecule/api-resource-ai-models
|
|
@@ -728,6 +728,16 @@ All available AI models, grouped by provider, ordered from most to least capable
|
|
|
728
728
|
To add or remove a model, edit this array. Both the server-side validation
|
|
729
729
|
and the public discovery endpoint will update automatically.
|
|
730
730
|
|
|
731
|
+
EVERY entry field that shapes the outbound request (`webSearchToolType`,
|
|
732
|
+
`supportedEffortLevels`/`defaultEffortLevel`/`effortBudgetTokens`,
|
|
733
|
+
`maxOutputTokens`, `regions`) must hold for the model's ACTUAL serving
|
|
734
|
+
host(s) in every region — not just look right in a docs table. The
|
|
735
|
+
pre-commit hook enforces this with a LIVE Synthase-shaped probe of each
|
|
736
|
+
added/changed entry (molecule-dev `verify:model-dispatch --staged-catalog`);
|
|
737
|
+
a value the host rejects fails the whole request for every user
|
|
738
|
+
(glm-5.3-flash 2026-08-28: a `webSearchToolType` both zhipu hosts reject
|
|
739
|
+
made every Synthase turn 400/422 while the model itself worked fine).
|
|
740
|
+
|
|
731
741
|
Effort is each model's OWN native value — there is no abstract scale (see
|
|
732
742
|
{@link ModelDefinition.supportedEffortLevels}):
|
|
733
743
|
|
|
@@ -841,6 +851,16 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
841
851
|
$1.65/$4.951), so nothing would ever select it — and carrying both would
|
|
842
852
|
put two selectable Alibaba flagships in one family. Revisit only if Alibaba
|
|
843
853
|
publishes it as a distinct first-party DashScope model id.)
|
|
854
|
+
(re-verified 2026-08-28 on the per-model pages, which publish USD rates per
|
|
855
|
+
region directly — a better source than the CNY pricing table: qwen3.8-max's
|
|
856
|
+
Singapore implicit-cache rate is $0.25, NOT the 20%/$0.40 the ZH cache doc's
|
|
857
|
+
standard table implies. That doc names qwen3.8-max, qwen3.8-flash and
|
|
858
|
+
qwen3.8-2.4t-a95b as exceptions and defers to the console; the model pages
|
|
859
|
+
are the console's figures. qwen3.7-max's $0.50 (20%) is confirmed unchanged.
|
|
860
|
+
qwen3.8-flash (2026-08-26) ADDED: $0.15/$0.47, implicit cache $0.016, 1M ctx
|
|
861
|
+
/ 131,072 out, Image+Text+Video input, tools, hybrid thinking ON by default
|
|
862
|
+
with thinking_budget. It is cn-region-only — api.deepinfra.com has no
|
|
863
|
+
Qwen/Qwen3.8-Flash — so ALIBABA_US_MODEL_MAP needs no entry.)
|
|
844
864
|
- Zhipu: https://docs.z.ai/guides/overview/pricing + docs.z.ai/guides/llm/
|
|
845
865
|
glm-5.3 (verified 2026-08-26: glm-5.3 shipped 2026-08-14 and IS in the
|
|
846
866
|
catalog — $1.40/$4.40, cached $0.26, i.e. glm-5.2's card unchanged, 1M ctx,
|
package/dist/models.d.ts
CHANGED
|
@@ -14,6 +14,16 @@ import type { ModelDefinition } from './types.js';
|
|
|
14
14
|
* To add or remove a model, edit this array. Both the server-side validation
|
|
15
15
|
* and the public discovery endpoint will update automatically.
|
|
16
16
|
*
|
|
17
|
+
* EVERY entry field that shapes the outbound request (`webSearchToolType`,
|
|
18
|
+
* `supportedEffortLevels`/`defaultEffortLevel`/`effortBudgetTokens`,
|
|
19
|
+
* `maxOutputTokens`, `regions`) must hold for the model's ACTUAL serving
|
|
20
|
+
* host(s) in every region — not just look right in a docs table. The
|
|
21
|
+
* pre-commit hook enforces this with a LIVE Synthase-shaped probe of each
|
|
22
|
+
* added/changed entry (molecule-dev `verify:model-dispatch --staged-catalog`);
|
|
23
|
+
* a value the host rejects fails the whole request for every user
|
|
24
|
+
* (glm-5.3-flash 2026-08-28: a `webSearchToolType` both zhipu hosts reject
|
|
25
|
+
* made every Synthase turn 400/422 while the model itself worked fine).
|
|
26
|
+
*
|
|
17
27
|
* Effort is each model's OWN native value — there is no abstract scale (see
|
|
18
28
|
* {@link ModelDefinition.supportedEffortLevels}):
|
|
19
29
|
* - A model driven by a provider-native effort/level param lists its provider
|
|
@@ -125,6 +135,16 @@ import type { ModelDefinition } from './types.js';
|
|
|
125
135
|
* $1.65/$4.951), so nothing would ever select it — and carrying both would
|
|
126
136
|
* put two selectable Alibaba flagships in one family. Revisit only if Alibaba
|
|
127
137
|
* publishes it as a distinct first-party DashScope model id.)
|
|
138
|
+
* (re-verified 2026-08-28 on the per-model pages, which publish USD rates per
|
|
139
|
+
* region directly — a better source than the CNY pricing table: qwen3.8-max's
|
|
140
|
+
* Singapore implicit-cache rate is $0.25, NOT the 20%/$0.40 the ZH cache doc's
|
|
141
|
+
* standard table implies. That doc names qwen3.8-max, qwen3.8-flash and
|
|
142
|
+
* qwen3.8-2.4t-a95b as exceptions and defers to the console; the model pages
|
|
143
|
+
* are the console's figures. qwen3.7-max's $0.50 (20%) is confirmed unchanged.
|
|
144
|
+
* qwen3.8-flash (2026-08-26) ADDED: $0.15/$0.47, implicit cache $0.016, 1M ctx
|
|
145
|
+
* / 131,072 out, Image+Text+Video input, tools, hybrid thinking ON by default
|
|
146
|
+
* with thinking_budget. It is cn-region-only — api.deepinfra.com has no
|
|
147
|
+
* Qwen/Qwen3.8-Flash — so ALIBABA_US_MODEL_MAP needs no entry.)
|
|
128
148
|
* - Zhipu: https://docs.z.ai/guides/overview/pricing + docs.z.ai/guides/llm/
|
|
129
149
|
* glm-5.3 (verified 2026-08-26: glm-5.3 shipped 2026-08-14 and IS in the
|
|
130
150
|
* catalog — $1.40/$4.40, cached $0.26, i.e. glm-5.2's card unchanged, 1M ctx,
|
package/dist/models.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD
|
|
1
|
+
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA6JG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAymDnC,CAAA"}
|
package/dist/models.js
CHANGED
|
@@ -13,6 +13,16 @@
|
|
|
13
13
|
* To add or remove a model, edit this array. Both the server-side validation
|
|
14
14
|
* and the public discovery endpoint will update automatically.
|
|
15
15
|
*
|
|
16
|
+
* EVERY entry field that shapes the outbound request (`webSearchToolType`,
|
|
17
|
+
* `supportedEffortLevels`/`defaultEffortLevel`/`effortBudgetTokens`,
|
|
18
|
+
* `maxOutputTokens`, `regions`) must hold for the model's ACTUAL serving
|
|
19
|
+
* host(s) in every region — not just look right in a docs table. The
|
|
20
|
+
* pre-commit hook enforces this with a LIVE Synthase-shaped probe of each
|
|
21
|
+
* added/changed entry (molecule-dev `verify:model-dispatch --staged-catalog`);
|
|
22
|
+
* a value the host rejects fails the whole request for every user
|
|
23
|
+
* (glm-5.3-flash 2026-08-28: a `webSearchToolType` both zhipu hosts reject
|
|
24
|
+
* made every Synthase turn 400/422 while the model itself worked fine).
|
|
25
|
+
*
|
|
16
26
|
* Effort is each model's OWN native value — there is no abstract scale (see
|
|
17
27
|
* {@link ModelDefinition.supportedEffortLevels}):
|
|
18
28
|
* - A model driven by a provider-native effort/level param lists its provider
|
|
@@ -124,6 +134,16 @@
|
|
|
124
134
|
* $1.65/$4.951), so nothing would ever select it — and carrying both would
|
|
125
135
|
* put two selectable Alibaba flagships in one family. Revisit only if Alibaba
|
|
126
136
|
* publishes it as a distinct first-party DashScope model id.)
|
|
137
|
+
* (re-verified 2026-08-28 on the per-model pages, which publish USD rates per
|
|
138
|
+
* region directly — a better source than the CNY pricing table: qwen3.8-max's
|
|
139
|
+
* Singapore implicit-cache rate is $0.25, NOT the 20%/$0.40 the ZH cache doc's
|
|
140
|
+
* standard table implies. That doc names qwen3.8-max, qwen3.8-flash and
|
|
141
|
+
* qwen3.8-2.4t-a95b as exceptions and defers to the console; the model pages
|
|
142
|
+
* are the console's figures. qwen3.7-max's $0.50 (20%) is confirmed unchanged.
|
|
143
|
+
* qwen3.8-flash (2026-08-26) ADDED: $0.15/$0.47, implicit cache $0.016, 1M ctx
|
|
144
|
+
* / 131,072 out, Image+Text+Video input, tools, hybrid thinking ON by default
|
|
145
|
+
* with thinking_budget. It is cn-region-only — api.deepinfra.com has no
|
|
146
|
+
* Qwen/Qwen3.8-Flash — so ALIBABA_US_MODEL_MAP needs no entry.)
|
|
127
147
|
* - Zhipu: https://docs.z.ai/guides/overview/pricing + docs.z.ai/guides/llm/
|
|
128
148
|
* glm-5.3 (verified 2026-08-26: glm-5.3 shipped 2026-08-14 and IS in the
|
|
129
149
|
* catalog — $1.40/$4.40, cached $0.26, i.e. glm-5.2's card unchanged, 1M ctx,
|
|
@@ -174,7 +194,11 @@ export const MODELS = [
|
|
|
174
194
|
supportsPromptCaching: true,
|
|
175
195
|
supportsTools: true,
|
|
176
196
|
webSearchToolType: 'web_search_20260209',
|
|
177
|
-
|
|
197
|
+
// Current server-tool type per the API reference (verified 2026-08-28);
|
|
198
|
+
// 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
|
|
199
|
+
// not a tool type — sending it as one is a 400. Not currently sent by
|
|
200
|
+
// Synthase (request-shape.ts forwards only webSearchToolType).
|
|
201
|
+
codeExecutionToolType: 'code_execution_20260521',
|
|
178
202
|
webFetchToolType: 'web_fetch_20260209',
|
|
179
203
|
inputPricePerMTok: 10,
|
|
180
204
|
outputPricePerMTok: 50,
|
|
@@ -214,7 +238,11 @@ export const MODELS = [
|
|
|
214
238
|
supportsPromptCaching: true,
|
|
215
239
|
supportsTools: true,
|
|
216
240
|
webSearchToolType: 'web_search_20260209',
|
|
217
|
-
|
|
241
|
+
// Current server-tool type per the API reference (verified 2026-08-28);
|
|
242
|
+
// 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
|
|
243
|
+
// not a tool type — sending it as one is a 400. Not currently sent by
|
|
244
|
+
// Synthase (request-shape.ts forwards only webSearchToolType).
|
|
245
|
+
codeExecutionToolType: 'code_execution_20260521',
|
|
218
246
|
webFetchToolType: 'web_fetch_20260209',
|
|
219
247
|
// Drop-in successor to Opus 4.8 at identical pricing.
|
|
220
248
|
inputPricePerMTok: 5,
|
|
@@ -243,7 +271,11 @@ export const MODELS = [
|
|
|
243
271
|
supportsPromptCaching: true,
|
|
244
272
|
supportsTools: true,
|
|
245
273
|
webSearchToolType: 'web_search_20260209',
|
|
246
|
-
|
|
274
|
+
// Current server-tool type per the API reference (verified 2026-08-28);
|
|
275
|
+
// 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
|
|
276
|
+
// not a tool type — sending it as one is a 400. Not currently sent by
|
|
277
|
+
// Synthase (request-shape.ts forwards only webSearchToolType).
|
|
278
|
+
codeExecutionToolType: 'code_execution_20260521',
|
|
247
279
|
webFetchToolType: 'web_fetch_20260209',
|
|
248
280
|
inputPricePerMTok: 5,
|
|
249
281
|
outputPricePerMTok: 25,
|
|
@@ -276,7 +308,11 @@ export const MODELS = [
|
|
|
276
308
|
supportsPromptCaching: true,
|
|
277
309
|
supportsTools: true,
|
|
278
310
|
webSearchToolType: 'web_search_20260209',
|
|
279
|
-
|
|
311
|
+
// Current server-tool type per the API reference (verified 2026-08-28);
|
|
312
|
+
// 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
|
|
313
|
+
// not a tool type — sending it as one is a 400. Not currently sent by
|
|
314
|
+
// Synthase (request-shape.ts forwards only webSearchToolType).
|
|
315
|
+
codeExecutionToolType: 'code_execution_20260521',
|
|
280
316
|
webFetchToolType: 'web_fetch_20260209',
|
|
281
317
|
// Standard pricing. Intro pricing ($2/$10) applies through 2026-08-31 —
|
|
282
318
|
// billed here at standard so metering never under-charges; revisit after.
|
|
@@ -307,7 +343,11 @@ export const MODELS = [
|
|
|
307
343
|
supportsPromptCaching: true,
|
|
308
344
|
supportsTools: true,
|
|
309
345
|
webSearchToolType: 'web_search_20260209',
|
|
310
|
-
|
|
346
|
+
// Current server-tool type per the API reference (verified 2026-08-28);
|
|
347
|
+
// 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
|
|
348
|
+
// not a tool type — sending it as one is a 400. Not currently sent by
|
|
349
|
+
// Synthase (request-shape.ts forwards only webSearchToolType).
|
|
350
|
+
codeExecutionToolType: 'code_execution_20260521',
|
|
311
351
|
webFetchToolType: 'web_fetch_20260209',
|
|
312
352
|
inputPricePerMTok: 5,
|
|
313
353
|
outputPricePerMTok: 25,
|
|
@@ -342,7 +382,11 @@ export const MODELS = [
|
|
|
342
382
|
supportsPromptCaching: true,
|
|
343
383
|
supportsTools: true,
|
|
344
384
|
webSearchToolType: 'web_search_20260209',
|
|
345
|
-
|
|
385
|
+
// Current server-tool type per the API reference (verified 2026-08-28);
|
|
386
|
+
// 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
|
|
387
|
+
// not a tool type — sending it as one is a 400. Not currently sent by
|
|
388
|
+
// Synthase (request-shape.ts forwards only webSearchToolType).
|
|
389
|
+
codeExecutionToolType: 'code_execution_20260521',
|
|
346
390
|
webFetchToolType: 'web_fetch_20260209',
|
|
347
391
|
inputPricePerMTok: 5,
|
|
348
392
|
outputPricePerMTok: 25,
|
|
@@ -373,7 +417,11 @@ export const MODELS = [
|
|
|
373
417
|
supportsPromptCaching: true,
|
|
374
418
|
supportsTools: true,
|
|
375
419
|
webSearchToolType: 'web_search_20260209',
|
|
376
|
-
|
|
420
|
+
// Current server-tool type per the API reference (verified 2026-08-28);
|
|
421
|
+
// 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
|
|
422
|
+
// not a tool type — sending it as one is a 400. Not currently sent by
|
|
423
|
+
// Synthase (request-shape.ts forwards only webSearchToolType).
|
|
424
|
+
codeExecutionToolType: 'code_execution_20260521',
|
|
377
425
|
webFetchToolType: 'web_fetch_20260209',
|
|
378
426
|
inputPricePerMTok: 3,
|
|
379
427
|
outputPricePerMTok: 15,
|
|
@@ -446,7 +494,11 @@ export const MODELS = [
|
|
|
446
494
|
supportsTools: true,
|
|
447
495
|
// Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
|
|
448
496
|
toolsRequireReasoningOff: true,
|
|
449
|
-
webSearchToolType:
|
|
497
|
+
// NO webSearchToolType: the OpenAI bond calls /v1/chat/completions, which
|
|
498
|
+
// has no web_search tool type (it is a Responses-API construct), and the
|
|
499
|
+
// bond deliberately forwards no server tools. Advertising one here surfaced
|
|
500
|
+
// web search in the system prompt while it could never work. Re-add when
|
|
501
|
+
// the bond moves to /v1/responses (verified 2026-08-28).
|
|
450
502
|
codeExecutionToolType: 'code_interpreter',
|
|
451
503
|
// LIST price. OpenAI ran a >20% PROMO from 2026-08-22 ($4/$20, cache read
|
|
452
504
|
// $0.40, cache write $5) — "GPT-5.6 Sol's promotional pricing is available
|
|
@@ -479,7 +531,11 @@ export const MODELS = [
|
|
|
479
531
|
supportsTools: true,
|
|
480
532
|
// Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
|
|
481
533
|
toolsRequireReasoningOff: true,
|
|
482
|
-
webSearchToolType:
|
|
534
|
+
// NO webSearchToolType: the OpenAI bond calls /v1/chat/completions, which
|
|
535
|
+
// has no web_search tool type (it is a Responses-API construct), and the
|
|
536
|
+
// bond deliberately forwards no server tools. Advertising one here surfaced
|
|
537
|
+
// web search in the system prompt while it could never work. Re-add when
|
|
538
|
+
// the bond moves to /v1/responses (verified 2026-08-28).
|
|
483
539
|
codeExecutionToolType: 'code_interpreter',
|
|
484
540
|
// Repriced 2026-07-30 (20% cut from $2.50/$15).
|
|
485
541
|
inputPricePerMTok: 2,
|
|
@@ -507,7 +563,11 @@ export const MODELS = [
|
|
|
507
563
|
supportsTools: true,
|
|
508
564
|
// Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
|
|
509
565
|
toolsRequireReasoningOff: true,
|
|
510
|
-
webSearchToolType:
|
|
566
|
+
// NO webSearchToolType: the OpenAI bond calls /v1/chat/completions, which
|
|
567
|
+
// has no web_search tool type (it is a Responses-API construct), and the
|
|
568
|
+
// bond deliberately forwards no server tools. Advertising one here surfaced
|
|
569
|
+
// web search in the system prompt while it could never work. Re-add when
|
|
570
|
+
// the bond moves to /v1/responses (verified 2026-08-28).
|
|
511
571
|
codeExecutionToolType: 'code_interpreter',
|
|
512
572
|
// Repriced 2026-07-30 (80% cut from $1/$6).
|
|
513
573
|
inputPricePerMTok: 0.2,
|
|
@@ -542,7 +602,11 @@ export const MODELS = [
|
|
|
542
602
|
supportsTools: true,
|
|
543
603
|
// Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
|
|
544
604
|
toolsRequireReasoningOff: true,
|
|
545
|
-
webSearchToolType:
|
|
605
|
+
// NO webSearchToolType: the OpenAI bond calls /v1/chat/completions, which
|
|
606
|
+
// has no web_search tool type (it is a Responses-API construct), and the
|
|
607
|
+
// bond deliberately forwards no server tools. Advertising one here surfaced
|
|
608
|
+
// web search in the system prompt while it could never work. Re-add when
|
|
609
|
+
// the bond moves to /v1/responses (verified 2026-08-28).
|
|
546
610
|
codeExecutionToolType: 'code_interpreter',
|
|
547
611
|
inputPricePerMTok: 5,
|
|
548
612
|
outputPricePerMTok: 30,
|
|
@@ -572,7 +636,11 @@ export const MODELS = [
|
|
|
572
636
|
supportsTools: true,
|
|
573
637
|
// Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
|
|
574
638
|
toolsRequireReasoningOff: true,
|
|
575
|
-
webSearchToolType:
|
|
639
|
+
// NO webSearchToolType: the OpenAI bond calls /v1/chat/completions, which
|
|
640
|
+
// has no web_search tool type (it is a Responses-API construct), and the
|
|
641
|
+
// bond deliberately forwards no server tools. Advertising one here surfaced
|
|
642
|
+
// web search in the system prompt while it could never work. Re-add when
|
|
643
|
+
// the bond moves to /v1/responses (verified 2026-08-28).
|
|
576
644
|
codeExecutionToolType: 'code_interpreter',
|
|
577
645
|
inputPricePerMTok: 2.5,
|
|
578
646
|
outputPricePerMTok: 15,
|
|
@@ -604,7 +672,11 @@ export const MODELS = [
|
|
|
604
672
|
supportsTools: true,
|
|
605
673
|
// Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
|
|
606
674
|
toolsRequireReasoningOff: true,
|
|
607
|
-
webSearchToolType:
|
|
675
|
+
// NO webSearchToolType: the OpenAI bond calls /v1/chat/completions, which
|
|
676
|
+
// has no web_search tool type (it is a Responses-API construct), and the
|
|
677
|
+
// bond deliberately forwards no server tools. Advertising one here surfaced
|
|
678
|
+
// web search in the system prompt while it could never work. Re-add when
|
|
679
|
+
// the bond moves to /v1/responses (verified 2026-08-28).
|
|
608
680
|
codeExecutionToolType: 'code_interpreter',
|
|
609
681
|
inputPricePerMTok: 0.75,
|
|
610
682
|
outputPricePerMTok: 4.5,
|
|
@@ -1407,16 +1479,24 @@ export const MODELS = [
|
|
|
1407
1479
|
supportedEffortLevels: ['4K', '8K', '16K', '32K'],
|
|
1408
1480
|
defaultEffortLevel: '8K',
|
|
1409
1481
|
effortBudgetTokens: { '4K': 4000, '8K': 8000, '16K': 16000, '32K': 32000 },
|
|
1410
|
-
//
|
|
1411
|
-
//
|
|
1412
|
-
// false
|
|
1482
|
+
// The provider's own model page DOES list Image/Text/Video input (checked
|
|
1483
|
+
// 2026-08-28, superseding the earlier "text generation only" reading of the
|
|
1484
|
+
// model catalog page). Kept false anyway: this model's DEFAULT region is the
|
|
1485
|
+
// DeepInfra US re-host, which serves it as `text-generation` — vision is not
|
|
1486
|
+
// offerable where it actually dispatches. Native-only qwen3.8-flash below
|
|
1487
|
+
// carries the flag.
|
|
1413
1488
|
supportsVision: false,
|
|
1414
1489
|
supportsPromptCaching: true,
|
|
1415
1490
|
supportsTools: true,
|
|
1416
1491
|
inputPricePerMTok: 2,
|
|
1417
1492
|
outputPricePerMTok: 6,
|
|
1418
|
-
// Implicit context cache
|
|
1419
|
-
|
|
1493
|
+
// Implicit context cache. The ZH cache doc's standard 20% does NOT apply
|
|
1494
|
+
// here: that doc names qwen3.8-max as an exception and points at the
|
|
1495
|
+
// console, and the model page publishes the real Singapore rate — $0.25,
|
|
1496
|
+
// i.e. 12.5% of input (verified 2026-08-28 on
|
|
1497
|
+
// alibabacloud.com/help/en/model-studio/qwen3-8-max). Implicit creation is
|
|
1498
|
+
// not billed beyond input, so write stays 1× input.
|
|
1499
|
+
cacheReadPricePerMTok: 0.25,
|
|
1420
1500
|
cacheWritePricePerMTok: 2,
|
|
1421
1501
|
regions: ['us', 'cn'],
|
|
1422
1502
|
// US = DeepInfra (Qwen/Qwen3.8-Max), verified 2026-08-14 against
|
|
@@ -1430,6 +1510,49 @@ export const MODELS = [
|
|
|
1430
1510
|
// Not published by Alibaba — best-effort estimate.
|
|
1431
1511
|
knowledgeCutoff: '2026-04-01',
|
|
1432
1512
|
},
|
|
1513
|
+
{
|
|
1514
|
+
id: 'qwen3.8-flash',
|
|
1515
|
+
provider: 'alibaba',
|
|
1516
|
+
label: 'Qwen3.8 Flash',
|
|
1517
|
+
description: 'Alibaba cheap tier — 1M context, multimodal, hybrid thinking',
|
|
1518
|
+
// Model page: context window 1,000,000; max input 991,808 (983,616 in
|
|
1519
|
+
// thinking mode); max chain-of-thought 262,144; max output 131,072.
|
|
1520
|
+
contextWindow: 1_000_000,
|
|
1521
|
+
maxOutputTokens: 131_072,
|
|
1522
|
+
// Same hybrid-thinking mechanism as qwen3.8-max — the deep-thinking doc
|
|
1523
|
+
// lists the "Qwen3.8 Flash series" as hybrid with thinking ON by default,
|
|
1524
|
+
// and thinking_budget applies — so effort scales the budget identically.
|
|
1525
|
+
supportsThinking: true,
|
|
1526
|
+
thinkingBudgetTokens: 8_000,
|
|
1527
|
+
thinkingConfigurable: true,
|
|
1528
|
+
supportedEffortLevels: ['4K', '8K', '16K', '32K'],
|
|
1529
|
+
defaultEffortLevel: '8K',
|
|
1530
|
+
effortBudgetTokens: { '4K': 4000, '8K': 8000, '16K': 16000, '32K': 32000 },
|
|
1531
|
+
// Model page lists Image / Text / Video input. Unlike qwen3.8-max this one
|
|
1532
|
+
// has no US re-host to lose it to — it only ever dispatches to the native
|
|
1533
|
+
// host, which serves the multimodal surface.
|
|
1534
|
+
supportsVision: true,
|
|
1535
|
+
supportsPromptCaching: true,
|
|
1536
|
+
supportsTools: true,
|
|
1537
|
+
// Singapore International list card, published in USD on the model page:
|
|
1538
|
+
// $0.15 / $0.47, implicit cache $0.016. (Beijing is cheaper at
|
|
1539
|
+
// $0.113/$0.382/$0.014; the bond calls the international endpoint.)
|
|
1540
|
+
inputPricePerMTok: 0.15,
|
|
1541
|
+
outputPricePerMTok: 0.47,
|
|
1542
|
+
// Implicit context cache. Like qwen3.8-max this model is an explicit
|
|
1543
|
+
// exception to the ZH doc's standard 20%; the published rate is $0.016
|
|
1544
|
+
// (~10.7% of input). Creation is not billed beyond input → write = 1×.
|
|
1545
|
+
cacheReadPricePerMTok: 0.016,
|
|
1546
|
+
cacheWritePricePerMTok: 0.15,
|
|
1547
|
+
// No US re-host exists — DeepInfra serves Qwen3.8-Max, Qwen3.8-27B and
|
|
1548
|
+
// Qwen3.8-2.4T-A95B but returns "model not found" for Qwen/Qwen3.8-Flash
|
|
1549
|
+
// (checked 2026-08-28), so this is pinned to the native host and bills the
|
|
1550
|
+
// card above. ALIBABA_US_MODEL_MAP therefore needs no entry.
|
|
1551
|
+
regions: ['cn'],
|
|
1552
|
+
// Not published by Alibaba — best-effort estimate, same 3.8 generation as
|
|
1553
|
+
// qwen3.8-max.
|
|
1554
|
+
knowledgeCutoff: '2026-04-01',
|
|
1555
|
+
},
|
|
1433
1556
|
{
|
|
1434
1557
|
id: 'qwen3.7-max',
|
|
1435
1558
|
provider: 'alibaba',
|
|
@@ -1583,6 +1706,11 @@ export const MODELS = [
|
|
|
1583
1706
|
// US default: DeepInfra (zai-org/GLM-5.3-Flash) bills exactly the list
|
|
1584
1707
|
// card — $0.15/$0.50, cache read 0.2× = $0.03. Verified live 2026-08-27.
|
|
1585
1708
|
regions: ['us', 'cn'],
|
|
1709
|
+
// NB: DeepInfra returns NO cached-token usage (verified live 2026-08-28:
|
|
1710
|
+
// an identical ~2.6k-token prefix twice reported full input both passes,
|
|
1711
|
+
// no prompt_tokens_details) — the us cacheRead rate below is informational
|
|
1712
|
+
// only; metering always bills full input there. Native z.ai reports and
|
|
1713
|
+
// discounts cached tokens correctly.
|
|
1586
1714
|
regionPricing: {
|
|
1587
1715
|
us: { inputPricePerMTok: 0.15, outputPricePerMTok: 0.5, cacheReadPricePerMTok: 0.03 },
|
|
1588
1716
|
},
|
|
@@ -1614,6 +1742,11 @@ export const MODELS = [
|
|
|
1614
1742
|
cacheWritePricePerMTok: 1.4,
|
|
1615
1743
|
// US default (DeepInfra bills ~half native). Verified 2026-08-01.
|
|
1616
1744
|
regions: ['us', 'cn'],
|
|
1745
|
+
// NB: DeepInfra returns NO cached-token usage (verified live 2026-08-28:
|
|
1746
|
+
// an identical ~2.6k-token prefix twice reported full input both passes,
|
|
1747
|
+
// no prompt_tokens_details) — the us cacheRead rate below is informational
|
|
1748
|
+
// only; metering always bills full input there. Native z.ai reports and
|
|
1749
|
+
// discounts cached tokens correctly.
|
|
1617
1750
|
regionPricing: {
|
|
1618
1751
|
us: { inputPricePerMTok: 0.75, outputPricePerMTok: 2.4, cacheReadPricePerMTok: 0.14 },
|
|
1619
1752
|
},
|
|
@@ -1650,6 +1783,11 @@ export const MODELS = [
|
|
|
1650
1783
|
cacheWritePricePerMTok: 1,
|
|
1651
1784
|
// US default (DeepInfra bills below native). Verified 2026-08-01.
|
|
1652
1785
|
regions: ['us', 'cn'],
|
|
1786
|
+
// NB: DeepInfra returns NO cached-token usage (verified live 2026-08-28:
|
|
1787
|
+
// an identical ~2.6k-token prefix twice reported full input both passes,
|
|
1788
|
+
// no prompt_tokens_details) — the us cacheRead rate below is informational
|
|
1789
|
+
// only; metering always bills full input there. Native z.ai reports and
|
|
1790
|
+
// discounts cached tokens correctly.
|
|
1653
1791
|
regionPricing: {
|
|
1654
1792
|
us: { inputPricePerMTok: 0.6, outputPricePerMTok: 2.08, cacheReadPricePerMTok: 0.12 },
|
|
1655
1793
|
},
|
package/package.json
CHANGED