@molecule/api-resource-ai-models 1.2.6 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -3,7 +3,7 @@ AUTO-GENERATED — DO NOT EDIT THIS FILE.
3
3
  Generated by `mlcl sync-docs` from the package's src/index.ts JSDoc + mlcl/registry.json.
4
4
  Edits here are overwritten on the next commit (molecule's pre-commit hook regenerates).
5
5
  To change this document, edit the module-level JSDoc in src/index.ts.
6
- Generated: 2026-08-27T00:37:24.634Z
6
+ Generated: 2026-08-28T19:44:44.140Z
7
7
  -->
8
8
 
9
9
  # @molecule/api-resource-ai-models
@@ -728,6 +728,16 @@ All available AI models, grouped by provider, ordered from most to least capable
728
728
  To add or remove a model, edit this array. Both the server-side validation
729
729
  and the public discovery endpoint will update automatically.
730
730
 
731
+ EVERY entry field that shapes the outbound request (`webSearchToolType`,
732
+ `supportedEffortLevels`/`defaultEffortLevel`/`effortBudgetTokens`,
733
+ `maxOutputTokens`, `regions`) must hold for the model's ACTUAL serving
734
+ host(s) in every region — not just look right in a docs table. The
735
+ pre-commit hook enforces this with a LIVE Synthase-shaped probe of each
736
+ added/changed entry (molecule-dev `verify:model-dispatch --staged-catalog`);
737
+ a value the host rejects fails the whole request for every user
738
+ (glm-5.3-flash 2026-08-28: a `webSearchToolType` both zhipu hosts reject
739
+ made every Synthase turn 400/422 while the model itself worked fine).
740
+
731
741
  Effort is each model's OWN native value — there is no abstract scale (see
732
742
  {@link ModelDefinition.supportedEffortLevels}):
733
743
 
@@ -841,6 +851,16 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
841
851
  $1.65/$4.951), so nothing would ever select it — and carrying both would
842
852
  put two selectable Alibaba flagships in one family. Revisit only if Alibaba
843
853
  publishes it as a distinct first-party DashScope model id.)
854
+ (re-verified 2026-08-28 on the per-model pages, which publish USD rates per
855
+ region directly — a better source than the CNY pricing table: qwen3.8-max's
856
+ Singapore implicit-cache rate is $0.25, NOT the 20%/$0.40 the ZH cache doc's
857
+ standard table implies. That doc names qwen3.8-max, qwen3.8-flash and
858
+ qwen3.8-2.4t-a95b as exceptions and defers to the console; the model pages
859
+ are the console's figures. qwen3.7-max's $0.50 (20%) is confirmed unchanged.
860
+ qwen3.8-flash (2026-08-26) ADDED: $0.15/$0.47, implicit cache $0.016, 1M ctx
861
+ / 131,072 out, Image+Text+Video input, tools, hybrid thinking ON by default
862
+ with thinking_budget. It is cn-region-only — api.deepinfra.com has no
863
+ Qwen/Qwen3.8-Flash — so ALIBABA_US_MODEL_MAP needs no entry.)
844
864
  - Zhipu: https://docs.z.ai/guides/overview/pricing + docs.z.ai/guides/llm/
845
865
  glm-5.3 (verified 2026-08-26: glm-5.3 shipped 2026-08-14 and IS in the
846
866
  catalog — $1.40/$4.40, cached $0.26, i.e. glm-5.2's card unchanged, 1M ctx,
package/dist/models.d.ts CHANGED
@@ -14,6 +14,16 @@ import type { ModelDefinition } from './types.js';
14
14
  * To add or remove a model, edit this array. Both the server-side validation
15
15
  * and the public discovery endpoint will update automatically.
16
16
  *
17
+ * EVERY entry field that shapes the outbound request (`webSearchToolType`,
18
+ * `supportedEffortLevels`/`defaultEffortLevel`/`effortBudgetTokens`,
19
+ * `maxOutputTokens`, `regions`) must hold for the model's ACTUAL serving
20
+ * host(s) in every region — not just look right in a docs table. The
21
+ * pre-commit hook enforces this with a LIVE Synthase-shaped probe of each
22
+ * added/changed entry (molecule-dev `verify:model-dispatch --staged-catalog`);
23
+ * a value the host rejects fails the whole request for every user
24
+ * (glm-5.3-flash 2026-08-28: a `webSearchToolType` both zhipu hosts reject
25
+ * made every Synthase turn 400/422 while the model itself worked fine).
26
+ *
17
27
  * Effort is each model's OWN native value — there is no abstract scale (see
18
28
  * {@link ModelDefinition.supportedEffortLevels}):
19
29
  * - A model driven by a provider-native effort/level param lists its provider
@@ -125,6 +135,16 @@ import type { ModelDefinition } from './types.js';
125
135
  * $1.65/$4.951), so nothing would ever select it — and carrying both would
126
136
  * put two selectable Alibaba flagships in one family. Revisit only if Alibaba
127
137
  * publishes it as a distinct first-party DashScope model id.)
138
+ * (re-verified 2026-08-28 on the per-model pages, which publish USD rates per
139
+ * region directly — a better source than the CNY pricing table: qwen3.8-max's
140
+ * Singapore implicit-cache rate is $0.25, NOT the 20%/$0.40 the ZH cache doc's
141
+ * standard table implies. That doc names qwen3.8-max, qwen3.8-flash and
142
+ * qwen3.8-2.4t-a95b as exceptions and defers to the console; the model pages
143
+ * are the console's figures. qwen3.7-max's $0.50 (20%) is confirmed unchanged.
144
+ * qwen3.8-flash (2026-08-26) ADDED: $0.15/$0.47, implicit cache $0.016, 1M ctx
145
+ * / 131,072 out, Image+Text+Video input, tools, hybrid thinking ON by default
146
+ * with thinking_budget. It is cn-region-only — api.deepinfra.com has no
147
+ * Qwen/Qwen3.8-Flash — so ALIBABA_US_MODEL_MAP needs no entry.)
128
148
  * - Zhipu: https://docs.z.ai/guides/overview/pricing + docs.z.ai/guides/llm/
129
149
  * glm-5.3 (verified 2026-08-26: glm-5.3 shipped 2026-08-14 and IS in the
130
150
  * catalog — $1.40/$4.40, cached $0.26, i.e. glm-5.2's card unchanged, 1M ctx,
@@ -1 +1 @@
1
- {"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAyIG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAm/CnC,CAAA"}
1
+ {"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA6JG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAymDnC,CAAA"}
package/dist/models.js CHANGED
@@ -13,6 +13,16 @@
13
13
  * To add or remove a model, edit this array. Both the server-side validation
14
14
  * and the public discovery endpoint will update automatically.
15
15
  *
16
+ * EVERY entry field that shapes the outbound request (`webSearchToolType`,
17
+ * `supportedEffortLevels`/`defaultEffortLevel`/`effortBudgetTokens`,
18
+ * `maxOutputTokens`, `regions`) must hold for the model's ACTUAL serving
19
+ * host(s) in every region — not just look right in a docs table. The
20
+ * pre-commit hook enforces this with a LIVE Synthase-shaped probe of each
21
+ * added/changed entry (molecule-dev `verify:model-dispatch --staged-catalog`);
22
+ * a value the host rejects fails the whole request for every user
23
+ * (glm-5.3-flash 2026-08-28: a `webSearchToolType` both zhipu hosts reject
24
+ * made every Synthase turn 400/422 while the model itself worked fine).
25
+ *
16
26
  * Effort is each model's OWN native value — there is no abstract scale (see
17
27
  * {@link ModelDefinition.supportedEffortLevels}):
18
28
  * - A model driven by a provider-native effort/level param lists its provider
@@ -124,6 +134,16 @@
124
134
  * $1.65/$4.951), so nothing would ever select it — and carrying both would
125
135
  * put two selectable Alibaba flagships in one family. Revisit only if Alibaba
126
136
  * publishes it as a distinct first-party DashScope model id.)
137
+ * (re-verified 2026-08-28 on the per-model pages, which publish USD rates per
138
+ * region directly — a better source than the CNY pricing table: qwen3.8-max's
139
+ * Singapore implicit-cache rate is $0.25, NOT the 20%/$0.40 the ZH cache doc's
140
+ * standard table implies. That doc names qwen3.8-max, qwen3.8-flash and
141
+ * qwen3.8-2.4t-a95b as exceptions and defers to the console; the model pages
142
+ * are the console's figures. qwen3.7-max's $0.50 (20%) is confirmed unchanged.
143
+ * qwen3.8-flash (2026-08-26) ADDED: $0.15/$0.47, implicit cache $0.016, 1M ctx
144
+ * / 131,072 out, Image+Text+Video input, tools, hybrid thinking ON by default
145
+ * with thinking_budget. It is cn-region-only — api.deepinfra.com has no
146
+ * Qwen/Qwen3.8-Flash — so ALIBABA_US_MODEL_MAP needs no entry.)
127
147
  * - Zhipu: https://docs.z.ai/guides/overview/pricing + docs.z.ai/guides/llm/
128
148
  * glm-5.3 (verified 2026-08-26: glm-5.3 shipped 2026-08-14 and IS in the
129
149
  * catalog — $1.40/$4.40, cached $0.26, i.e. glm-5.2's card unchanged, 1M ctx,
@@ -174,7 +194,11 @@ export const MODELS = [
174
194
  supportsPromptCaching: true,
175
195
  supportsTools: true,
176
196
  webSearchToolType: 'web_search_20260209',
177
- codeExecutionToolType: 'code_execution_20250825',
197
+ // Current server-tool type per the API reference (verified 2026-08-28);
198
+ // 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
199
+ // not a tool type — sending it as one is a 400. Not currently sent by
200
+ // Synthase (request-shape.ts forwards only webSearchToolType).
201
+ codeExecutionToolType: 'code_execution_20260521',
178
202
  webFetchToolType: 'web_fetch_20260209',
179
203
  inputPricePerMTok: 10,
180
204
  outputPricePerMTok: 50,
@@ -214,7 +238,11 @@ export const MODELS = [
214
238
  supportsPromptCaching: true,
215
239
  supportsTools: true,
216
240
  webSearchToolType: 'web_search_20260209',
217
- codeExecutionToolType: 'code_execution_20250825',
241
+ // Current server-tool type per the API reference (verified 2026-08-28);
242
+ // 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
243
+ // not a tool type — sending it as one is a 400. Not currently sent by
244
+ // Synthase (request-shape.ts forwards only webSearchToolType).
245
+ codeExecutionToolType: 'code_execution_20260521',
218
246
  webFetchToolType: 'web_fetch_20260209',
219
247
  // Drop-in successor to Opus 4.8 at identical pricing.
220
248
  inputPricePerMTok: 5,
@@ -243,7 +271,11 @@ export const MODELS = [
243
271
  supportsPromptCaching: true,
244
272
  supportsTools: true,
245
273
  webSearchToolType: 'web_search_20260209',
246
- codeExecutionToolType: 'code_execution_20250825',
274
+ // Current server-tool type per the API reference (verified 2026-08-28);
275
+ // 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
276
+ // not a tool type — sending it as one is a 400. Not currently sent by
277
+ // Synthase (request-shape.ts forwards only webSearchToolType).
278
+ codeExecutionToolType: 'code_execution_20260521',
247
279
  webFetchToolType: 'web_fetch_20260209',
248
280
  inputPricePerMTok: 5,
249
281
  outputPricePerMTok: 25,
@@ -276,7 +308,11 @@ export const MODELS = [
276
308
  supportsPromptCaching: true,
277
309
  supportsTools: true,
278
310
  webSearchToolType: 'web_search_20260209',
279
- codeExecutionToolType: 'code_execution_20250825',
311
+ // Current server-tool type per the API reference (verified 2026-08-28);
312
+ // 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
313
+ // not a tool type — sending it as one is a 400. Not currently sent by
314
+ // Synthase (request-shape.ts forwards only webSearchToolType).
315
+ codeExecutionToolType: 'code_execution_20260521',
280
316
  webFetchToolType: 'web_fetch_20260209',
281
317
  // Standard pricing. Intro pricing ($2/$10) applies through 2026-08-31 —
282
318
  // billed here at standard so metering never under-charges; revisit after.
@@ -307,7 +343,11 @@ export const MODELS = [
307
343
  supportsPromptCaching: true,
308
344
  supportsTools: true,
309
345
  webSearchToolType: 'web_search_20260209',
310
- codeExecutionToolType: 'code_execution_20250825',
346
+ // Current server-tool type per the API reference (verified 2026-08-28);
347
+ // 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
348
+ // not a tool type — sending it as one is a 400. Not currently sent by
349
+ // Synthase (request-shape.ts forwards only webSearchToolType).
350
+ codeExecutionToolType: 'code_execution_20260521',
311
351
  webFetchToolType: 'web_fetch_20260209',
312
352
  inputPricePerMTok: 5,
313
353
  outputPricePerMTok: 25,
@@ -342,7 +382,11 @@ export const MODELS = [
342
382
  supportsPromptCaching: true,
343
383
  supportsTools: true,
344
384
  webSearchToolType: 'web_search_20260209',
345
- codeExecutionToolType: 'code_execution_20250825',
385
+ // Current server-tool type per the API reference (verified 2026-08-28);
386
+ // 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
387
+ // not a tool type — sending it as one is a 400. Not currently sent by
388
+ // Synthase (request-shape.ts forwards only webSearchToolType).
389
+ codeExecutionToolType: 'code_execution_20260521',
346
390
  webFetchToolType: 'web_fetch_20260209',
347
391
  inputPricePerMTok: 5,
348
392
  outputPricePerMTok: 25,
@@ -373,7 +417,11 @@ export const MODELS = [
373
417
  supportsPromptCaching: true,
374
418
  supportsTools: true,
375
419
  webSearchToolType: 'web_search_20260209',
376
- codeExecutionToolType: 'code_execution_20250825',
420
+ // Current server-tool type per the API reference (verified 2026-08-28);
421
+ // 'code_execution_20250825' was the BETA HEADER date (code-execution-2025-08-25),
422
+ // not a tool type — sending it as one is a 400. Not currently sent by
423
+ // Synthase (request-shape.ts forwards only webSearchToolType).
424
+ codeExecutionToolType: 'code_execution_20260521',
377
425
  webFetchToolType: 'web_fetch_20260209',
378
426
  inputPricePerMTok: 3,
379
427
  outputPricePerMTok: 15,
@@ -446,7 +494,11 @@ export const MODELS = [
446
494
  supportsTools: true,
447
495
  // Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
448
496
  toolsRequireReasoningOff: true,
449
- webSearchToolType: 'web_search',
497
+ // NO webSearchToolType: the OpenAI bond calls /v1/chat/completions, which
498
+ // has no web_search tool type (it is a Responses-API construct), and the
499
+ // bond deliberately forwards no server tools. Advertising one here surfaced
500
+ // web search in the system prompt while it could never work. Re-add when
501
+ // the bond moves to /v1/responses (verified 2026-08-28).
450
502
  codeExecutionToolType: 'code_interpreter',
451
503
  // LIST price. OpenAI ran a >20% PROMO from 2026-08-22 ($4/$20, cache read
452
504
  // $0.40, cache write $5) — "GPT-5.6 Sol's promotional pricing is available
@@ -479,7 +531,11 @@ export const MODELS = [
479
531
  supportsTools: true,
480
532
  // Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
481
533
  toolsRequireReasoningOff: true,
482
- webSearchToolType: 'web_search',
534
+ // NO webSearchToolType: the OpenAI bond calls /v1/chat/completions, which
535
+ // has no web_search tool type (it is a Responses-API construct), and the
536
+ // bond deliberately forwards no server tools. Advertising one here surfaced
537
+ // web search in the system prompt while it could never work. Re-add when
538
+ // the bond moves to /v1/responses (verified 2026-08-28).
483
539
  codeExecutionToolType: 'code_interpreter',
484
540
  // Repriced 2026-07-30 (20% cut from $2.50/$15).
485
541
  inputPricePerMTok: 2,
@@ -507,7 +563,11 @@ export const MODELS = [
507
563
  supportsTools: true,
508
564
  // Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
509
565
  toolsRequireReasoningOff: true,
510
- webSearchToolType: 'web_search',
566
+ // NO webSearchToolType: the OpenAI bond calls /v1/chat/completions, which
567
+ // has no web_search tool type (it is a Responses-API construct), and the
568
+ // bond deliberately forwards no server tools. Advertising one here surfaced
569
+ // web search in the system prompt while it could never work. Re-add when
570
+ // the bond moves to /v1/responses (verified 2026-08-28).
511
571
  codeExecutionToolType: 'code_interpreter',
512
572
  // Repriced 2026-07-30 (80% cut from $1/$6).
513
573
  inputPricePerMTok: 0.2,
@@ -542,7 +602,11 @@ export const MODELS = [
542
602
  supportsTools: true,
543
603
  // Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
544
604
  toolsRequireReasoningOff: true,
545
- webSearchToolType: 'web_search',
605
+ // NO webSearchToolType: the OpenAI bond calls /v1/chat/completions, which
606
+ // has no web_search tool type (it is a Responses-API construct), and the
607
+ // bond deliberately forwards no server tools. Advertising one here surfaced
608
+ // web search in the system prompt while it could never work. Re-add when
609
+ // the bond moves to /v1/responses (verified 2026-08-28).
546
610
  codeExecutionToolType: 'code_interpreter',
547
611
  inputPricePerMTok: 5,
548
612
  outputPricePerMTok: 30,
@@ -572,7 +636,11 @@ export const MODELS = [
572
636
  supportsTools: true,
573
637
  // Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
574
638
  toolsRequireReasoningOff: true,
575
- webSearchToolType: 'web_search',
639
+ // NO webSearchToolType: the OpenAI bond calls /v1/chat/completions, which
640
+ // has no web_search tool type (it is a Responses-API construct), and the
641
+ // bond deliberately forwards no server tools. Advertising one here surfaced
642
+ // web search in the system prompt while it could never work. Re-add when
643
+ // the bond moves to /v1/responses (verified 2026-08-28).
576
644
  codeExecutionToolType: 'code_interpreter',
577
645
  inputPricePerMTok: 2.5,
578
646
  outputPricePerMTok: 15,
@@ -604,7 +672,11 @@ export const MODELS = [
604
672
  supportsTools: true,
605
673
  // Tools + ANY reasoning is a 400 on /v1/chat/completions for this family.
606
674
  toolsRequireReasoningOff: true,
607
- webSearchToolType: 'web_search',
675
+ // NO webSearchToolType: the OpenAI bond calls /v1/chat/completions, which
676
+ // has no web_search tool type (it is a Responses-API construct), and the
677
+ // bond deliberately forwards no server tools. Advertising one here surfaced
678
+ // web search in the system prompt while it could never work. Re-add when
679
+ // the bond moves to /v1/responses (verified 2026-08-28).
608
680
  codeExecutionToolType: 'code_interpreter',
609
681
  inputPricePerMTok: 0.75,
610
682
  outputPricePerMTok: 4.5,
@@ -1407,16 +1479,24 @@ export const MODELS = [
1407
1479
  supportedEffortLevels: ['4K', '8K', '16K', '32K'],
1408
1480
  defaultEffortLevel: '8K',
1409
1481
  effortBudgetTokens: { '4K': 4000, '8K': 8000, '16K': 16000, '32K': 32000 },
1410
- // models.dev claims image+video input, but Alibaba's own model catalog
1411
- // lists qwen3.8-max under text generation (VL remains a separate line) —
1412
- // false until the provider's page says otherwise.
1482
+ // The provider's own model page DOES list Image/Text/Video input (checked
1483
+ // 2026-08-28, superseding the earlier "text generation only" reading of the
1484
+ // model catalog page). Kept false anyway: this model's DEFAULT region is the
1485
+ // DeepInfra US re-host, which serves it as `text-generation` — vision is not
1486
+ // offerable where it actually dispatches. Native-only qwen3.8-flash below
1487
+ // carries the flag.
1413
1488
  supportsVision: false,
1414
1489
  supportsPromptCaching: true,
1415
1490
  supportsTools: true,
1416
1491
  inputPricePerMTok: 2,
1417
1492
  outputPricePerMTok: 6,
1418
- // Implicit context cache: read = 20% of input, no write premium.
1419
- cacheReadPricePerMTok: 0.4,
1493
+ // Implicit context cache. The ZH cache doc's standard 20% does NOT apply
1494
+ // here: that doc names qwen3.8-max as an exception and points at the
1495
+ // console, and the model page publishes the real Singapore rate — $0.25,
1496
+ // i.e. 12.5% of input (verified 2026-08-28 on
1497
+ // alibabacloud.com/help/en/model-studio/qwen3-8-max). Implicit creation is
1498
+ // not billed beyond input, so write stays 1× input.
1499
+ cacheReadPricePerMTok: 0.25,
1420
1500
  cacheWritePricePerMTok: 2,
1421
1501
  regions: ['us', 'cn'],
1422
1502
  // US = DeepInfra (Qwen/Qwen3.8-Max), verified 2026-08-14 against
@@ -1430,6 +1510,49 @@ export const MODELS = [
1430
1510
  // Not published by Alibaba — best-effort estimate.
1431
1511
  knowledgeCutoff: '2026-04-01',
1432
1512
  },
1513
+ {
1514
+ id: 'qwen3.8-flash',
1515
+ provider: 'alibaba',
1516
+ label: 'Qwen3.8 Flash',
1517
+ description: 'Alibaba cheap tier — 1M context, multimodal, hybrid thinking',
1518
+ // Model page: context window 1,000,000; max input 991,808 (983,616 in
1519
+ // thinking mode); max chain-of-thought 262,144; max output 131,072.
1520
+ contextWindow: 1_000_000,
1521
+ maxOutputTokens: 131_072,
1522
+ // Same hybrid-thinking mechanism as qwen3.8-max — the deep-thinking doc
1523
+ // lists the "Qwen3.8 Flash series" as hybrid with thinking ON by default,
1524
+ // and thinking_budget applies — so effort scales the budget identically.
1525
+ supportsThinking: true,
1526
+ thinkingBudgetTokens: 8_000,
1527
+ thinkingConfigurable: true,
1528
+ supportedEffortLevels: ['4K', '8K', '16K', '32K'],
1529
+ defaultEffortLevel: '8K',
1530
+ effortBudgetTokens: { '4K': 4000, '8K': 8000, '16K': 16000, '32K': 32000 },
1531
+ // Model page lists Image / Text / Video input. Unlike qwen3.8-max this one
1532
+ // has no US re-host to lose it to — it only ever dispatches to the native
1533
+ // host, which serves the multimodal surface.
1534
+ supportsVision: true,
1535
+ supportsPromptCaching: true,
1536
+ supportsTools: true,
1537
+ // Singapore International list card, published in USD on the model page:
1538
+ // $0.15 / $0.47, implicit cache $0.016. (Beijing is cheaper at
1539
+ // $0.113/$0.382/$0.014; the bond calls the international endpoint.)
1540
+ inputPricePerMTok: 0.15,
1541
+ outputPricePerMTok: 0.47,
1542
+ // Implicit context cache. Like qwen3.8-max this model is an explicit
1543
+ // exception to the ZH doc's standard 20%; the published rate is $0.016
1544
+ // (~10.7% of input). Creation is not billed beyond input → write = 1×.
1545
+ cacheReadPricePerMTok: 0.016,
1546
+ cacheWritePricePerMTok: 0.15,
1547
+ // No US re-host exists — DeepInfra serves Qwen3.8-Max, Qwen3.8-27B and
1548
+ // Qwen3.8-2.4T-A95B but returns "model not found" for Qwen/Qwen3.8-Flash
1549
+ // (checked 2026-08-28), so this is pinned to the native host and bills the
1550
+ // card above. ALIBABA_US_MODEL_MAP therefore needs no entry.
1551
+ regions: ['cn'],
1552
+ // Not published by Alibaba — best-effort estimate, same 3.8 generation as
1553
+ // qwen3.8-max.
1554
+ knowledgeCutoff: '2026-04-01',
1555
+ },
1433
1556
  {
1434
1557
  id: 'qwen3.7-max',
1435
1558
  provider: 'alibaba',
@@ -1583,6 +1706,11 @@ export const MODELS = [
1583
1706
  // US default: DeepInfra (zai-org/GLM-5.3-Flash) bills exactly the list
1584
1707
  // card — $0.15/$0.50, cache read 0.2× = $0.03. Verified live 2026-08-27.
1585
1708
  regions: ['us', 'cn'],
1709
+ // NB: DeepInfra returns NO cached-token usage (verified live 2026-08-28:
1710
+ // an identical ~2.6k-token prefix twice reported full input both passes,
1711
+ // no prompt_tokens_details) — the us cacheRead rate below is informational
1712
+ // only; metering always bills full input there. Native z.ai reports and
1713
+ // discounts cached tokens correctly.
1586
1714
  regionPricing: {
1587
1715
  us: { inputPricePerMTok: 0.15, outputPricePerMTok: 0.5, cacheReadPricePerMTok: 0.03 },
1588
1716
  },
@@ -1614,6 +1742,11 @@ export const MODELS = [
1614
1742
  cacheWritePricePerMTok: 1.4,
1615
1743
  // US default (DeepInfra bills ~half native). Verified 2026-08-01.
1616
1744
  regions: ['us', 'cn'],
1745
+ // NB: DeepInfra returns NO cached-token usage (verified live 2026-08-28:
1746
+ // an identical ~2.6k-token prefix twice reported full input both passes,
1747
+ // no prompt_tokens_details) — the us cacheRead rate below is informational
1748
+ // only; metering always bills full input there. Native z.ai reports and
1749
+ // discounts cached tokens correctly.
1617
1750
  regionPricing: {
1618
1751
  us: { inputPricePerMTok: 0.75, outputPricePerMTok: 2.4, cacheReadPricePerMTok: 0.14 },
1619
1752
  },
@@ -1650,6 +1783,11 @@ export const MODELS = [
1650
1783
  cacheWritePricePerMTok: 1,
1651
1784
  // US default (DeepInfra bills below native). Verified 2026-08-01.
1652
1785
  regions: ['us', 'cn'],
1786
+ // NB: DeepInfra returns NO cached-token usage (verified live 2026-08-28:
1787
+ // an identical ~2.6k-token prefix twice reported full input both passes,
1788
+ // no prompt_tokens_details) — the us cacheRead rate below is informational
1789
+ // only; metering always bills full input there. Native z.ai reports and
1790
+ // discounts cached tokens correctly.
1653
1791
  regionPricing: {
1654
1792
  us: { inputPricePerMTok: 0.6, outputPricePerMTok: 2.08, cacheReadPricePerMTok: 0.12 },
1655
1793
  },
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@molecule/api-resource-ai-models",
3
- "version": "1.2.6",
3
+ "version": "1.3.0",
4
4
  "description": "AI model catalog — server-side source of truth plus an authentication-gated discovery endpoint",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",