@molecule/api-resource-ai-models 1.2.1 → 1.2.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -7
- package/dist/models.d.ts +16 -6
- package/dist/models.d.ts.map +1 -1
- package/dist/models.js +71 -78
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -3,7 +3,7 @@ AUTO-GENERATED — DO NOT EDIT THIS FILE.
|
|
|
3
3
|
Generated by `mlcl sync-docs` from the package's src/index.ts JSDoc + mlcl/registry.json.
|
|
4
4
|
Edits here are overwritten on the next commit (molecule's pre-commit hook regenerates).
|
|
5
5
|
To change this document, edit the module-level JSDoc in src/index.ts.
|
|
6
|
-
Generated: 2026-08-
|
|
6
|
+
Generated: 2026-08-18T03:40:46.175Z
|
|
7
7
|
-->
|
|
8
8
|
|
|
9
9
|
# @molecule/api-resource-ai-models
|
|
@@ -789,12 +789,12 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
789
789
|
grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
|
|
790
790
|
grok-code-fast-1 no longer listed — retires 2026-08-15)
|
|
791
791
|
- DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
|
|
792
|
-
2026-08-
|
|
793
|
-
never in this catalog. V4-Pro
|
|
794
|
-
|
|
795
|
-
|
|
796
|
-
|
|
797
|
-
|
|
792
|
+
2026-08-18; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
|
|
793
|
+
never in this catalog. The V4-Pro-GA price RISE effective 2026-08-16T16:00Z
|
|
794
|
+
has LANDED and is folded into the base fields, along with the peak-hour 2×
|
|
795
|
+
the same card introduced. The rate card gives the peak windows as
|
|
796
|
+
"01:00 - 04:00 and 06:00 - 10:00 UTC (all other hours are off-peak)" with
|
|
797
|
+
no day qualifier — DAILY, as `peakPricing` models them.)
|
|
798
798
|
- Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
|
|
799
799
|
the US re-host (kimi-k3 flagship 2026-07-16
|
|
800
800
|
— 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
|
|
@@ -817,6 +817,16 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
817
817
|
125%) — the EN edition of that doc simply lags (zero qwen3.8 mentions),
|
|
818
818
|
which an earlier pass misread as "excepted/console-only". qwen3.7-max
|
|
819
819
|
still runs its 50%-off promo — billed here at list, $2.50/$7.50)
|
|
820
|
+
(re-verified 2026-08-15 against api.deepinfra.com/models/: qwen3.8-max's US
|
|
821
|
+
re-host rates are unchanged — $1.65/$4.951, cache read 0.1248× input =
|
|
822
|
+
$0.206. DeepInfra also began serving `Qwen/Qwen3.8-2.4T-A95B` on
|
|
823
|
+
2026-08-12; by DeepInfra's own description it is the OPEN-WEIGHT variant of
|
|
824
|
+
Qwen3.8 Max (2.4T MoE, 95B active), 262,144 ctx / 131,072 out, $2/$6 with
|
|
825
|
+
cache read 0.1× input. NOT added: it is the same tier as qwen3.8-max, which
|
|
826
|
+
the catalog already offers, and it costs MORE on that very host ($2/$6 vs
|
|
827
|
+
$1.65/$4.951), so nothing would ever select it — and carrying both would
|
|
828
|
+
put two selectable Alibaba flagships in one family. Revisit only if Alibaba
|
|
829
|
+
publishes it as a distinct first-party DashScope model id.)
|
|
820
830
|
- Zhipu: https://docs.z.ai/guides/overview/pricing (unchanged; glm-5.2 is
|
|
821
831
|
the newest — "GLM-5.3/5.5" rumors have no released ids as of 2026-07-28)
|
|
822
832
|
|
package/dist/models.d.ts
CHANGED
|
@@ -73,12 +73,12 @@ import type { ModelDefinition } from './types.js';
|
|
|
73
73
|
* grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
|
|
74
74
|
* grok-code-fast-1 no longer listed — retires 2026-08-15)
|
|
75
75
|
* - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
|
|
76
|
-
* 2026-08-
|
|
77
|
-
* never in this catalog. V4-Pro
|
|
78
|
-
*
|
|
79
|
-
*
|
|
80
|
-
*
|
|
81
|
-
*
|
|
76
|
+
* 2026-08-18; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
|
|
77
|
+
* never in this catalog. The V4-Pro-GA price RISE effective 2026-08-16T16:00Z
|
|
78
|
+
* has LANDED and is folded into the base fields, along with the peak-hour 2×
|
|
79
|
+
* the same card introduced. The rate card gives the peak windows as
|
|
80
|
+
* "01:00 - 04:00 and 06:00 - 10:00 UTC (all other hours are off-peak)" with
|
|
81
|
+
* no day qualifier — DAILY, as `peakPricing` models them.)
|
|
82
82
|
* - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
|
|
83
83
|
* the US re-host (kimi-k3 flagship 2026-07-16
|
|
84
84
|
* — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
|
|
@@ -101,6 +101,16 @@ import type { ModelDefinition } from './types.js';
|
|
|
101
101
|
* 125%) — the EN edition of that doc simply lags (zero qwen3.8 mentions),
|
|
102
102
|
* which an earlier pass misread as "excepted/console-only". qwen3.7-max
|
|
103
103
|
* still runs its 50%-off promo — billed here at list, $2.50/$7.50)
|
|
104
|
+
* (re-verified 2026-08-15 against api.deepinfra.com/models/: qwen3.8-max's US
|
|
105
|
+
* re-host rates are unchanged — $1.65/$4.951, cache read 0.1248× input =
|
|
106
|
+
* $0.206. DeepInfra also began serving `Qwen/Qwen3.8-2.4T-A95B` on
|
|
107
|
+
* 2026-08-12; by DeepInfra's own description it is the OPEN-WEIGHT variant of
|
|
108
|
+
* Qwen3.8 Max (2.4T MoE, 95B active), 262,144 ctx / 131,072 out, $2/$6 with
|
|
109
|
+
* cache read 0.1× input. NOT added: it is the same tier as qwen3.8-max, which
|
|
110
|
+
* the catalog already offers, and it costs MORE on that very host ($2/$6 vs
|
|
111
|
+
* $1.65/$4.951), so nothing would ever select it — and carrying both would
|
|
112
|
+
* put two selectable Alibaba flagships in one family. Revisit only if Alibaba
|
|
113
|
+
* publishes it as a distinct first-party DashScope model id.)
|
|
104
114
|
* - Zhipu: https://docs.z.ai/guides/overview/pricing (unchanged; glm-5.2 is
|
|
105
115
|
* the newest — "GLM-5.3/5.5" rumors have no released ids as of 2026-07-28)
|
|
106
116
|
*
|
package/dist/models.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD
|
|
1
|
+
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA6GG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAi5CnC,CAAA"}
|
package/dist/models.js
CHANGED
|
@@ -72,12 +72,12 @@
|
|
|
72
72
|
* grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
|
|
73
73
|
* grok-code-fast-1 no longer listed — retires 2026-08-15)
|
|
74
74
|
* - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
|
|
75
|
-
* 2026-08-
|
|
76
|
-
* never in this catalog. V4-Pro
|
|
77
|
-
*
|
|
78
|
-
*
|
|
79
|
-
*
|
|
80
|
-
*
|
|
75
|
+
* 2026-08-18; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
|
|
76
|
+
* never in this catalog. The V4-Pro-GA price RISE effective 2026-08-16T16:00Z
|
|
77
|
+
* has LANDED and is folded into the base fields, along with the peak-hour 2×
|
|
78
|
+
* the same card introduced. The rate card gives the peak windows as
|
|
79
|
+
* "01:00 - 04:00 and 06:00 - 10:00 UTC (all other hours are off-peak)" with
|
|
80
|
+
* no day qualifier — DAILY, as `peakPricing` models them.)
|
|
81
81
|
* - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
|
|
82
82
|
* the US re-host (kimi-k3 flagship 2026-07-16
|
|
83
83
|
* — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
|
|
@@ -100,6 +100,16 @@
|
|
|
100
100
|
* 125%) — the EN edition of that doc simply lags (zero qwen3.8 mentions),
|
|
101
101
|
* which an earlier pass misread as "excepted/console-only". qwen3.7-max
|
|
102
102
|
* still runs its 50%-off promo — billed here at list, $2.50/$7.50)
|
|
103
|
+
* (re-verified 2026-08-15 against api.deepinfra.com/models/: qwen3.8-max's US
|
|
104
|
+
* re-host rates are unchanged — $1.65/$4.951, cache read 0.1248× input =
|
|
105
|
+
* $0.206. DeepInfra also began serving `Qwen/Qwen3.8-2.4T-A95B` on
|
|
106
|
+
* 2026-08-12; by DeepInfra's own description it is the OPEN-WEIGHT variant of
|
|
107
|
+
* Qwen3.8 Max (2.4T MoE, 95B active), 262,144 ctx / 131,072 out, $2/$6 with
|
|
108
|
+
* cache read 0.1× input. NOT added: it is the same tier as qwen3.8-max, which
|
|
109
|
+
* the catalog already offers, and it costs MORE on that very host ($2/$6 vs
|
|
110
|
+
* $1.65/$4.951), so nothing would ever select it — and carrying both would
|
|
111
|
+
* put two selectable Alibaba flagships in one family. Revisit only if Alibaba
|
|
112
|
+
* publishes it as a distinct first-party DashScope model id.)
|
|
103
113
|
* - Zhipu: https://docs.z.ai/guides/overview/pricing (unchanged; glm-5.2 is
|
|
104
114
|
* the newest — "GLM-5.3/5.5" rumors have no released ids as of 2026-07-28)
|
|
105
115
|
*
|
|
@@ -886,35 +896,31 @@ export const MODELS = [
|
|
|
886
896
|
// DeepSeek
|
|
887
897
|
// Verified: https://api-docs.deepseek.com/quick_start/pricing
|
|
888
898
|
// https://api-docs.deepseek.com/guides/thinking_mode
|
|
889
|
-
// https://api-docs.deepseek.com/updates/ (2026-08-
|
|
899
|
+
// https://api-docs.deepseek.com/updates/ (2026-08-18)
|
|
890
900
|
// 2026-08-13: V4-Pro GA — and with it the price rise that the "coming soon"
|
|
891
|
-
// note below had been waiting on. It
|
|
892
|
-
//
|
|
893
|
-
//
|
|
894
|
-
//
|
|
895
|
-
//
|
|
896
|
-
//
|
|
897
|
-
// come back here rather than as flat rates.
|
|
901
|
+
// note below had been waiting on. It was staged as `scheduledPricing`
|
|
902
|
+
// effective 2026-08-16T16:00Z; that instant has PASSED and the rates are now
|
|
903
|
+
// folded into the base fields, re-verified 2026-08-18 against the live rate
|
|
904
|
+
// card. The card is off-peak/peak with peak exactly 2× off-peak, so it maps
|
|
905
|
+
// onto base rates + `peakPricing` multiplier 2 — which is why the peak
|
|
906
|
+
// windows removed in July come back here rather than as flat rates.
|
|
898
907
|
// pro off-peak 0.66 / 1.98, cache hit 0.022 (peak 1.32 / 3.96 / 0.044)
|
|
899
908
|
// flash off-peak 0.22 / 0.66, cache hit 0.007 (peak 0.44 / 1.32 / 0.014)
|
|
900
|
-
// Cache HITS
|
|
901
|
-
// (12.1×) at peak — and agentic input is ~94% cache hits, so effective
|
|
902
|
-
// cost
|
|
903
|
-
// (flash is `freeTier`, pro is the free-tier planner)
|
|
904
|
-
//
|
|
905
|
-
// 1. WEEKDAYS OR DAILY. The rate card says only "Peak hours
|
|
906
|
-
// 04:00 and 06:00 - 10:00 UTC (all other hours are
|
|
907
|
-
//
|
|
908
|
-
//
|
|
909
|
-
//
|
|
910
|
-
//
|
|
911
|
-
//
|
|
912
|
-
//
|
|
913
|
-
//
|
|
914
|
-
//
|
|
915
|
-
// BELOW CN's new off-peak on both input and output — CN wins only on
|
|
916
|
-
// cache reads. Re-derive against measured cache-hit ratios before
|
|
917
|
-
// leaving the default where it is.
|
|
909
|
+
// Cache HITS were the real move — pro 0.003625 → 0.022 (6.1×) off-peak,
|
|
910
|
+
// 0.044 (12.1×) at peak — and agentic input is ~94% cache hits, so effective
|
|
911
|
+
// input cost rose far more than the list prices suggest. Both are free-tier
|
|
912
|
+
// models (flash is `freeTier`, pro is the free-tier planner).
|
|
913
|
+
// The two post-landing re-verifications, both settled 2026-08-18:
|
|
914
|
+
// 1. WEEKDAYS OR DAILY — DAILY. The rate card still says only "Peak hours
|
|
915
|
+
// are 01:00 - 04:00 and 06:00 - 10:00 UTC (all other hours are
|
|
916
|
+
// off-peak)" with no day qualifier (press coverage had described them as
|
|
917
|
+
// weekday-only). `peakPricing` has no day-of-week concept, so if the
|
|
918
|
+
// provider ever qualifies these by day this over-bills every weekend
|
|
919
|
+
// peak window and needs the FIELD extended, not the numbers nudged.
|
|
920
|
+
// 2. THE CN-VS-US DEFAULT — settled per model, and they differ. Post-rise
|
|
921
|
+
// DeepInfra's US flash rates (0.08/0.18/0.016) are BELOW CN's off-peak
|
|
922
|
+
// on both input and output, so Flash defaults US; Pro's US re-host still
|
|
923
|
+
// bills ~2.3× its native card, so Pro stays CN. See each entry.
|
|
918
924
|
// 2026-07-31: DeepSeek-V4-Flash OFFICIAL API launched in public beta — the
|
|
919
925
|
// SAME `deepseek-v4-flash` id now serves the re-post-trained 0731 build
|
|
920
926
|
// (same architecture/size; much stronger agent benchmarks — beats
|
|
@@ -944,11 +950,13 @@ export const MODELS = [
|
|
|
944
950
|
supportsVision: false,
|
|
945
951
|
supportsPromptCaching: true,
|
|
946
952
|
supportsTools: true,
|
|
947
|
-
|
|
948
|
-
|
|
953
|
+
// Off-peak rates; peak is `peakPricing.multiplier` × these (see below).
|
|
954
|
+
inputPricePerMTok: 0.66,
|
|
955
|
+
outputPricePerMTok: 1.98,
|
|
949
956
|
// DeepSeek automatic context cache: absolute cache-hit price ($/M).
|
|
950
|
-
cacheReadPricePerMTok: 0.
|
|
951
|
-
|
|
957
|
+
cacheReadPricePerMTok: 0.022,
|
|
958
|
+
// DeepSeek charges no cache-write premium — write bills at input.
|
|
959
|
+
cacheWritePricePerMTok: 0.66,
|
|
952
960
|
// Native-China DEFAULT (owner decision 2026-08-01, re-derived 2026-08-14):
|
|
953
961
|
// the US re-host (DeepInfra) bills ~3× list and ~28× cache reads, and
|
|
954
962
|
// agentic input is ~94% cache hits, so US processing ran ~5.7× native on
|
|
@@ -972,28 +980,18 @@ export const MODELS = [
|
|
|
972
980
|
regionPricing: {
|
|
973
981
|
us: { inputPricePerMTok: 1.3, outputPricePerMTok: 2.6, cacheReadPricePerMTok: 0.1 },
|
|
974
982
|
},
|
|
975
|
-
// The peak-hour 2× surcharge is
|
|
976
|
-
//
|
|
977
|
-
//
|
|
978
|
-
//
|
|
979
|
-
//
|
|
980
|
-
//
|
|
981
|
-
|
|
982
|
-
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
|
|
986
|
-
|
|
987
|
-
// DeepSeek charges no cache-write premium — write bills at input.
|
|
988
|
-
cacheWritePricePerMTok: 0.66,
|
|
989
|
-
peakPricing: {
|
|
990
|
-
windows: [
|
|
991
|
-
{ startMinuteUtc: 60, endMinuteUtc: 240 },
|
|
992
|
-
{ startMinuteUtc: 360, endMinuteUtc: 600 },
|
|
993
|
-
],
|
|
994
|
-
multiplier: 2,
|
|
995
|
-
},
|
|
996
|
-
source: 'https://api-docs.deepseek.com/quick_start/pricing/',
|
|
983
|
+
// The peak-hour 2× surcharge is ON the rate card and LIVE since
|
|
984
|
+
// 2026-08-16T16:00Z. Peak = 01:00-04:00 and 06:00-10:00 UTC (Beijing
|
|
985
|
+
// business hours), daily, at exactly 2× the off-peak rates above. Windows
|
|
986
|
+
// like these were once pre-wired ahead of the card and over-billed every
|
|
987
|
+
// peak-window turn for weeks — hence the rule that they only exist here
|
|
988
|
+
// once the provider's own page shows them, which it now does.
|
|
989
|
+
peakPricing: {
|
|
990
|
+
windows: [
|
|
991
|
+
{ startMinuteUtc: 60, endMinuteUtc: 240 },
|
|
992
|
+
{ startMinuteUtc: 360, endMinuteUtc: 600 },
|
|
993
|
+
],
|
|
994
|
+
multiplier: 2,
|
|
997
995
|
},
|
|
998
996
|
// Not published by DeepSeek — best-effort estimate.
|
|
999
997
|
knowledgeCutoff: '2025-07-01',
|
|
@@ -1017,11 +1015,13 @@ export const MODELS = [
|
|
|
1017
1015
|
// executor — the model the IDE picks when none is chosen. Exactly one model
|
|
1018
1016
|
// in this catalog may carry freeTier (enforced by lookup.test.ts).
|
|
1019
1017
|
freeTier: true,
|
|
1020
|
-
|
|
1021
|
-
|
|
1018
|
+
// Off-peak rates; peak is `peakPricing.multiplier` × these (see below).
|
|
1019
|
+
inputPricePerMTok: 0.22,
|
|
1020
|
+
outputPricePerMTok: 0.66,
|
|
1022
1021
|
// DeepSeek automatic context cache: absolute cache-hit price ($/M).
|
|
1023
|
-
cacheReadPricePerMTok: 0.
|
|
1024
|
-
|
|
1022
|
+
cacheReadPricePerMTok: 0.007,
|
|
1023
|
+
// DeepSeek charges no cache-write premium — write bills at input.
|
|
1024
|
+
cacheWritePricePerMTok: 0.22,
|
|
1025
1025
|
// US (DeepInfra) DEFAULT as of 2026-08-16 — flipped from CN when DeepSeek's
|
|
1026
1026
|
// rise landed (owner decision 2026-08-14). CN was cheaper on real traffic
|
|
1027
1027
|
// only because of its cache reads; the rise takes those from $0.0028 to
|
|
@@ -1043,22 +1043,15 @@ export const MODELS = [
|
|
|
1043
1043
|
regionPricing: {
|
|
1044
1044
|
us: { inputPricePerMTok: 0.08, outputPricePerMTok: 0.18, cacheReadPricePerMTok: 0.016 },
|
|
1045
1045
|
},
|
|
1046
|
-
// Peak-hour surcharge
|
|
1047
|
-
|
|
1048
|
-
|
|
1049
|
-
|
|
1050
|
-
|
|
1051
|
-
|
|
1052
|
-
|
|
1053
|
-
|
|
1054
|
-
|
|
1055
|
-
windows: [
|
|
1056
|
-
{ startMinuteUtc: 60, endMinuteUtc: 240 },
|
|
1057
|
-
{ startMinuteUtc: 360, endMinuteUtc: 600 },
|
|
1058
|
-
],
|
|
1059
|
-
multiplier: 2,
|
|
1060
|
-
},
|
|
1061
|
-
source: 'https://api-docs.deepseek.com/quick_start/pricing/',
|
|
1046
|
+
// Peak-hour surcharge live since 2026-08-16T16:00Z (see deepseek-v4-pro).
|
|
1047
|
+
// It applies to the NATIVE CN card only — this model defaults to the US
|
|
1048
|
+
// re-host, which is flat, so most turns never take it.
|
|
1049
|
+
peakPricing: {
|
|
1050
|
+
windows: [
|
|
1051
|
+
{ startMinuteUtc: 60, endMinuteUtc: 240 },
|
|
1052
|
+
{ startMinuteUtc: 360, endMinuteUtc: 600 },
|
|
1053
|
+
],
|
|
1054
|
+
multiplier: 2,
|
|
1062
1055
|
},
|
|
1063
1056
|
// Not published by DeepSeek — best-effort estimate.
|
|
1064
1057
|
knowledgeCutoff: '2025-07-01',
|
package/package.json
CHANGED