@molecule/api-resource-ai-models 1.5.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -11
- package/dist/models.d.ts +27 -10
- package/dist/models.d.ts.map +1 -1
- package/dist/models.js +190 -36
- package/package.json +2 -2
package/README.md
CHANGED
|
@@ -3,7 +3,7 @@ AUTO-GENERATED — DO NOT EDIT THIS FILE.
|
|
|
3
3
|
Generated by `mlcl sync-docs` from the package's src/index.ts JSDoc + mlcl/registry.json.
|
|
4
4
|
Edits here are overwritten on the next commit (molecule's pre-commit hook regenerates).
|
|
5
5
|
To change this document, edit the module-level JSDoc in src/index.ts.
|
|
6
|
-
Generated: 2026-09-
|
|
6
|
+
Generated: 2026-09-10T11:48:36.810Z
|
|
7
7
|
-->
|
|
8
8
|
|
|
9
9
|
# @molecule/api-resource-ai-models
|
|
@@ -860,17 +860,26 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
860
860
|
not modeled; reasoning_effort low|medium|high default high, image input;
|
|
861
861
|
grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
|
|
862
862
|
grok-code-fast-1 no longer listed — retires 2026-08-15)
|
|
863
|
-
- DeepSeek: https://api-docs.deepseek.com/quick_start/pricing
|
|
864
|
-
2026-
|
|
865
|
-
never in this catalog. The
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
|
|
870
|
-
|
|
863
|
+
- DeepSeek: https://api-docs.deepseek.com/quick_start/pricing +
|
|
864
|
+
/updates/ (verified 2026-09-10; legacy deepseek-chat/-reasoner ids fully
|
|
865
|
+
retired 2026-07-24 — never in this catalog. The 2026-09-10 re-read caught
|
|
866
|
+
the V4.1-Flash release DAY-OF: new evergreen id `deepseek-flash` at
|
|
867
|
+
off-peak miss $0.15 / hit $0.003 / out $0.6 (peak ×2, same Mon-Fri UTC
|
|
868
|
+
windows, 1M ctx / 384K out), the V4 flash + vision-exp ids RETIRED with
|
|
869
|
+
their names "temporarily routed" to V4.1 Flash at Flash prices (the
|
|
870
|
+
`deepseek-v4-flash` entry is repriced to that card), and `deepseek-v4-pro`
|
|
871
|
+
announced to route to V4.1 Flash at Flash prices from 2026-09-14 12:00
|
|
872
|
+
Beijing (staged as `scheduledPricing` there); the pro card itself is
|
|
873
|
+
unchanged (V4-Pro-0813, 0.66/1.98/0.022 off-peak). The card qualifies the
|
|
874
|
+
peak windows BY DAY — "01:00 - 04:00 and 06:00 - 10:00 UTC, Monday through
|
|
875
|
+
Friday (all other hours are off-peak)" — so the windows are
|
|
871
876
|
`daysOfWeekUtc`-restricted rather than daily. `deepseek-v4-flash-vision-exp`
|
|
872
|
-
|
|
873
|
-
|
|
877
|
+
is retired with the rest of V4 flash and stays deliberately uncatalogued.
|
|
878
|
+
The US re-host rates (DeepInfra) are unchanged from the 2026-09-06 read —
|
|
879
|
+
`DeepSeek-V4-Flash-0731` at $0.06/$0.18, cache read $0.015; DeepInfra also
|
|
880
|
+
listed `deepseek-ai/DeepSeek-V4.1-Flash` on 2026-09-10 at $0.30/$1.20
|
|
881
|
+
(cache read 0.02× input) — 2× native, not wired. See each entry's
|
|
882
|
+
`regionPricing`.)
|
|
874
883
|
- Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
|
|
875
884
|
the US re-host (kimi-k3 flagship 2026-07-16
|
|
876
885
|
— 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
|
|
@@ -878,6 +887,9 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
878
887
|
constraint that kept kimi-k2.7-code out. BOTH are now in the catalog: the
|
|
879
888
|
moonshot bond gained preserved thinking (reasoning replayed through tool
|
|
880
889
|
loops), so kimi-k3 is the Moonshot pick.)
|
|
890
|
+
(re-verified 2026-09-10 on platform.kimi.ai/docs/models: kimi-k2.5 was
|
|
891
|
+
officially discontinued 2026-08-31 — calls 404 — so its entry is now
|
|
892
|
+
`disabled: true`. kimi-k3 / kimi-k2.6 / kimi-k2.7-code all still Active.)
|
|
881
893
|
- MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
|
|
882
894
|
minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
|
|
883
895
|
- Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
|
|
@@ -929,6 +941,11 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
929
941
|
DeepInfra re-host zai-org/GLM-5.3-Flash, which bills exactly the list card
|
|
930
942
|
($0.15/$0.50, cache read 0.2x = $0.03 — verified live 2026-08-27), mapped
|
|
931
943
|
in molecule-dev's ZHIPU_US_MODEL_MAP.)
|
|
944
|
+
(re-verified 2026-09-10 on https://docs.z.ai/guides/overview/pricing: the
|
|
945
|
+
promo ended on schedule 2026-09-09 24:00 UTC+8 and the page prints the list
|
|
946
|
+
card again — $0.15/$0.50, cached input $0.03, no promo footnote. models.dev
|
|
947
|
+
still carries the expired promo rate, so the KNOWN_DIVERGENCES entry now
|
|
948
|
+
runs to a fixed 2026-12-09 re-verify date.)
|
|
932
949
|
|
|
933
950
|
Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
|
|
934
951
|
where the provider doesn't publish one; the provider sources above verify
|
package/dist/models.d.ts
CHANGED
|
@@ -128,17 +128,26 @@ import type { ModelDefinition } from './types.js';
|
|
|
128
128
|
* not modeled; reasoning_effort low|medium|high default high, image input;
|
|
129
129
|
* grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
|
|
130
130
|
* grok-code-fast-1 no longer listed — retires 2026-08-15)
|
|
131
|
-
* - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing
|
|
132
|
-
* 2026-
|
|
133
|
-
* never in this catalog. The
|
|
134
|
-
*
|
|
135
|
-
*
|
|
136
|
-
*
|
|
137
|
-
*
|
|
138
|
-
*
|
|
131
|
+
* - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing +
|
|
132
|
+
* /updates/ (verified 2026-09-10; legacy deepseek-chat/-reasoner ids fully
|
|
133
|
+
* retired 2026-07-24 — never in this catalog. The 2026-09-10 re-read caught
|
|
134
|
+
* the V4.1-Flash release DAY-OF: new evergreen id `deepseek-flash` at
|
|
135
|
+
* off-peak miss $0.15 / hit $0.003 / out $0.6 (peak ×2, same Mon-Fri UTC
|
|
136
|
+
* windows, 1M ctx / 384K out), the V4 flash + vision-exp ids RETIRED with
|
|
137
|
+
* their names "temporarily routed" to V4.1 Flash at Flash prices (the
|
|
138
|
+
* `deepseek-v4-flash` entry is repriced to that card), and `deepseek-v4-pro`
|
|
139
|
+
* announced to route to V4.1 Flash at Flash prices from 2026-09-14 12:00
|
|
140
|
+
* Beijing (staged as `scheduledPricing` there); the pro card itself is
|
|
141
|
+
* unchanged (V4-Pro-0813, 0.66/1.98/0.022 off-peak). The card qualifies the
|
|
142
|
+
* peak windows BY DAY — "01:00 - 04:00 and 06:00 - 10:00 UTC, Monday through
|
|
143
|
+
* Friday (all other hours are off-peak)" — so the windows are
|
|
139
144
|
* `daysOfWeekUtc`-restricted rather than daily. `deepseek-v4-flash-vision-exp`
|
|
140
|
-
*
|
|
141
|
-
*
|
|
145
|
+
* is retired with the rest of V4 flash and stays deliberately uncatalogued.
|
|
146
|
+
* The US re-host rates (DeepInfra) are unchanged from the 2026-09-06 read —
|
|
147
|
+
* `DeepSeek-V4-Flash-0731` at $0.06/$0.18, cache read $0.015; DeepInfra also
|
|
148
|
+
* listed `deepseek-ai/DeepSeek-V4.1-Flash` on 2026-09-10 at $0.30/$1.20
|
|
149
|
+
* (cache read 0.02× input) — 2× native, not wired. See each entry's
|
|
150
|
+
* `regionPricing`.)
|
|
142
151
|
* - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
|
|
143
152
|
* the US re-host (kimi-k3 flagship 2026-07-16
|
|
144
153
|
* — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
|
|
@@ -146,6 +155,9 @@ import type { ModelDefinition } from './types.js';
|
|
|
146
155
|
* constraint that kept kimi-k2.7-code out. BOTH are now in the catalog: the
|
|
147
156
|
* moonshot bond gained preserved thinking (reasoning replayed through tool
|
|
148
157
|
* loops), so kimi-k3 is the Moonshot pick.)
|
|
158
|
+
* (re-verified 2026-09-10 on platform.kimi.ai/docs/models: kimi-k2.5 was
|
|
159
|
+
* officially discontinued 2026-08-31 — calls 404 — so its entry is now
|
|
160
|
+
* `disabled: true`. kimi-k3 / kimi-k2.6 / kimi-k2.7-code all still Active.)
|
|
149
161
|
* - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
|
|
150
162
|
* minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
|
|
151
163
|
* - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
|
|
@@ -197,6 +209,11 @@ import type { ModelDefinition } from './types.js';
|
|
|
197
209
|
* DeepInfra re-host zai-org/GLM-5.3-Flash, which bills exactly the list card
|
|
198
210
|
* ($0.15/$0.50, cache read 0.2x = $0.03 — verified live 2026-08-27), mapped
|
|
199
211
|
* in molecule-dev's ZHIPU_US_MODEL_MAP.)
|
|
212
|
+
* (re-verified 2026-09-10 on https://docs.z.ai/guides/overview/pricing: the
|
|
213
|
+
* promo ended on schedule 2026-09-09 24:00 UTC+8 and the page prints the list
|
|
214
|
+
* card again — $0.15/$0.50, cached input $0.03, no promo footnote. models.dev
|
|
215
|
+
* still carries the expired promo rate, so the KNOWN_DIVERGENCES entry now
|
|
216
|
+
* runs to a fixed 2026-12-09 re-verify date.)
|
|
200
217
|
*
|
|
201
218
|
* Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
|
|
202
219
|
* where the provider doesn't publish one; the provider sources above verify
|
package/dist/models.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAQjD
|
|
1
|
+
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAQjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkNG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAi1DnC,CAAA"}
|
package/dist/models.js
CHANGED
|
@@ -132,17 +132,26 @@ const WEEKDAYS_UTC = [1, 2, 3, 4, 5];
|
|
|
132
132
|
* not modeled; reasoning_effort low|medium|high default high, image input;
|
|
133
133
|
* grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
|
|
134
134
|
* grok-code-fast-1 no longer listed — retires 2026-08-15)
|
|
135
|
-
* - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing
|
|
136
|
-
* 2026-
|
|
137
|
-
* never in this catalog. The
|
|
138
|
-
*
|
|
139
|
-
*
|
|
140
|
-
*
|
|
141
|
-
*
|
|
142
|
-
*
|
|
135
|
+
* - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing +
|
|
136
|
+
* /updates/ (verified 2026-09-10; legacy deepseek-chat/-reasoner ids fully
|
|
137
|
+
* retired 2026-07-24 — never in this catalog. The 2026-09-10 re-read caught
|
|
138
|
+
* the V4.1-Flash release DAY-OF: new evergreen id `deepseek-flash` at
|
|
139
|
+
* off-peak miss $0.15 / hit $0.003 / out $0.6 (peak ×2, same Mon-Fri UTC
|
|
140
|
+
* windows, 1M ctx / 384K out), the V4 flash + vision-exp ids RETIRED with
|
|
141
|
+
* their names "temporarily routed" to V4.1 Flash at Flash prices (the
|
|
142
|
+
* `deepseek-v4-flash` entry is repriced to that card), and `deepseek-v4-pro`
|
|
143
|
+
* announced to route to V4.1 Flash at Flash prices from 2026-09-14 12:00
|
|
144
|
+
* Beijing (staged as `scheduledPricing` there); the pro card itself is
|
|
145
|
+
* unchanged (V4-Pro-0813, 0.66/1.98/0.022 off-peak). The card qualifies the
|
|
146
|
+
* peak windows BY DAY — "01:00 - 04:00 and 06:00 - 10:00 UTC, Monday through
|
|
147
|
+
* Friday (all other hours are off-peak)" — so the windows are
|
|
143
148
|
* `daysOfWeekUtc`-restricted rather than daily. `deepseek-v4-flash-vision-exp`
|
|
144
|
-
*
|
|
145
|
-
*
|
|
149
|
+
* is retired with the rest of V4 flash and stays deliberately uncatalogued.
|
|
150
|
+
* The US re-host rates (DeepInfra) are unchanged from the 2026-09-06 read —
|
|
151
|
+
* `DeepSeek-V4-Flash-0731` at $0.06/$0.18, cache read $0.015; DeepInfra also
|
|
152
|
+
* listed `deepseek-ai/DeepSeek-V4.1-Flash` on 2026-09-10 at $0.30/$1.20
|
|
153
|
+
* (cache read 0.02× input) — 2× native, not wired. See each entry's
|
|
154
|
+
* `regionPricing`.)
|
|
146
155
|
* - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
|
|
147
156
|
* the US re-host (kimi-k3 flagship 2026-07-16
|
|
148
157
|
* — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
|
|
@@ -150,6 +159,9 @@ const WEEKDAYS_UTC = [1, 2, 3, 4, 5];
|
|
|
150
159
|
* constraint that kept kimi-k2.7-code out. BOTH are now in the catalog: the
|
|
151
160
|
* moonshot bond gained preserved thinking (reasoning replayed through tool
|
|
152
161
|
* loops), so kimi-k3 is the Moonshot pick.)
|
|
162
|
+
* (re-verified 2026-09-10 on platform.kimi.ai/docs/models: kimi-k2.5 was
|
|
163
|
+
* officially discontinued 2026-08-31 — calls 404 — so its entry is now
|
|
164
|
+
* `disabled: true`. kimi-k3 / kimi-k2.6 / kimi-k2.7-code all still Active.)
|
|
153
165
|
* - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
|
|
154
166
|
* minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
|
|
155
167
|
* - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
|
|
@@ -201,6 +213,11 @@ const WEEKDAYS_UTC = [1, 2, 3, 4, 5];
|
|
|
201
213
|
* DeepInfra re-host zai-org/GLM-5.3-Flash, which bills exactly the list card
|
|
202
214
|
* ($0.15/$0.50, cache read 0.2x = $0.03 — verified live 2026-08-27), mapped
|
|
203
215
|
* in molecule-dev's ZHIPU_US_MODEL_MAP.)
|
|
216
|
+
* (re-verified 2026-09-10 on https://docs.z.ai/guides/overview/pricing: the
|
|
217
|
+
* promo ended on schedule 2026-09-09 24:00 UTC+8 and the page prints the list
|
|
218
|
+
* card again — $0.15/$0.50, cached input $0.03, no promo footnote. models.dev
|
|
219
|
+
* still carries the expired promo rate, so the KNOWN_DIVERGENCES entry now
|
|
220
|
+
* runs to a fixed 2026-12-09 re-verify date.)
|
|
204
221
|
*
|
|
205
222
|
* Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
|
|
206
223
|
* where the provider doesn't publish one; the provider sources above verify
|
|
@@ -657,11 +674,11 @@ export const MODELS = [
|
|
|
657
674
|
// Cached input 0.1× input, cache write 1.25× input (see section note).
|
|
658
675
|
cacheReadPricePerMTok: 0.02,
|
|
659
676
|
cacheWritePricePerMTok: 0.25,
|
|
660
|
-
//
|
|
661
|
-
//
|
|
662
|
-
// outright
|
|
663
|
-
//
|
|
664
|
-
|
|
677
|
+
// No freeTierRegions: Luna was the free-tier PLAN default 2026-08-18 →
|
|
678
|
+
// 2026-09-10 via this carve-out; the free planner is now `deepseek-flash`
|
|
679
|
+
// (outright freeTier), so the carve-out is removed — a stale one would
|
|
680
|
+
// claim a free-tier relationship Luna no longer has (see deepseek-v4-pro's
|
|
681
|
+
// note for the same removal pattern).
|
|
665
682
|
// Not published — best-effort estimate.
|
|
666
683
|
knowledgeCutoff: '2026-03-01',
|
|
667
684
|
},
|
|
@@ -1140,7 +1157,25 @@ export const MODELS = [
|
|
|
1140
1157
|
// DeepSeek
|
|
1141
1158
|
// Verified: https://api-docs.deepseek.com/quick_start/pricing
|
|
1142
1159
|
// https://api-docs.deepseek.com/guides/thinking_mode
|
|
1143
|
-
// https://api-docs.deepseek.com/updates/ (2026-
|
|
1160
|
+
// https://api-docs.deepseek.com/updates/ (re-read 2026-09-10)
|
|
1161
|
+
// 2026-09-10: DeepSeek-V4.1-Flash RELEASED (updates page, same day) under the
|
|
1162
|
+
// NEW id `deepseek-flash` — "Change the model name to deepseek-flash to call
|
|
1163
|
+
// the latest V4.1 Flash model." Its card CUT the flash rates to a third of
|
|
1164
|
+
// the 2026-08-16 rise:
|
|
1165
|
+
// flash (V4.1) off-peak miss 0.15 / hit 0.003 / out 0.6 (peak ×2, same
|
|
1166
|
+
// Mon-Fri UTC windows; 1M ctx / 384K out; concurrency 2500)
|
|
1167
|
+
// and the announcement claims "native multimodal visual understanding", so
|
|
1168
|
+
// the new entry carries supportsVision. It also "outperforms V4 Pro on
|
|
1169
|
+
// performance, cost, and speed", and two retirement facts landed with it:
|
|
1170
|
+
// 1. V4 Flash + V4 Flash Vision Exp are RETIRED; the legacy ids
|
|
1171
|
+
// `deepseek-v4-flash` / `deepseek-v4-flash-vision-exp` are "temporarily
|
|
1172
|
+
// routed" to V4.1 Flash and billed at Flash prices — so the v4-flash
|
|
1173
|
+
// entry below is repriced to the V4.1 card (its id still answers), and
|
|
1174
|
+
// leans on that "temporary" routing until molecule-dev's default ids
|
|
1175
|
+
// move to `deepseek-flash`.
|
|
1176
|
+
// 2. From 2026-09-14 12:00 Beijing (04:00 UTC), until V4.1 Pro ships, ALL
|
|
1177
|
+
// `deepseek-v4-pro` requests route to V4.1 Flash at Flash prices —
|
|
1178
|
+
// staged as `scheduledPricing` on the pro entry below.
|
|
1144
1179
|
// 2026-08-13: V4-Pro GA — and with it the price rise that the "coming soon"
|
|
1145
1180
|
// note below had been waiting on. It was staged as `scheduledPricing`
|
|
1146
1181
|
// effective 2026-08-16T16:00Z; that instant has PASSED and the rates are now
|
|
@@ -1242,8 +1277,32 @@ export const MODELS = [
|
|
|
1242
1277
|
],
|
|
1243
1278
|
multiplier: 2,
|
|
1244
1279
|
},
|
|
1280
|
+
// ANNOUNCED 2026-09-10 (updates page): from 2026-09-14 12:00 Beijing time
|
|
1281
|
+
// (04:00 UTC), and until V4.1 Pro ships, every `deepseek-v4-pro` request is
|
|
1282
|
+
// routed to V4.1 Flash and billed at the V4.1 FLASH price — so the base
|
|
1283
|
+
// rates become the flash card below (the peak windows are identical, so
|
|
1284
|
+
// they carry through unchanged). The US `regionPricing` above is DeepInfra's
|
|
1285
|
+
// own card for its V4-Pro copy and is NOT touched by the native routing.
|
|
1286
|
+
// Fold into the base fields once the instant has passed (the freshness
|
|
1287
|
+
// gate gives models.dev its scheduled-landing grace meanwhile).
|
|
1288
|
+
scheduledPricing: {
|
|
1289
|
+
effectiveFrom: '2026-09-14T04:00:00Z',
|
|
1290
|
+
inputPricePerMTok: 0.15,
|
|
1291
|
+
outputPricePerMTok: 0.6,
|
|
1292
|
+
cacheReadPricePerMTok: 0.003,
|
|
1293
|
+
cacheWritePricePerMTok: 0.15,
|
|
1294
|
+
source: 'https://api-docs.deepseek.com/updates/ (2026-09-10): deepseek-v4-pro → V4.1 Flash at Flash prices from 2026-09-14 12:00 Beijing',
|
|
1295
|
+
},
|
|
1245
1296
|
// Not published by DeepSeek — best-effort estimate.
|
|
1246
1297
|
knowledgeCutoff: '2025-07-01',
|
|
1298
|
+
// Deprecated the day it stops being itself: DeepSeek's own testing has
|
|
1299
|
+
// V4.1 Flash "comprehensively surpassing" Pro on performance/cost/speed,
|
|
1300
|
+
// and from 2026-09-14 12:00 Beijing every `deepseek-v4-pro` request routes
|
|
1301
|
+
// to V4.1 Flash at Flash prices (see scheduledPricing above) until V4.1 Pro
|
|
1302
|
+
// ships — at which point this becomes a new entry's succession problem, not
|
|
1303
|
+
// this one's. Until the 14th it still serves real V4-Pro-0813 weights at
|
|
1304
|
+
// the Pro card, so it stays selectable for existing selections.
|
|
1305
|
+
deprecatedAt: '2026-09-14',
|
|
1247
1306
|
},
|
|
1248
1307
|
{
|
|
1249
1308
|
id: 'deepseek-v4-flash',
|
|
@@ -1260,37 +1319,62 @@ export const MODELS = [
|
|
|
1260
1319
|
supportsVision: false,
|
|
1261
1320
|
supportsPromptCaching: true,
|
|
1262
1321
|
supportsTools: true,
|
|
1263
|
-
//
|
|
1264
|
-
//
|
|
1265
|
-
//
|
|
1266
|
-
freeTier
|
|
1267
|
-
//
|
|
1268
|
-
|
|
1269
|
-
|
|
1322
|
+
// freeTier moved to `deepseek-flash` (2026-09-10) together with
|
|
1323
|
+
// molecule-dev's default ids — the free tier follows the go-forward
|
|
1324
|
+
// evergreen id, not a retired name whose routing DeepSeek calls
|
|
1325
|
+
// "temporary". Exactly one model in this catalog may carry freeTier
|
|
1326
|
+
// (enforced by lookup.test.ts).
|
|
1327
|
+
// The V4.1 Flash card (2026-09-10). This legacy id is RETIRED and
|
|
1328
|
+
// "temporarily routed" to V4.1 Flash, billed at Flash prices — so these ARE
|
|
1329
|
+
// the rates the id answers with today (they replace the 0.22/0.66/0.007
|
|
1330
|
+
// card the same id served from 2026-08-16; same in-place succession as the
|
|
1331
|
+
// 2026-07-31 0731 re-post-train). Peak is `peakPricing.multiplier` × these.
|
|
1332
|
+
inputPricePerMTok: 0.15,
|
|
1333
|
+
outputPricePerMTok: 0.6,
|
|
1270
1334
|
// DeepSeek automatic context cache: absolute cache-hit price ($/M).
|
|
1271
|
-
cacheReadPricePerMTok: 0.
|
|
1335
|
+
cacheReadPricePerMTok: 0.003,
|
|
1272
1336
|
// DeepSeek charges no cache-write premium — write bills at input.
|
|
1273
|
-
cacheWritePricePerMTok: 0.
|
|
1337
|
+
cacheWritePricePerMTok: 0.15,
|
|
1274
1338
|
// US (DeepInfra) DEFAULT as of 2026-08-16 — flipped from CN when DeepSeek's
|
|
1275
1339
|
// rise landed (owner decision 2026-08-14). CN was cheaper on real traffic
|
|
1276
1340
|
// only because of its cache reads; the rise takes those from $0.0028 to
|
|
1277
|
-
// $0.007 (peak $0.014) against DeepInfra's flat $0.
|
|
1278
|
-
// enough to carry the
|
|
1341
|
+
// $0.007 (peak $0.014) against DeepInfra's flat $0.015, which is no longer
|
|
1342
|
+
// enough to carry the 3.7x CN now loses on BOTH fresh input (0.22 vs 0.06)
|
|
1343
|
+
// and output (0.66 vs 0.18). On
|
|
1279
1344
|
// the agentic mix this model actually serves (~94% cache hits) US is
|
|
1280
|
-
// cheaper at EVERY hour: 0.
|
|
1345
|
+
// cheaper at EVERY hour: 0.103c/turn flat vs 0.152c off-peak and 0.303c at
|
|
1281
1346
|
// peak. It is also flat-rate, so free-tier cost stops varying by Beijing
|
|
1282
1347
|
// business hours. Re-derive if the cache-hit ratio drops much below ~90%,
|
|
1283
1348
|
// where CN's cheaper reads start winning again. This deliberately splits
|
|
1284
1349
|
// the plan/execute pair across regions — Pro stays CN because its US
|
|
1285
1350
|
// re-host is ~2.3x its own native rate even after the rise.
|
|
1351
|
+
// Re-derived 2026-09-10 against the repriced CN card: CN's cache hits are
|
|
1352
|
+
// now 5× cheaper than the re-host's ($0.003 vs $0.015), so CN wins on
|
|
1353
|
+
// blended input (0.0118 vs 0.0177 $/MTok off-peak) — but US wins on output
|
|
1354
|
+
// at 0.18 vs 0.6/1.2, and CN only takes a turn whose output is under ~1.4%
|
|
1355
|
+
// of its input. At real agentic output ratios (5–25%) US still wins every
|
|
1356
|
+
// hour, so the US default holds.
|
|
1357
|
+
// MIGRATED 2026-09-10: molecule-dev's default literals (EXECUTOR_MODEL /
|
|
1358
|
+
// EXECUTOR_MODEL_CUSTOM / DEFAULT_CHAT_MODEL / COMPACTION_MODEL /
|
|
1359
|
+
// COMMIT_MESSAGE_MODEL / FREE_TIER_MODELS.execute) now point at
|
|
1360
|
+
// `deepseek-flash` (cn-native). This entry stays selectable — not
|
|
1361
|
+
// `supersededBy` — because its US leg (DeepInfra 0731 at $0.06/$0.18 flat,
|
|
1362
|
+
// older-but-solid weights DeepInfra keeps serving) is still the cheapest
|
|
1363
|
+
// flash serving there is and saved selections should keep working; it is
|
|
1364
|
+
// merely hidden from offering via `deprecatedAt`.
|
|
1286
1365
|
regions: ['us', 'cn'],
|
|
1287
|
-
// US = DeepInfra, verified 2026-
|
|
1366
|
+
// US = DeepInfra, verified 2026-09-06 against the id the bond actually
|
|
1288
1367
|
// sends: `deepseek-ai/DeepSeek-V4-Flash-0731`, the official release that
|
|
1289
1368
|
// supersedes the preview weights still served under the un-dated id
|
|
1290
|
-
// (cents_per_input_token 0.
|
|
1291
|
-
// rate_per_input_token_cached 0.
|
|
1369
|
+
// (cents_per_input_token 0.000006, cents_per_output_token 0.000018,
|
|
1370
|
+
// rate_per_input_token_cached 0.25 → cache read = 0.25 × input).
|
|
1371
|
+
// DeepInfra CUT the 0731 input rate 0.08 → 0.06 (cache read 0.016 →
|
|
1372
|
+
// 0.015); output held at 0.18. The un-dated `DeepSeek-V4-Flash` id moved
|
|
1373
|
+
// the other way (0.09 input, cache 0.2× = 0.018), so the two ids no longer
|
|
1374
|
+
// price alike — this entry tracks the dated one the modelMap sends, and the
|
|
1375
|
+
// gap widens the case for the US default this model already carries.
|
|
1292
1376
|
regionPricing: {
|
|
1293
|
-
us: { inputPricePerMTok: 0.
|
|
1377
|
+
us: { inputPricePerMTok: 0.06, outputPricePerMTok: 0.18, cacheReadPricePerMTok: 0.015 },
|
|
1294
1378
|
},
|
|
1295
1379
|
// Peak-hour surcharge live since 2026-08-16T16:00Z, Mon-Fri (see
|
|
1296
1380
|
// deepseek-v4-pro). It applies to the NATIVE CN card only — this model
|
|
@@ -1304,6 +1388,67 @@ export const MODELS = [
|
|
|
1304
1388
|
},
|
|
1305
1389
|
// Not published by DeepSeek — best-effort estimate.
|
|
1306
1390
|
knowledgeCutoff: '2025-07-01',
|
|
1391
|
+
// Retired upstream 2026-09-10 (V4 flash ids; legacy name "temporarily
|
|
1392
|
+
// routed" to V4.1 Flash). See the regions note above for why this is
|
|
1393
|
+
// deprecatedAt — hidden from offering, still selectable — rather than
|
|
1394
|
+
// supersededBy.
|
|
1395
|
+
deprecatedAt: '2026-09-10',
|
|
1396
|
+
},
|
|
1397
|
+
{
|
|
1398
|
+
// Released 2026-09-10 (updates page) — the go-forward EVERGREEN flash id.
|
|
1399
|
+
// Same rates the repriced legacy entry above carries natively, but under
|
|
1400
|
+
// the id DeepSeek actually wants called, with the multimodal surface the
|
|
1401
|
+
// announcement claims ("native multimodal visual understanding") and none
|
|
1402
|
+
// of the retirement ambiguity of a "temporarily routed" legacy name.
|
|
1403
|
+
id: 'deepseek-flash',
|
|
1404
|
+
provider: 'deepseek',
|
|
1405
|
+
label: 'DeepSeek V4.1 Flash',
|
|
1406
|
+
description: 'Newest ultra-cheap & fast flash — vision-capable agentic coding',
|
|
1407
|
+
contextWindow: 1_000_000,
|
|
1408
|
+
maxOutputTokens: 384_000,
|
|
1409
|
+
// Non-thinking (see section note) — same bond treatment as the family:
|
|
1410
|
+
// the executor tool loop does not replay reasoning_content.
|
|
1411
|
+
supportsThinking: false,
|
|
1412
|
+
thinkingBudgetTokens: 0,
|
|
1413
|
+
thinkingConfigurable: false,
|
|
1414
|
+
supportsVision: true,
|
|
1415
|
+
supportsPromptCaching: true,
|
|
1416
|
+
supportsTools: true,
|
|
1417
|
+
// Free-tier default (moved from deepseek-v4-flash 2026-09-10): the
|
|
1418
|
+
// cheapest capable executor on the go-forward id — fast, non-thinking,
|
|
1419
|
+
// tool-calling, and now vision-capable. Serving cost is ~2× the retired
|
|
1420
|
+
// US leg on output-heavy turns (CN 0.6/1.2 output vs DeepInfra's flat
|
|
1421
|
+
// 0.18) — tiers.ts's free-turn budget math derives from this entry.
|
|
1422
|
+
// Exactly one model in this catalog may carry freeTier (lookup.test.ts).
|
|
1423
|
+
freeTier: true,
|
|
1424
|
+
// Off-peak V4.1 Flash card; peak is `peakPricing.multiplier` × these.
|
|
1425
|
+
inputPricePerMTok: 0.15,
|
|
1426
|
+
outputPricePerMTok: 0.6,
|
|
1427
|
+
// DeepSeek automatic context cache: absolute cache-hit price ($/M).
|
|
1428
|
+
cacheReadPricePerMTok: 0.003,
|
|
1429
|
+
// DeepSeek charges no cache-write premium — write bills at input.
|
|
1430
|
+
cacheWritePricePerMTok: 0.15,
|
|
1431
|
+
// PINNED NATIVE — no US re-host is wired. DeepInfra does serve
|
|
1432
|
+
// `deepseek-ai/DeepSeek-V4.1-Flash` (listed 2026-09-10) at $0.30/$1.20,
|
|
1433
|
+
// cache read 0.02× input = $0.006 — exactly 2× this native off-peak card —
|
|
1434
|
+
// but offering a us region needs a molecule-dev US_MODEL_MAP entry
|
|
1435
|
+
// (region-model-maps.ts) plus a regionPricing row here, and that pair is a
|
|
1436
|
+
// deliberate molecule-dev change (it also can't beat this card on the
|
|
1437
|
+
// agentic mix, unlike the legacy 0731 re-host). Single-entry ['cn'] pins
|
|
1438
|
+
// dispatch to the native endpoint regardless of per-model region choice.
|
|
1439
|
+
regions: ['cn'],
|
|
1440
|
+
// Same peak windows as the family — the V4.1 card footnote is verbatim the
|
|
1441
|
+
// 2026-08-31 sentence: 01:00-04:00 and 06:00-10:00 UTC, Mon-Fri, ×2.
|
|
1442
|
+
peakPricing: {
|
|
1443
|
+
windows: [
|
|
1444
|
+
{ startMinuteUtc: 60, endMinuteUtc: 240, daysOfWeekUtc: WEEKDAYS_UTC },
|
|
1445
|
+
{ startMinuteUtc: 360, endMinuteUtc: 600, daysOfWeekUtc: WEEKDAYS_UTC },
|
|
1446
|
+
],
|
|
1447
|
+
multiplier: 2,
|
|
1448
|
+
},
|
|
1449
|
+
// Not published by DeepSeek — best-effort estimate (family estimate; the
|
|
1450
|
+
// V4.1 announcement lists benchmarks but no training cutoff).
|
|
1451
|
+
knowledgeCutoff: '2025-07-01',
|
|
1307
1452
|
},
|
|
1308
1453
|
// ---------------------------------------------------------------------------
|
|
1309
1454
|
// Moonshot (Kimi)
|
|
@@ -1464,8 +1609,15 @@ export const MODELS = [
|
|
|
1464
1609
|
knowledgeCutoff: '2024-04-01',
|
|
1465
1610
|
// Two generations behind. `supersededBy` names the current selectable Kimi
|
|
1466
1611
|
// (k3) rather than the also-superseded k2.6, so a saved selection resolves
|
|
1467
|
-
// forward in one hop.
|
|
1612
|
+
// forward in one hop. Moonshot RETIRED kimi-k2.5 on 2026-08-31 —
|
|
1613
|
+
// platform.kimi.ai/docs/models says calls 404, and a live 2026-09-10
|
|
1614
|
+
// dispatch probe confirmed the native host refusing it. The DeepInfra
|
|
1615
|
+
// re-host still answered inference on 2026-09-10 but is DELISTED from its
|
|
1616
|
+
// model catalog (detail page 404, /models/list unpruned) — borrowed time.
|
|
1617
|
+
// Disabled, never deleted: historical usage stays priceable and saved
|
|
1618
|
+
// selections still resolve forward.
|
|
1468
1619
|
deprecatedAt: '2026-04-01',
|
|
1620
|
+
disabled: true,
|
|
1469
1621
|
supersededBy: 'kimi-k3',
|
|
1470
1622
|
},
|
|
1471
1623
|
// ---------------------------------------------------------------------------
|
|
@@ -1830,9 +1982,11 @@ export const MODELS = [
|
|
|
1830
1982
|
supportsPromptCaching: true,
|
|
1831
1983
|
supportsTools: true,
|
|
1832
1984
|
webSearchToolType: 'web_search',
|
|
1833
|
-
//
|
|
1834
|
-
//
|
|
1835
|
-
//
|
|
1985
|
+
// List card, re-verified 2026-09-10 on the Z.ai pricing page after the
|
|
1986
|
+
// 50%-off launch promo ($0.075/$0.25, cached $0.015) ended on schedule
|
|
1987
|
+
// 2026-09-09 24:00 UTC+8 — this IS what the provider bills now; models.dev
|
|
1988
|
+
// still carries the expired promo rate (KNOWN_DIVERGENCES entry in
|
|
1989
|
+
// check-model-freshness, re-verify date 2026-12-09).
|
|
1836
1990
|
inputPricePerMTok: 0.15,
|
|
1837
1991
|
outputPricePerMTok: 0.5,
|
|
1838
1992
|
// GLM context cache: read 0.2× input, no write premium.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@molecule/api-resource-ai-models",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.6.0",
|
|
4
4
|
"description": "AI model catalog — server-side source of truth plus an authentication-gated discovery endpoint",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/index.js",
|
|
@@ -32,7 +32,7 @@
|
|
|
32
32
|
"@molecule/api-resource": "1.0.1",
|
|
33
33
|
"@types/node": "26.1.2",
|
|
34
34
|
"typescript": "6.0.3",
|
|
35
|
-
"vitest": "4.1.
|
|
35
|
+
"vitest": "4.1.11"
|
|
36
36
|
},
|
|
37
37
|
"peerDependencies": {
|
|
38
38
|
"@molecule/api-bond": "^1.0.1",
|