@molecule/api-resource-ai-models 1.5.1 → 1.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -14
- package/dist/models.d.ts +27 -13
- package/dist/models.d.ts.map +1 -1
- package/dist/models.js +181 -32
- package/package.json +2 -2
package/README.md
CHANGED
|
@@ -3,7 +3,7 @@ AUTO-GENERATED — DO NOT EDIT THIS FILE.
|
|
|
3
3
|
Generated by `mlcl sync-docs` from the package's src/index.ts JSDoc + mlcl/registry.json.
|
|
4
4
|
Edits here are overwritten on the next commit (molecule's pre-commit hook regenerates).
|
|
5
5
|
To change this document, edit the module-level JSDoc in src/index.ts.
|
|
6
|
-
Generated: 2026-09-
|
|
6
|
+
Generated: 2026-09-10T11:48:36.810Z
|
|
7
7
|
-->
|
|
8
8
|
|
|
9
9
|
# @molecule/api-resource-ai-models
|
|
@@ -860,20 +860,26 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
860
860
|
not modeled; reasoning_effort low|medium|high default high, image input;
|
|
861
861
|
grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
|
|
862
862
|
grok-code-fast-1 no longer listed — retires 2026-08-15)
|
|
863
|
-
- DeepSeek: https://api-docs.deepseek.com/quick_start/pricing
|
|
864
|
-
2026-09-
|
|
865
|
-
never in this catalog. The
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
|
|
870
|
-
|
|
863
|
+
- DeepSeek: https://api-docs.deepseek.com/quick_start/pricing +
|
|
864
|
+
/updates/ (verified 2026-09-10; legacy deepseek-chat/-reasoner ids fully
|
|
865
|
+
retired 2026-07-24 — never in this catalog. The 2026-09-10 re-read caught
|
|
866
|
+
the V4.1-Flash release DAY-OF: new evergreen id `deepseek-flash` at
|
|
867
|
+
off-peak miss $0.15 / hit $0.003 / out $0.6 (peak ×2, same Mon-Fri UTC
|
|
868
|
+
windows, 1M ctx / 384K out), the V4 flash + vision-exp ids RETIRED with
|
|
869
|
+
their names "temporarily routed" to V4.1 Flash at Flash prices (the
|
|
870
|
+
`deepseek-v4-flash` entry is repriced to that card), and `deepseek-v4-pro`
|
|
871
|
+
announced to route to V4.1 Flash at Flash prices from 2026-09-14 12:00
|
|
872
|
+
Beijing (staged as `scheduledPricing` there); the pro card itself is
|
|
873
|
+
unchanged (V4-Pro-0813, 0.66/1.98/0.022 off-peak). The card qualifies the
|
|
874
|
+
peak windows BY DAY — "01:00 - 04:00 and 06:00 - 10:00 UTC, Monday through
|
|
875
|
+
Friday (all other hours are off-peak)" — so the windows are
|
|
871
876
|
`daysOfWeekUtc`-restricted rather than daily. `deepseek-v4-flash-vision-exp`
|
|
872
|
-
|
|
873
|
-
|
|
874
|
-
|
|
875
|
-
`DeepSeek-V4-Flash
|
|
876
|
-
|
|
877
|
+
is retired with the rest of V4 flash and stays deliberately uncatalogued.
|
|
878
|
+
The US re-host rates (DeepInfra) are unchanged from the 2026-09-06 read —
|
|
879
|
+
`DeepSeek-V4-Flash-0731` at $0.06/$0.18, cache read $0.015; DeepInfra also
|
|
880
|
+
listed `deepseek-ai/DeepSeek-V4.1-Flash` on 2026-09-10 at $0.30/$1.20
|
|
881
|
+
(cache read 0.02× input) — 2× native, not wired. See each entry's
|
|
882
|
+
`regionPricing`.)
|
|
877
883
|
- Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
|
|
878
884
|
the US re-host (kimi-k3 flagship 2026-07-16
|
|
879
885
|
— 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
|
|
@@ -881,6 +887,9 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
881
887
|
constraint that kept kimi-k2.7-code out. BOTH are now in the catalog: the
|
|
882
888
|
moonshot bond gained preserved thinking (reasoning replayed through tool
|
|
883
889
|
loops), so kimi-k3 is the Moonshot pick.)
|
|
890
|
+
(re-verified 2026-09-10 on platform.kimi.ai/docs/models: kimi-k2.5 was
|
|
891
|
+
officially discontinued 2026-08-31 — calls 404 — so its entry is now
|
|
892
|
+
`disabled: true`. kimi-k3 / kimi-k2.6 / kimi-k2.7-code all still Active.)
|
|
884
893
|
- MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
|
|
885
894
|
minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
|
|
886
895
|
- Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
|
|
@@ -932,6 +941,11 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
932
941
|
DeepInfra re-host zai-org/GLM-5.3-Flash, which bills exactly the list card
|
|
933
942
|
($0.15/$0.50, cache read 0.2x = $0.03 — verified live 2026-08-27), mapped
|
|
934
943
|
in molecule-dev's ZHIPU_US_MODEL_MAP.)
|
|
944
|
+
(re-verified 2026-09-10 on https://docs.z.ai/guides/overview/pricing: the
|
|
945
|
+
promo ended on schedule 2026-09-09 24:00 UTC+8 and the page prints the list
|
|
946
|
+
card again — $0.15/$0.50, cached input $0.03, no promo footnote. models.dev
|
|
947
|
+
still carries the expired promo rate, so the KNOWN_DIVERGENCES entry now
|
|
948
|
+
runs to a fixed 2026-12-09 re-verify date.)
|
|
935
949
|
|
|
936
950
|
Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
|
|
937
951
|
where the provider doesn't publish one; the provider sources above verify
|
package/dist/models.d.ts
CHANGED
|
@@ -128,20 +128,26 @@ import type { ModelDefinition } from './types.js';
|
|
|
128
128
|
* not modeled; reasoning_effort low|medium|high default high, image input;
|
|
129
129
|
* grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
|
|
130
130
|
* grok-code-fast-1 no longer listed — retires 2026-08-15)
|
|
131
|
-
* - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing
|
|
132
|
-
* 2026-09-
|
|
133
|
-
* never in this catalog. The
|
|
134
|
-
*
|
|
135
|
-
*
|
|
136
|
-
*
|
|
137
|
-
*
|
|
138
|
-
*
|
|
131
|
+
* - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing +
|
|
132
|
+
* /updates/ (verified 2026-09-10; legacy deepseek-chat/-reasoner ids fully
|
|
133
|
+
* retired 2026-07-24 — never in this catalog. The 2026-09-10 re-read caught
|
|
134
|
+
* the V4.1-Flash release DAY-OF: new evergreen id `deepseek-flash` at
|
|
135
|
+
* off-peak miss $0.15 / hit $0.003 / out $0.6 (peak ×2, same Mon-Fri UTC
|
|
136
|
+
* windows, 1M ctx / 384K out), the V4 flash + vision-exp ids RETIRED with
|
|
137
|
+
* their names "temporarily routed" to V4.1 Flash at Flash prices (the
|
|
138
|
+
* `deepseek-v4-flash` entry is repriced to that card), and `deepseek-v4-pro`
|
|
139
|
+
* announced to route to V4.1 Flash at Flash prices from 2026-09-14 12:00
|
|
140
|
+
* Beijing (staged as `scheduledPricing` there); the pro card itself is
|
|
141
|
+
* unchanged (V4-Pro-0813, 0.66/1.98/0.022 off-peak). The card qualifies the
|
|
142
|
+
* peak windows BY DAY — "01:00 - 04:00 and 06:00 - 10:00 UTC, Monday through
|
|
143
|
+
* Friday (all other hours are off-peak)" — so the windows are
|
|
139
144
|
* `daysOfWeekUtc`-restricted rather than daily. `deepseek-v4-flash-vision-exp`
|
|
140
|
-
*
|
|
141
|
-
*
|
|
142
|
-
*
|
|
143
|
-
* `DeepSeek-V4-Flash
|
|
144
|
-
*
|
|
145
|
+
* is retired with the rest of V4 flash and stays deliberately uncatalogued.
|
|
146
|
+
* The US re-host rates (DeepInfra) are unchanged from the 2026-09-06 read —
|
|
147
|
+
* `DeepSeek-V4-Flash-0731` at $0.06/$0.18, cache read $0.015; DeepInfra also
|
|
148
|
+
* listed `deepseek-ai/DeepSeek-V4.1-Flash` on 2026-09-10 at $0.30/$1.20
|
|
149
|
+
* (cache read 0.02× input) — 2× native, not wired. See each entry's
|
|
150
|
+
* `regionPricing`.)
|
|
145
151
|
* - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
|
|
146
152
|
* the US re-host (kimi-k3 flagship 2026-07-16
|
|
147
153
|
* — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
|
|
@@ -149,6 +155,9 @@ import type { ModelDefinition } from './types.js';
|
|
|
149
155
|
* constraint that kept kimi-k2.7-code out. BOTH are now in the catalog: the
|
|
150
156
|
* moonshot bond gained preserved thinking (reasoning replayed through tool
|
|
151
157
|
* loops), so kimi-k3 is the Moonshot pick.)
|
|
158
|
+
* (re-verified 2026-09-10 on platform.kimi.ai/docs/models: kimi-k2.5 was
|
|
159
|
+
* officially discontinued 2026-08-31 — calls 404 — so its entry is now
|
|
160
|
+
* `disabled: true`. kimi-k3 / kimi-k2.6 / kimi-k2.7-code all still Active.)
|
|
152
161
|
* - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
|
|
153
162
|
* minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
|
|
154
163
|
* - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
|
|
@@ -200,6 +209,11 @@ import type { ModelDefinition } from './types.js';
|
|
|
200
209
|
* DeepInfra re-host zai-org/GLM-5.3-Flash, which bills exactly the list card
|
|
201
210
|
* ($0.15/$0.50, cache read 0.2x = $0.03 — verified live 2026-08-27), mapped
|
|
202
211
|
* in molecule-dev's ZHIPU_US_MODEL_MAP.)
|
|
212
|
+
* (re-verified 2026-09-10 on https://docs.z.ai/guides/overview/pricing: the
|
|
213
|
+
* promo ended on schedule 2026-09-09 24:00 UTC+8 and the page prints the list
|
|
214
|
+
* card again — $0.15/$0.50, cached input $0.03, no promo footnote. models.dev
|
|
215
|
+
* still carries the expired promo rate, so the KNOWN_DIVERGENCES entry now
|
|
216
|
+
* runs to a fixed 2026-12-09 re-verify date.)
|
|
203
217
|
*
|
|
204
218
|
* Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
|
|
205
219
|
* where the provider doesn't publish one; the provider sources above verify
|
package/dist/models.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAQjD
|
|
1
|
+
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAQjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkNG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAq1DnC,CAAA"}
|
package/dist/models.js
CHANGED
|
@@ -132,20 +132,26 @@ const WEEKDAYS_UTC = [1, 2, 3, 4, 5];
|
|
|
132
132
|
* not modeled; reasoning_effort low|medium|high default high, image input;
|
|
133
133
|
* grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
|
|
134
134
|
* grok-code-fast-1 no longer listed — retires 2026-08-15)
|
|
135
|
-
* - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing
|
|
136
|
-
* 2026-09-
|
|
137
|
-
* never in this catalog. The
|
|
138
|
-
*
|
|
139
|
-
*
|
|
140
|
-
*
|
|
141
|
-
*
|
|
142
|
-
*
|
|
135
|
+
* - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing +
|
|
136
|
+
* /updates/ (verified 2026-09-10; legacy deepseek-chat/-reasoner ids fully
|
|
137
|
+
* retired 2026-07-24 — never in this catalog. The 2026-09-10 re-read caught
|
|
138
|
+
* the V4.1-Flash release DAY-OF: new evergreen id `deepseek-flash` at
|
|
139
|
+
* off-peak miss $0.15 / hit $0.003 / out $0.6 (peak ×2, same Mon-Fri UTC
|
|
140
|
+
* windows, 1M ctx / 384K out), the V4 flash + vision-exp ids RETIRED with
|
|
141
|
+
* their names "temporarily routed" to V4.1 Flash at Flash prices (the
|
|
142
|
+
* `deepseek-v4-flash` entry is repriced to that card), and `deepseek-v4-pro`
|
|
143
|
+
* announced to route to V4.1 Flash at Flash prices from 2026-09-14 12:00
|
|
144
|
+
* Beijing (staged as `scheduledPricing` there); the pro card itself is
|
|
145
|
+
* unchanged (V4-Pro-0813, 0.66/1.98/0.022 off-peak). The card qualifies the
|
|
146
|
+
* peak windows BY DAY — "01:00 - 04:00 and 06:00 - 10:00 UTC, Monday through
|
|
147
|
+
* Friday (all other hours are off-peak)" — so the windows are
|
|
143
148
|
* `daysOfWeekUtc`-restricted rather than daily. `deepseek-v4-flash-vision-exp`
|
|
144
|
-
*
|
|
145
|
-
*
|
|
146
|
-
*
|
|
147
|
-
* `DeepSeek-V4-Flash
|
|
148
|
-
*
|
|
149
|
+
* is retired with the rest of V4 flash and stays deliberately uncatalogued.
|
|
150
|
+
* The US re-host rates (DeepInfra) are unchanged from the 2026-09-06 read —
|
|
151
|
+
* `DeepSeek-V4-Flash-0731` at $0.06/$0.18, cache read $0.015; DeepInfra also
|
|
152
|
+
* listed `deepseek-ai/DeepSeek-V4.1-Flash` on 2026-09-10 at $0.30/$1.20
|
|
153
|
+
* (cache read 0.02× input) — 2× native, not wired. See each entry's
|
|
154
|
+
* `regionPricing`.)
|
|
149
155
|
* - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
|
|
150
156
|
* the US re-host (kimi-k3 flagship 2026-07-16
|
|
151
157
|
* — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
|
|
@@ -153,6 +159,9 @@ const WEEKDAYS_UTC = [1, 2, 3, 4, 5];
|
|
|
153
159
|
* constraint that kept kimi-k2.7-code out. BOTH are now in the catalog: the
|
|
154
160
|
* moonshot bond gained preserved thinking (reasoning replayed through tool
|
|
155
161
|
* loops), so kimi-k3 is the Moonshot pick.)
|
|
162
|
+
* (re-verified 2026-09-10 on platform.kimi.ai/docs/models: kimi-k2.5 was
|
|
163
|
+
* officially discontinued 2026-08-31 — calls 404 — so its entry is now
|
|
164
|
+
* `disabled: true`. kimi-k3 / kimi-k2.6 / kimi-k2.7-code all still Active.)
|
|
156
165
|
* - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
|
|
157
166
|
* minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
|
|
158
167
|
* - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
|
|
@@ -204,6 +213,11 @@ const WEEKDAYS_UTC = [1, 2, 3, 4, 5];
|
|
|
204
213
|
* DeepInfra re-host zai-org/GLM-5.3-Flash, which bills exactly the list card
|
|
205
214
|
* ($0.15/$0.50, cache read 0.2x = $0.03 — verified live 2026-08-27), mapped
|
|
206
215
|
* in molecule-dev's ZHIPU_US_MODEL_MAP.)
|
|
216
|
+
* (re-verified 2026-09-10 on https://docs.z.ai/guides/overview/pricing: the
|
|
217
|
+
* promo ended on schedule 2026-09-09 24:00 UTC+8 and the page prints the list
|
|
218
|
+
* card again — $0.15/$0.50, cached input $0.03, no promo footnote. models.dev
|
|
219
|
+
* still carries the expired promo rate, so the KNOWN_DIVERGENCES entry now
|
|
220
|
+
* runs to a fixed 2026-12-09 re-verify date.)
|
|
207
221
|
*
|
|
208
222
|
* Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
|
|
209
223
|
* where the provider doesn't publish one; the provider sources above verify
|
|
@@ -660,11 +674,11 @@ export const MODELS = [
|
|
|
660
674
|
// Cached input 0.1× input, cache write 1.25× input (see section note).
|
|
661
675
|
cacheReadPricePerMTok: 0.02,
|
|
662
676
|
cacheWritePricePerMTok: 0.25,
|
|
663
|
-
//
|
|
664
|
-
//
|
|
665
|
-
// outright
|
|
666
|
-
//
|
|
667
|
-
|
|
677
|
+
// No freeTierRegions: Luna was the free-tier PLAN default 2026-08-18 →
|
|
678
|
+
// 2026-09-10 via this carve-out; the free planner is now `deepseek-flash`
|
|
679
|
+
// (outright freeTier), so the carve-out is removed — a stale one would
|
|
680
|
+
// claim a free-tier relationship Luna no longer has (see deepseek-v4-pro's
|
|
681
|
+
// note for the same removal pattern).
|
|
668
682
|
// Not published — best-effort estimate.
|
|
669
683
|
knowledgeCutoff: '2026-03-01',
|
|
670
684
|
},
|
|
@@ -1143,7 +1157,25 @@ export const MODELS = [
|
|
|
1143
1157
|
// DeepSeek
|
|
1144
1158
|
// Verified: https://api-docs.deepseek.com/quick_start/pricing
|
|
1145
1159
|
// https://api-docs.deepseek.com/guides/thinking_mode
|
|
1146
|
-
// https://api-docs.deepseek.com/updates/ (2026-
|
|
1160
|
+
// https://api-docs.deepseek.com/updates/ (re-read 2026-09-10)
|
|
1161
|
+
// 2026-09-10: DeepSeek-V4.1-Flash RELEASED (updates page, same day) under the
|
|
1162
|
+
// NEW id `deepseek-flash` — "Change the model name to deepseek-flash to call
|
|
1163
|
+
// the latest V4.1 Flash model." Its card CUT the flash rates to a third of
|
|
1164
|
+
// the 2026-08-16 rise:
|
|
1165
|
+
// flash (V4.1) off-peak miss 0.15 / hit 0.003 / out 0.6 (peak ×2, same
|
|
1166
|
+
// Mon-Fri UTC windows; 1M ctx / 384K out; concurrency 2500)
|
|
1167
|
+
// and the announcement claims "native multimodal visual understanding", so
|
|
1168
|
+
// the new entry carries supportsVision. It also "outperforms V4 Pro on
|
|
1169
|
+
// performance, cost, and speed", and two retirement facts landed with it:
|
|
1170
|
+
// 1. V4 Flash + V4 Flash Vision Exp are RETIRED; the legacy ids
|
|
1171
|
+
// `deepseek-v4-flash` / `deepseek-v4-flash-vision-exp` are "temporarily
|
|
1172
|
+
// routed" to V4.1 Flash and billed at Flash prices — so the v4-flash
|
|
1173
|
+
// entry below is repriced to the V4.1 card (its id still answers), and
|
|
1174
|
+
// leans on that "temporary" routing until molecule-dev's default ids
|
|
1175
|
+
// move to `deepseek-flash`.
|
|
1176
|
+
// 2. From 2026-09-14 12:00 Beijing (04:00 UTC), until V4.1 Pro ships, ALL
|
|
1177
|
+
// `deepseek-v4-pro` requests route to V4.1 Flash at Flash prices —
|
|
1178
|
+
// staged as `scheduledPricing` on the pro entry below.
|
|
1147
1179
|
// 2026-08-13: V4-Pro GA — and with it the price rise that the "coming soon"
|
|
1148
1180
|
// note below had been waiting on. It was staged as `scheduledPricing`
|
|
1149
1181
|
// effective 2026-08-16T16:00Z; that instant has PASSED and the rates are now
|
|
@@ -1245,8 +1277,32 @@ export const MODELS = [
|
|
|
1245
1277
|
],
|
|
1246
1278
|
multiplier: 2,
|
|
1247
1279
|
},
|
|
1280
|
+
// ANNOUNCED 2026-09-10 (updates page): from 2026-09-14 12:00 Beijing time
|
|
1281
|
+
// (04:00 UTC), and until V4.1 Pro ships, every `deepseek-v4-pro` request is
|
|
1282
|
+
// routed to V4.1 Flash and billed at the V4.1 FLASH price — so the base
|
|
1283
|
+
// rates become the flash card below (the peak windows are identical, so
|
|
1284
|
+
// they carry through unchanged). The US `regionPricing` above is DeepInfra's
|
|
1285
|
+
// own card for its V4-Pro copy and is NOT touched by the native routing.
|
|
1286
|
+
// Fold into the base fields once the instant has passed (the freshness
|
|
1287
|
+
// gate gives models.dev its scheduled-landing grace meanwhile).
|
|
1288
|
+
scheduledPricing: {
|
|
1289
|
+
effectiveFrom: '2026-09-14T04:00:00Z',
|
|
1290
|
+
inputPricePerMTok: 0.15,
|
|
1291
|
+
outputPricePerMTok: 0.6,
|
|
1292
|
+
cacheReadPricePerMTok: 0.003,
|
|
1293
|
+
cacheWritePricePerMTok: 0.15,
|
|
1294
|
+
source: 'https://api-docs.deepseek.com/updates/ (2026-09-10): deepseek-v4-pro → V4.1 Flash at Flash prices from 2026-09-14 12:00 Beijing',
|
|
1295
|
+
},
|
|
1248
1296
|
// Not published by DeepSeek — best-effort estimate.
|
|
1249
1297
|
knowledgeCutoff: '2025-07-01',
|
|
1298
|
+
// Deprecated the day it stops being itself: DeepSeek's own testing has
|
|
1299
|
+
// V4.1 Flash "comprehensively surpassing" Pro on performance/cost/speed,
|
|
1300
|
+
// and from 2026-09-14 12:00 Beijing every `deepseek-v4-pro` request routes
|
|
1301
|
+
// to V4.1 Flash at Flash prices (see scheduledPricing above) until V4.1 Pro
|
|
1302
|
+
// ships — at which point this becomes a new entry's succession problem, not
|
|
1303
|
+
// this one's. Until the 14th it still serves real V4-Pro-0813 weights at
|
|
1304
|
+
// the Pro card, so it stays selectable for existing selections.
|
|
1305
|
+
deprecatedAt: '2026-09-14',
|
|
1250
1306
|
},
|
|
1251
1307
|
{
|
|
1252
1308
|
id: 'deepseek-v4-flash',
|
|
@@ -1263,17 +1319,22 @@ export const MODELS = [
|
|
|
1263
1319
|
supportsVision: false,
|
|
1264
1320
|
supportsPromptCaching: true,
|
|
1265
1321
|
supportsTools: true,
|
|
1266
|
-
//
|
|
1267
|
-
//
|
|
1268
|
-
//
|
|
1269
|
-
freeTier
|
|
1270
|
-
//
|
|
1271
|
-
|
|
1272
|
-
|
|
1322
|
+
// freeTier moved to `deepseek-flash` (2026-09-10) together with
|
|
1323
|
+
// molecule-dev's default ids — the free tier follows the go-forward
|
|
1324
|
+
// evergreen id, not a retired name whose routing DeepSeek calls
|
|
1325
|
+
// "temporary". Exactly one model in this catalog may carry freeTier
|
|
1326
|
+
// (enforced by lookup.test.ts).
|
|
1327
|
+
// The V4.1 Flash card (2026-09-10). This legacy id is RETIRED and
|
|
1328
|
+
// "temporarily routed" to V4.1 Flash, billed at Flash prices — so these ARE
|
|
1329
|
+
// the rates the id answers with today (they replace the 0.22/0.66/0.007
|
|
1330
|
+
// card the same id served from 2026-08-16; same in-place succession as the
|
|
1331
|
+
// 2026-07-31 0731 re-post-train). Peak is `peakPricing.multiplier` × these.
|
|
1332
|
+
inputPricePerMTok: 0.15,
|
|
1333
|
+
outputPricePerMTok: 0.6,
|
|
1273
1334
|
// DeepSeek automatic context cache: absolute cache-hit price ($/M).
|
|
1274
|
-
cacheReadPricePerMTok: 0.
|
|
1335
|
+
cacheReadPricePerMTok: 0.003,
|
|
1275
1336
|
// DeepSeek charges no cache-write premium — write bills at input.
|
|
1276
|
-
cacheWritePricePerMTok: 0.
|
|
1337
|
+
cacheWritePricePerMTok: 0.15,
|
|
1277
1338
|
// US (DeepInfra) DEFAULT as of 2026-08-16 — flipped from CN when DeepSeek's
|
|
1278
1339
|
// rise landed (owner decision 2026-08-14). CN was cheaper on real traffic
|
|
1279
1340
|
// only because of its cache reads; the rise takes those from $0.0028 to
|
|
@@ -1287,6 +1348,20 @@ export const MODELS = [
|
|
|
1287
1348
|
// where CN's cheaper reads start winning again. This deliberately splits
|
|
1288
1349
|
// the plan/execute pair across regions — Pro stays CN because its US
|
|
1289
1350
|
// re-host is ~2.3x its own native rate even after the rise.
|
|
1351
|
+
// Re-derived 2026-09-10 against the repriced CN card: CN's cache hits are
|
|
1352
|
+
// now 5× cheaper than the re-host's ($0.003 vs $0.015), so CN wins on
|
|
1353
|
+
// blended input (0.0118 vs 0.0177 $/MTok off-peak) — but US wins on output
|
|
1354
|
+
// at 0.18 vs 0.6/1.2, and CN only takes a turn whose output is under ~1.4%
|
|
1355
|
+
// of its input. At real agentic output ratios (5–25%) US still wins every
|
|
1356
|
+
// hour, so the US default holds.
|
|
1357
|
+
// MIGRATED 2026-09-10: molecule-dev's default literals (EXECUTOR_MODEL /
|
|
1358
|
+
// EXECUTOR_MODEL_CUSTOM / DEFAULT_CHAT_MODEL / COMPACTION_MODEL /
|
|
1359
|
+
// COMMIT_MESSAGE_MODEL / FREE_TIER_MODELS.execute) now point at
|
|
1360
|
+
// `deepseek-flash` (cn-native). This entry stays selectable — not
|
|
1361
|
+
// `supersededBy` — because its US leg (DeepInfra 0731 at $0.06/$0.18 flat,
|
|
1362
|
+
// older-but-solid weights DeepInfra keeps serving) is still the cheapest
|
|
1363
|
+
// flash serving there is and saved selections should keep working; it is
|
|
1364
|
+
// merely hidden from offering via `deprecatedAt`.
|
|
1290
1365
|
regions: ['us', 'cn'],
|
|
1291
1366
|
// US = DeepInfra, verified 2026-09-06 against the id the bond actually
|
|
1292
1367
|
// sends: `deepseek-ai/DeepSeek-V4-Flash-0731`, the official release that
|
|
@@ -1313,6 +1388,71 @@ export const MODELS = [
|
|
|
1313
1388
|
},
|
|
1314
1389
|
// Not published by DeepSeek — best-effort estimate.
|
|
1315
1390
|
knowledgeCutoff: '2025-07-01',
|
|
1391
|
+
// Retired upstream 2026-09-10 (V4 flash ids; legacy name "temporarily
|
|
1392
|
+
// routed" to V4.1 Flash). See the regions note above for why this is
|
|
1393
|
+
// deprecatedAt — hidden from offering, still selectable — rather than
|
|
1394
|
+
// supersededBy.
|
|
1395
|
+
deprecatedAt: '2026-09-10',
|
|
1396
|
+
},
|
|
1397
|
+
{
|
|
1398
|
+
// Released 2026-09-10 (updates page) — the go-forward EVERGREEN flash id.
|
|
1399
|
+
// Same rates the repriced legacy entry above carries natively, but under
|
|
1400
|
+
// the id DeepSeek actually wants called, with the multimodal surface the
|
|
1401
|
+
// announcement claims ("native multimodal visual understanding") and none
|
|
1402
|
+
// of the retirement ambiguity of a "temporarily routed" legacy name.
|
|
1403
|
+
id: 'deepseek-flash',
|
|
1404
|
+
provider: 'deepseek',
|
|
1405
|
+
label: 'DeepSeek V4.1 Flash',
|
|
1406
|
+
description: 'Newest ultra-cheap & fast flash — vision-capable agentic coding',
|
|
1407
|
+
contextWindow: 1_000_000,
|
|
1408
|
+
maxOutputTokens: 384_000,
|
|
1409
|
+
// Non-thinking (see section note) — same bond treatment as the family:
|
|
1410
|
+
// the executor tool loop does not replay reasoning_content.
|
|
1411
|
+
supportsThinking: false,
|
|
1412
|
+
thinkingBudgetTokens: 0,
|
|
1413
|
+
thinkingConfigurable: false,
|
|
1414
|
+
supportsVision: true,
|
|
1415
|
+
supportsPromptCaching: true,
|
|
1416
|
+
supportsTools: true,
|
|
1417
|
+
// Free-tier default (moved from deepseek-v4-flash 2026-09-10): the
|
|
1418
|
+
// cheapest capable executor on the go-forward id — fast, non-thinking,
|
|
1419
|
+
// tool-calling, and now vision-capable. Serving cost is ~2× the retired
|
|
1420
|
+
// US leg on output-heavy turns (CN 0.6/1.2 output vs DeepInfra's flat
|
|
1421
|
+
// 0.18) — tiers.ts's free-turn budget math derives from this entry.
|
|
1422
|
+
// Exactly one model in this catalog may carry freeTier (lookup.test.ts).
|
|
1423
|
+
freeTier: true,
|
|
1424
|
+
// Off-peak V4.1 Flash card; peak is `peakPricing.multiplier` × these.
|
|
1425
|
+
inputPricePerMTok: 0.15,
|
|
1426
|
+
outputPricePerMTok: 0.6,
|
|
1427
|
+
// DeepSeek automatic context cache: absolute cache-hit price ($/M).
|
|
1428
|
+
cacheReadPricePerMTok: 0.003,
|
|
1429
|
+
// DeepSeek charges no cache-write premium — write bills at input.
|
|
1430
|
+
cacheWritePricePerMTok: 0.15,
|
|
1431
|
+
// Native first (the default region); the DeepInfra US re-host
|
|
1432
|
+
// (`deepseek-ai/DeepSeek-V4.1-Flash`, molecule-dev region-model-maps.ts)
|
|
1433
|
+
// is offered as a SECOND region since 2026-09-14, when the native endpoint
|
|
1434
|
+
// queued every completion for hours (SSE keep-alives, no token, three
|
|
1435
|
+
// Synthase turns lost) while DeepInfra answered at once. Its card, read
|
|
1436
|
+
// live from api.deepinfra.com/models on 2026-09-14: $0.20/$0.60, cache
|
|
1437
|
+
// read 3% of input = $0.006 — roughly 2× this native off-peak card on
|
|
1438
|
+
// cache reads, so native stays the default and `us` is the fallback a
|
|
1439
|
+
// project picks when native is unavailable.
|
|
1440
|
+
regions: ['cn', 'us'],
|
|
1441
|
+
regionPricing: {
|
|
1442
|
+
us: { inputPricePerMTok: 0.2, outputPricePerMTok: 0.6, cacheReadPricePerMTok: 0.006 },
|
|
1443
|
+
},
|
|
1444
|
+
// Same peak windows as the family — the V4.1 card footnote is verbatim the
|
|
1445
|
+
// 2026-08-31 sentence: 01:00-04:00 and 06:00-10:00 UTC, Mon-Fri, ×2.
|
|
1446
|
+
peakPricing: {
|
|
1447
|
+
windows: [
|
|
1448
|
+
{ startMinuteUtc: 60, endMinuteUtc: 240, daysOfWeekUtc: WEEKDAYS_UTC },
|
|
1449
|
+
{ startMinuteUtc: 360, endMinuteUtc: 600, daysOfWeekUtc: WEEKDAYS_UTC },
|
|
1450
|
+
],
|
|
1451
|
+
multiplier: 2,
|
|
1452
|
+
},
|
|
1453
|
+
// Not published by DeepSeek — best-effort estimate (family estimate; the
|
|
1454
|
+
// V4.1 announcement lists benchmarks but no training cutoff).
|
|
1455
|
+
knowledgeCutoff: '2025-07-01',
|
|
1316
1456
|
},
|
|
1317
1457
|
// ---------------------------------------------------------------------------
|
|
1318
1458
|
// Moonshot (Kimi)
|
|
@@ -1473,8 +1613,15 @@ export const MODELS = [
|
|
|
1473
1613
|
knowledgeCutoff: '2024-04-01',
|
|
1474
1614
|
// Two generations behind. `supersededBy` names the current selectable Kimi
|
|
1475
1615
|
// (k3) rather than the also-superseded k2.6, so a saved selection resolves
|
|
1476
|
-
// forward in one hop.
|
|
1616
|
+
// forward in one hop. Moonshot RETIRED kimi-k2.5 on 2026-08-31 —
|
|
1617
|
+
// platform.kimi.ai/docs/models says calls 404, and a live 2026-09-10
|
|
1618
|
+
// dispatch probe confirmed the native host refusing it. The DeepInfra
|
|
1619
|
+
// re-host still answered inference on 2026-09-10 but is DELISTED from its
|
|
1620
|
+
// model catalog (detail page 404, /models/list unpruned) — borrowed time.
|
|
1621
|
+
// Disabled, never deleted: historical usage stays priceable and saved
|
|
1622
|
+
// selections still resolve forward.
|
|
1477
1623
|
deprecatedAt: '2026-04-01',
|
|
1624
|
+
disabled: true,
|
|
1478
1625
|
supersededBy: 'kimi-k3',
|
|
1479
1626
|
},
|
|
1480
1627
|
// ---------------------------------------------------------------------------
|
|
@@ -1839,9 +1986,11 @@ export const MODELS = [
|
|
|
1839
1986
|
supportsPromptCaching: true,
|
|
1840
1987
|
supportsTools: true,
|
|
1841
1988
|
webSearchToolType: 'web_search',
|
|
1842
|
-
//
|
|
1843
|
-
//
|
|
1844
|
-
//
|
|
1989
|
+
// List card, re-verified 2026-09-10 on the Z.ai pricing page after the
|
|
1990
|
+
// 50%-off launch promo ($0.075/$0.25, cached $0.015) ended on schedule
|
|
1991
|
+
// 2026-09-09 24:00 UTC+8 — this IS what the provider bills now; models.dev
|
|
1992
|
+
// still carries the expired promo rate (KNOWN_DIVERGENCES entry in
|
|
1993
|
+
// check-model-freshness, re-verify date 2026-12-09).
|
|
1845
1994
|
inputPricePerMTok: 0.15,
|
|
1846
1995
|
outputPricePerMTok: 0.5,
|
|
1847
1996
|
// GLM context cache: read 0.2× input, no write premium.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@molecule/api-resource-ai-models",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.6.1",
|
|
4
4
|
"description": "AI model catalog — server-side source of truth plus an authentication-gated discovery endpoint",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/index.js",
|
|
@@ -32,7 +32,7 @@
|
|
|
32
32
|
"@molecule/api-resource": "1.0.1",
|
|
33
33
|
"@types/node": "26.1.2",
|
|
34
34
|
"typescript": "6.0.3",
|
|
35
|
-
"vitest": "4.1.
|
|
35
|
+
"vitest": "4.1.11"
|
|
36
36
|
},
|
|
37
37
|
"peerDependencies": {
|
|
38
38
|
"@molecule/api-bond": "^1.0.1",
|