@molecule/api-resource-ai-models 1.5.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -3,7 +3,7 @@ AUTO-GENERATED — DO NOT EDIT THIS FILE.
3
3
  Generated by `mlcl sync-docs` from the package's src/index.ts JSDoc + mlcl/registry.json.
4
4
  Edits here are overwritten on the next commit (molecule's pre-commit hook regenerates).
5
5
  To change this document, edit the module-level JSDoc in src/index.ts.
6
- Generated: 2026-09-02T19:41:05.247Z
6
+ Generated: 2026-09-10T11:48:36.810Z
7
7
  -->
8
8
 
9
9
  # @molecule/api-resource-ai-models
@@ -860,17 +860,26 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
860
860
  not modeled; reasoning_effort low|medium|high default high, image input;
861
861
  grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
862
862
  grok-code-fast-1 no longer listed — retires 2026-08-15)
863
- - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
864
- 2026-08-31; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
865
- never in this catalog. The V4-Pro-GA price RISE effective 2026-08-16T16:00Z
866
- has LANDED and is folded into the base fields, along with the peak-hour 2×
867
- the same card introduced; every rate re-read on the card 2026-08-31 and
868
- unchanged. The card now qualifies the windows BY DAY — "01:00 - 04:00 and
869
- 06:00 - 10:00 UTC, Monday through Friday (all other hours are off-peak)" —
870
- where on 2026-08-18 it carried no day qualifier at all, so the windows are
863
+ - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing +
864
+ /updates/ (verified 2026-09-10; legacy deepseek-chat/-reasoner ids fully
865
+ retired 2026-07-24 — never in this catalog. The 2026-09-10 re-read caught
866
+ the V4.1-Flash release DAY-OF: new evergreen id `deepseek-flash` at
867
+ off-peak miss $0.15 / hit $0.003 / out $0.6 (peak ×2, same Mon-Fri UTC
868
+ windows, 1M ctx / 384K out), the V4 flash + vision-exp ids RETIRED with
869
+ their names "temporarily routed" to V4.1 Flash at Flash prices (the
870
+ `deepseek-v4-flash` entry is repriced to that card), and `deepseek-v4-pro`
871
+ announced to route to V4.1 Flash at Flash prices from 2026-09-14 12:00
872
+ Beijing (staged as `scheduledPricing` there); the pro card itself is
873
+ unchanged (V4-Pro-0813, 0.66/1.98/0.022 off-peak). The card qualifies the
874
+ peak windows BY DAY — "01:00 - 04:00 and 06:00 - 10:00 UTC, Monday through
875
+ Friday (all other hours are off-peak)" — so the windows are
871
876
  `daysOfWeekUtc`-restricted rather than daily. `deepseek-v4-flash-vision-exp`
872
- also appears on the card at flash's rates: EXPERIMENTAL and vision-only-new,
873
- deliberately not catalogued.)
877
+ is retired with the rest of V4 flash and stays deliberately uncatalogued.
878
+ The US re-host rates (DeepInfra) are unchanged from the 2026-09-06 read —
879
+ `DeepSeek-V4-Flash-0731` at $0.06/$0.18, cache read $0.015; DeepInfra also
880
+ listed `deepseek-ai/DeepSeek-V4.1-Flash` on 2026-09-10 at $0.30/$1.20
881
+ (cache read 0.02× input) — 2× native, not wired. See each entry's
882
+ `regionPricing`.)
874
883
  - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
875
884
  the US re-host (kimi-k3 flagship 2026-07-16
876
885
  — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
@@ -878,6 +887,9 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
878
887
  constraint that kept kimi-k2.7-code out. BOTH are now in the catalog: the
879
888
  moonshot bond gained preserved thinking (reasoning replayed through tool
880
889
  loops), so kimi-k3 is the Moonshot pick.)
890
+ (re-verified 2026-09-10 on platform.kimi.ai/docs/models: kimi-k2.5 was
891
+ officially discontinued 2026-08-31 — calls 404 — so its entry is now
892
+ `disabled: true`. kimi-k3 / kimi-k2.6 / kimi-k2.7-code all still Active.)
881
893
  - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
882
894
  minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
883
895
  - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
@@ -929,6 +941,11 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
929
941
  DeepInfra re-host zai-org/GLM-5.3-Flash, which bills exactly the list card
930
942
  ($0.15/$0.50, cache read 0.2x = $0.03 — verified live 2026-08-27), mapped
931
943
  in molecule-dev's ZHIPU_US_MODEL_MAP.)
944
+ (re-verified 2026-09-10 on https://docs.z.ai/guides/overview/pricing: the
945
+ promo ended on schedule 2026-09-09 24:00 UTC+8 and the page prints the list
946
+ card again — $0.15/$0.50, cached input $0.03, no promo footnote. models.dev
947
+ still carries the expired promo rate, so the KNOWN_DIVERGENCES entry now
948
+ runs to a fixed 2026-12-09 re-verify date.)
932
949
 
933
950
  Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
934
951
  where the provider doesn't publish one; the provider sources above verify
package/dist/models.d.ts CHANGED
@@ -128,17 +128,26 @@ import type { ModelDefinition } from './types.js';
128
128
  * not modeled; reasoning_effort low|medium|high default high, image input;
129
129
  * grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
130
130
  * grok-code-fast-1 no longer listed — retires 2026-08-15)
131
- * - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
132
- * 2026-08-31; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
133
- * never in this catalog. The V4-Pro-GA price RISE effective 2026-08-16T16:00Z
134
- * has LANDED and is folded into the base fields, along with the peak-hour 2×
135
- * the same card introduced; every rate re-read on the card 2026-08-31 and
136
- * unchanged. The card now qualifies the windows BY DAY — "01:00 - 04:00 and
137
- * 06:00 - 10:00 UTC, Monday through Friday (all other hours are off-peak)" —
138
- * where on 2026-08-18 it carried no day qualifier at all, so the windows are
131
+ * - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing +
132
+ * /updates/ (verified 2026-09-10; legacy deepseek-chat/-reasoner ids fully
133
+ * retired 2026-07-24 — never in this catalog. The 2026-09-10 re-read caught
134
+ * the V4.1-Flash release DAY-OF: new evergreen id `deepseek-flash` at
135
+ * off-peak miss $0.15 / hit $0.003 / out $0.6 (peak ×2, same Mon-Fri UTC
136
+ * windows, 1M ctx / 384K out), the V4 flash + vision-exp ids RETIRED with
137
+ * their names "temporarily routed" to V4.1 Flash at Flash prices (the
138
+ * `deepseek-v4-flash` entry is repriced to that card), and `deepseek-v4-pro`
139
+ * announced to route to V4.1 Flash at Flash prices from 2026-09-14 12:00
140
+ * Beijing (staged as `scheduledPricing` there); the pro card itself is
141
+ * unchanged (V4-Pro-0813, 0.66/1.98/0.022 off-peak). The card qualifies the
142
+ * peak windows BY DAY — "01:00 - 04:00 and 06:00 - 10:00 UTC, Monday through
143
+ * Friday (all other hours are off-peak)" — so the windows are
139
144
  * `daysOfWeekUtc`-restricted rather than daily. `deepseek-v4-flash-vision-exp`
140
- * also appears on the card at flash's rates: EXPERIMENTAL and vision-only-new,
141
- * deliberately not catalogued.)
145
+ * is retired with the rest of V4 flash and stays deliberately uncatalogued.
146
+ * The US re-host rates (DeepInfra) are unchanged from the 2026-09-06 read —
147
+ * `DeepSeek-V4-Flash-0731` at $0.06/$0.18, cache read $0.015; DeepInfra also
148
+ * listed `deepseek-ai/DeepSeek-V4.1-Flash` on 2026-09-10 at $0.30/$1.20
149
+ * (cache read 0.02× input) — 2× native, not wired. See each entry's
150
+ * `regionPricing`.)
142
151
  * - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
143
152
  * the US re-host (kimi-k3 flagship 2026-07-16
144
153
  * — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
@@ -146,6 +155,9 @@ import type { ModelDefinition } from './types.js';
146
155
  * constraint that kept kimi-k2.7-code out. BOTH are now in the catalog: the
147
156
  * moonshot bond gained preserved thinking (reasoning replayed through tool
148
157
  * loops), so kimi-k3 is the Moonshot pick.)
158
+ * (re-verified 2026-09-10 on platform.kimi.ai/docs/models: kimi-k2.5 was
159
+ * officially discontinued 2026-08-31 — calls 404 — so its entry is now
160
+ * `disabled: true`. kimi-k3 / kimi-k2.6 / kimi-k2.7-code all still Active.)
149
161
  * - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
150
162
  * minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
151
163
  * - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
@@ -197,6 +209,11 @@ import type { ModelDefinition } from './types.js';
197
209
  * DeepInfra re-host zai-org/GLM-5.3-Flash, which bills exactly the list card
198
210
  * ($0.15/$0.50, cache read 0.2x = $0.03 — verified live 2026-08-27), mapped
199
211
  * in molecule-dev's ZHIPU_US_MODEL_MAP.)
212
+ * (re-verified 2026-09-10 on https://docs.z.ai/guides/overview/pricing: the
213
+ * promo ended on schedule 2026-09-09 24:00 UTC+8 and the page prints the list
214
+ * card again — $0.15/$0.50, cached input $0.03, no promo footnote. models.dev
215
+ * still carries the expired promo rate, so the KNOWN_DIVERGENCES entry now
216
+ * runs to a fixed 2026-12-09 re-verify date.)
200
217
  *
201
218
  * Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
202
219
  * where the provider doesn't publish one; the provider sources above verify
@@ -1 +1 @@
1
- {"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAQjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAiMG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAusDnC,CAAA"}
1
+ {"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAQjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkNG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAi1DnC,CAAA"}
package/dist/models.js CHANGED
@@ -132,17 +132,26 @@ const WEEKDAYS_UTC = [1, 2, 3, 4, 5];
132
132
  * not modeled; reasoning_effort low|medium|high default high, image input;
133
133
  * grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
134
134
  * grok-code-fast-1 no longer listed — retires 2026-08-15)
135
- * - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
136
- * 2026-08-31; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
137
- * never in this catalog. The V4-Pro-GA price RISE effective 2026-08-16T16:00Z
138
- * has LANDED and is folded into the base fields, along with the peak-hour 2×
139
- * the same card introduced; every rate re-read on the card 2026-08-31 and
140
- * unchanged. The card now qualifies the windows BY DAY — "01:00 - 04:00 and
141
- * 06:00 - 10:00 UTC, Monday through Friday (all other hours are off-peak)" —
142
- * where on 2026-08-18 it carried no day qualifier at all, so the windows are
135
+ * - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing +
136
+ * /updates/ (verified 2026-09-10; legacy deepseek-chat/-reasoner ids fully
137
+ * retired 2026-07-24 — never in this catalog. The 2026-09-10 re-read caught
138
+ * the V4.1-Flash release DAY-OF: new evergreen id `deepseek-flash` at
139
+ * off-peak miss $0.15 / hit $0.003 / out $0.6 (peak ×2, same Mon-Fri UTC
140
+ * windows, 1M ctx / 384K out), the V4 flash + vision-exp ids RETIRED with
141
+ * their names "temporarily routed" to V4.1 Flash at Flash prices (the
142
+ * `deepseek-v4-flash` entry is repriced to that card), and `deepseek-v4-pro`
143
+ * announced to route to V4.1 Flash at Flash prices from 2026-09-14 12:00
144
+ * Beijing (staged as `scheduledPricing` there); the pro card itself is
145
+ * unchanged (V4-Pro-0813, 0.66/1.98/0.022 off-peak). The card qualifies the
146
+ * peak windows BY DAY — "01:00 - 04:00 and 06:00 - 10:00 UTC, Monday through
147
+ * Friday (all other hours are off-peak)" — so the windows are
143
148
  * `daysOfWeekUtc`-restricted rather than daily. `deepseek-v4-flash-vision-exp`
144
- * also appears on the card at flash's rates: EXPERIMENTAL and vision-only-new,
145
- * deliberately not catalogued.)
149
+ * is retired with the rest of V4 flash and stays deliberately uncatalogued.
150
+ * The US re-host rates (DeepInfra) are unchanged from the 2026-09-06 read —
151
+ * `DeepSeek-V4-Flash-0731` at $0.06/$0.18, cache read $0.015; DeepInfra also
152
+ * listed `deepseek-ai/DeepSeek-V4.1-Flash` on 2026-09-10 at $0.30/$1.20
153
+ * (cache read 0.02× input) — 2× native, not wired. See each entry's
154
+ * `regionPricing`.)
146
155
  * - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
147
156
  * the US re-host (kimi-k3 flagship 2026-07-16
148
157
  * — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
@@ -150,6 +159,9 @@ const WEEKDAYS_UTC = [1, 2, 3, 4, 5];
150
159
  * constraint that kept kimi-k2.7-code out. BOTH are now in the catalog: the
151
160
  * moonshot bond gained preserved thinking (reasoning replayed through tool
152
161
  * loops), so kimi-k3 is the Moonshot pick.)
162
+ * (re-verified 2026-09-10 on platform.kimi.ai/docs/models: kimi-k2.5 was
163
+ * officially discontinued 2026-08-31 — calls 404 — so its entry is now
164
+ * `disabled: true`. kimi-k3 / kimi-k2.6 / kimi-k2.7-code all still Active.)
153
165
  * - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
154
166
  * minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
155
167
  * - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
@@ -201,6 +213,11 @@ const WEEKDAYS_UTC = [1, 2, 3, 4, 5];
201
213
  * DeepInfra re-host zai-org/GLM-5.3-Flash, which bills exactly the list card
202
214
  * ($0.15/$0.50, cache read 0.2x = $0.03 — verified live 2026-08-27), mapped
203
215
  * in molecule-dev's ZHIPU_US_MODEL_MAP.)
216
+ * (re-verified 2026-09-10 on https://docs.z.ai/guides/overview/pricing: the
217
+ * promo ended on schedule 2026-09-09 24:00 UTC+8 and the page prints the list
218
+ * card again — $0.15/$0.50, cached input $0.03, no promo footnote. models.dev
219
+ * still carries the expired promo rate, so the KNOWN_DIVERGENCES entry now
220
+ * runs to a fixed 2026-12-09 re-verify date.)
204
221
  *
205
222
  * Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
206
223
  * where the provider doesn't publish one; the provider sources above verify
@@ -657,11 +674,11 @@ export const MODELS = [
657
674
  // Cached input 0.1× input, cache write 1.25× input (see section note).
658
675
  cacheReadPricePerMTok: 0.02,
659
676
  cacheWritePricePerMTok: 0.25,
660
- // The free-tier PLAN default (2026-08-18) — us-only OpenAI, so this carve-out
661
- // is what lets `freeTierAllows` permit it for plan mode without making it
662
- // outright `freeTier` (which would free it in every mode). Mirrors how
663
- // minimax-m3 was scoped as the prior free planner.
664
- freeTierRegions: ['us'],
677
+ // No freeTierRegions: Luna was the free-tier PLAN default 2026-08-18 →
678
+ // 2026-09-10 via this carve-out; the free planner is now `deepseek-flash`
679
+ // (outright freeTier), so the carve-out is removed — a stale one would
680
+ // claim a free-tier relationship Luna no longer has (see deepseek-v4-pro's
681
+ // note for the same removal pattern).
665
682
  // Not published — best-effort estimate.
666
683
  knowledgeCutoff: '2026-03-01',
667
684
  },
@@ -1140,7 +1157,25 @@ export const MODELS = [
1140
1157
  // DeepSeek
1141
1158
  // Verified: https://api-docs.deepseek.com/quick_start/pricing
1142
1159
  // https://api-docs.deepseek.com/guides/thinking_mode
1143
- // https://api-docs.deepseek.com/updates/ (2026-08-18)
1160
+ // https://api-docs.deepseek.com/updates/ (re-read 2026-09-10)
1161
+ // 2026-09-10: DeepSeek-V4.1-Flash RELEASED (updates page, same day) under the
1162
+ // NEW id `deepseek-flash` — "Change the model name to deepseek-flash to call
1163
+ // the latest V4.1 Flash model." Its card CUT the flash rates to a third of
1164
+ // the 2026-08-16 rise:
1165
+ // flash (V4.1) off-peak miss 0.15 / hit 0.003 / out 0.6 (peak ×2, same
1166
+ // Mon-Fri UTC windows; 1M ctx / 384K out; concurrency 2500)
1167
+ // and the announcement claims "native multimodal visual understanding", so
1168
+ // the new entry carries supportsVision. It also "outperforms V4 Pro on
1169
+ // performance, cost, and speed", and two retirement facts landed with it:
1170
+ // 1. V4 Flash + V4 Flash Vision Exp are RETIRED; the legacy ids
1171
+ // `deepseek-v4-flash` / `deepseek-v4-flash-vision-exp` are "temporarily
1172
+ // routed" to V4.1 Flash and billed at Flash prices — so the v4-flash
1173
+ // entry below is repriced to the V4.1 card (its id still answers), and
1174
+ // leans on that "temporary" routing until molecule-dev's default ids
1175
+ // move to `deepseek-flash`.
1176
+ // 2. From 2026-09-14 12:00 Beijing (04:00 UTC), until V4.1 Pro ships, ALL
1177
+ // `deepseek-v4-pro` requests route to V4.1 Flash at Flash prices —
1178
+ // staged as `scheduledPricing` on the pro entry below.
1144
1179
  // 2026-08-13: V4-Pro GA — and with it the price rise that the "coming soon"
1145
1180
  // note below had been waiting on. It was staged as `scheduledPricing`
1146
1181
  // effective 2026-08-16T16:00Z; that instant has PASSED and the rates are now
@@ -1242,8 +1277,32 @@ export const MODELS = [
1242
1277
  ],
1243
1278
  multiplier: 2,
1244
1279
  },
1280
+ // ANNOUNCED 2026-09-10 (updates page): from 2026-09-14 12:00 Beijing time
1281
+ // (04:00 UTC), and until V4.1 Pro ships, every `deepseek-v4-pro` request is
1282
+ // routed to V4.1 Flash and billed at the V4.1 FLASH price — so the base
1283
+ // rates become the flash card below (the peak windows are identical, so
1284
+ // they carry through unchanged). The US `regionPricing` above is DeepInfra's
1285
+ // own card for its V4-Pro copy and is NOT touched by the native routing.
1286
+ // Fold into the base fields once the instant has passed (the freshness
1287
+ // gate gives models.dev its scheduled-landing grace meanwhile).
1288
+ scheduledPricing: {
1289
+ effectiveFrom: '2026-09-14T04:00:00Z',
1290
+ inputPricePerMTok: 0.15,
1291
+ outputPricePerMTok: 0.6,
1292
+ cacheReadPricePerMTok: 0.003,
1293
+ cacheWritePricePerMTok: 0.15,
1294
+ source: 'https://api-docs.deepseek.com/updates/ (2026-09-10): deepseek-v4-pro → V4.1 Flash at Flash prices from 2026-09-14 12:00 Beijing',
1295
+ },
1245
1296
  // Not published by DeepSeek — best-effort estimate.
1246
1297
  knowledgeCutoff: '2025-07-01',
1298
+ // Deprecated the day it stops being itself: DeepSeek's own testing has
1299
+ // V4.1 Flash "comprehensively surpassing" Pro on performance/cost/speed,
1300
+ // and from 2026-09-14 12:00 Beijing every `deepseek-v4-pro` request routes
1301
+ // to V4.1 Flash at Flash prices (see scheduledPricing above) until V4.1 Pro
1302
+ // ships — at which point this becomes a new entry's succession problem, not
1303
+ // this one's. Until the 14th it still serves real V4-Pro-0813 weights at
1304
+ // the Pro card, so it stays selectable for existing selections.
1305
+ deprecatedAt: '2026-09-14',
1247
1306
  },
1248
1307
  {
1249
1308
  id: 'deepseek-v4-flash',
@@ -1260,37 +1319,62 @@ export const MODELS = [
1260
1319
  supportsVision: false,
1261
1320
  supportsPromptCaching: true,
1262
1321
  supportsTools: true,
1263
- // Free-tier default: cheapest model + a fast, non-thinking tool-calling
1264
- // executor — the model the IDE picks when none is chosen. Exactly one model
1265
- // in this catalog may carry freeTier (enforced by lookup.test.ts).
1266
- freeTier: true,
1267
- // Off-peak rates; peak is `peakPricing.multiplier` × these (see below).
1268
- inputPricePerMTok: 0.22,
1269
- outputPricePerMTok: 0.66,
1322
+ // freeTier moved to `deepseek-flash` (2026-09-10) together with
1323
+ // molecule-dev's default ids — the free tier follows the go-forward
1324
+ // evergreen id, not a retired name whose routing DeepSeek calls
1325
+ // "temporary". Exactly one model in this catalog may carry freeTier
1326
+ // (enforced by lookup.test.ts).
1327
+ // The V4.1 Flash card (2026-09-10). This legacy id is RETIRED and
1328
+ // "temporarily routed" to V4.1 Flash, billed at Flash prices — so these ARE
1329
+ // the rates the id answers with today (they replace the 0.22/0.66/0.007
1330
+ // card the same id served from 2026-08-16; same in-place succession as the
1331
+ // 2026-07-31 0731 re-post-train). Peak is `peakPricing.multiplier` × these.
1332
+ inputPricePerMTok: 0.15,
1333
+ outputPricePerMTok: 0.6,
1270
1334
  // DeepSeek automatic context cache: absolute cache-hit price ($/M).
1271
- cacheReadPricePerMTok: 0.007,
1335
+ cacheReadPricePerMTok: 0.003,
1272
1336
  // DeepSeek charges no cache-write premium — write bills at input.
1273
- cacheWritePricePerMTok: 0.22,
1337
+ cacheWritePricePerMTok: 0.15,
1274
1338
  // US (DeepInfra) DEFAULT as of 2026-08-16 — flipped from CN when DeepSeek's
1275
1339
  // rise landed (owner decision 2026-08-14). CN was cheaper on real traffic
1276
1340
  // only because of its cache reads; the rise takes those from $0.0028 to
1277
- // $0.007 (peak $0.014) against DeepInfra's flat $0.016, which is no longer
1278
- // enough to carry the 1.6-3.1x it now loses on fresh input and output. On
1341
+ // $0.007 (peak $0.014) against DeepInfra's flat $0.015, which is no longer
1342
+ // enough to carry the 3.7x CN now loses on BOTH fresh input (0.22 vs 0.06)
1343
+ // and output (0.66 vs 0.18). On
1279
1344
  // the agentic mix this model actually serves (~94% cache hits) US is
1280
- // cheaper at EVERY hour: 0.114c/turn flat vs 0.152c off-peak and 0.303c at
1345
+ // cheaper at EVERY hour: 0.103c/turn flat vs 0.152c off-peak and 0.303c at
1281
1346
  // peak. It is also flat-rate, so free-tier cost stops varying by Beijing
1282
1347
  // business hours. Re-derive if the cache-hit ratio drops much below ~90%,
1283
1348
  // where CN's cheaper reads start winning again. This deliberately splits
1284
1349
  // the plan/execute pair across regions — Pro stays CN because its US
1285
1350
  // re-host is ~2.3x its own native rate even after the rise.
1351
+ // Re-derived 2026-09-10 against the repriced CN card: CN's cache hits are
1352
+ // now 5× cheaper than the re-host's ($0.003 vs $0.015), so CN wins on
1353
+ // blended input (0.0118 vs 0.0177 $/MTok off-peak) — but US wins on output
1354
+ // at 0.18 vs 0.6/1.2, and CN only takes a turn whose output is under ~1.4%
1355
+ // of its input. At real agentic output ratios (5–25%) US still wins every
1356
+ // hour, so the US default holds.
1357
+ // MIGRATED 2026-09-10: molecule-dev's default literals (EXECUTOR_MODEL /
1358
+ // EXECUTOR_MODEL_CUSTOM / DEFAULT_CHAT_MODEL / COMPACTION_MODEL /
1359
+ // COMMIT_MESSAGE_MODEL / FREE_TIER_MODELS.execute) now point at
1360
+ // `deepseek-flash` (cn-native). This entry stays selectable — not
1361
+ // `supersededBy` — because its US leg (DeepInfra 0731 at $0.06/$0.18 flat,
1362
+ // older-but-solid weights DeepInfra keeps serving) is still the cheapest
1363
+ // flash serving there is and saved selections should keep working; it is
1364
+ // merely hidden from offering via `deprecatedAt`.
1286
1365
  regions: ['us', 'cn'],
1287
- // US = DeepInfra, verified 2026-08-13 against the id the bond actually
1366
+ // US = DeepInfra, verified 2026-09-06 against the id the bond actually
1288
1367
  // sends: `deepseek-ai/DeepSeek-V4-Flash-0731`, the official release that
1289
1368
  // supersedes the preview weights still served under the un-dated id
1290
- // (cents_per_input_token 0.000008, cents_per_output_token 0.000018,
1291
- // rate_per_input_token_cached 0.2 → cache read = 0.2 × input).
1369
+ // (cents_per_input_token 0.000006, cents_per_output_token 0.000018,
1370
+ // rate_per_input_token_cached 0.25 → cache read = 0.25 × input).
1371
+ // DeepInfra CUT the 0731 input rate 0.08 → 0.06 (cache read 0.016 →
1372
+ // 0.015); output held at 0.18. The un-dated `DeepSeek-V4-Flash` id moved
1373
+ // the other way (0.09 input, cache 0.2× = 0.018), so the two ids no longer
1374
+ // price alike — this entry tracks the dated one the modelMap sends, and the
1375
+ // gap widens the case for the US default this model already carries.
1292
1376
  regionPricing: {
1293
- us: { inputPricePerMTok: 0.08, outputPricePerMTok: 0.18, cacheReadPricePerMTok: 0.016 },
1377
+ us: { inputPricePerMTok: 0.06, outputPricePerMTok: 0.18, cacheReadPricePerMTok: 0.015 },
1294
1378
  },
1295
1379
  // Peak-hour surcharge live since 2026-08-16T16:00Z, Mon-Fri (see
1296
1380
  // deepseek-v4-pro). It applies to the NATIVE CN card only — this model
@@ -1304,6 +1388,67 @@ export const MODELS = [
1304
1388
  },
1305
1389
  // Not published by DeepSeek — best-effort estimate.
1306
1390
  knowledgeCutoff: '2025-07-01',
1391
+ // Retired upstream 2026-09-10 (V4 flash ids; legacy name "temporarily
1392
+ // routed" to V4.1 Flash). See the regions note above for why this is
1393
+ // deprecatedAt — hidden from offering, still selectable — rather than
1394
+ // supersededBy.
1395
+ deprecatedAt: '2026-09-10',
1396
+ },
1397
+ {
1398
+ // Released 2026-09-10 (updates page) — the go-forward EVERGREEN flash id.
1399
+ // Same rates the repriced legacy entry above carries natively, but under
1400
+ // the id DeepSeek actually wants called, with the multimodal surface the
1401
+ // announcement claims ("native multimodal visual understanding") and none
1402
+ // of the retirement ambiguity of a "temporarily routed" legacy name.
1403
+ id: 'deepseek-flash',
1404
+ provider: 'deepseek',
1405
+ label: 'DeepSeek V4.1 Flash',
1406
+ description: 'Newest ultra-cheap & fast flash — vision-capable agentic coding',
1407
+ contextWindow: 1_000_000,
1408
+ maxOutputTokens: 384_000,
1409
+ // Non-thinking (see section note) — same bond treatment as the family:
1410
+ // the executor tool loop does not replay reasoning_content.
1411
+ supportsThinking: false,
1412
+ thinkingBudgetTokens: 0,
1413
+ thinkingConfigurable: false,
1414
+ supportsVision: true,
1415
+ supportsPromptCaching: true,
1416
+ supportsTools: true,
1417
+ // Free-tier default (moved from deepseek-v4-flash 2026-09-10): the
1418
+ // cheapest capable executor on the go-forward id — fast, non-thinking,
1419
+ // tool-calling, and now vision-capable. Serving cost is ~2× the retired
1420
+ // US leg on output-heavy turns (CN 0.6/1.2 output vs DeepInfra's flat
1421
+ // 0.18) — tiers.ts's free-turn budget math derives from this entry.
1422
+ // Exactly one model in this catalog may carry freeTier (lookup.test.ts).
1423
+ freeTier: true,
1424
+ // Off-peak V4.1 Flash card; peak is `peakPricing.multiplier` × these.
1425
+ inputPricePerMTok: 0.15,
1426
+ outputPricePerMTok: 0.6,
1427
+ // DeepSeek automatic context cache: absolute cache-hit price ($/M).
1428
+ cacheReadPricePerMTok: 0.003,
1429
+ // DeepSeek charges no cache-write premium — write bills at input.
1430
+ cacheWritePricePerMTok: 0.15,
1431
+ // PINNED NATIVE — no US re-host is wired. DeepInfra does serve
1432
+ // `deepseek-ai/DeepSeek-V4.1-Flash` (listed 2026-09-10) at $0.30/$1.20,
1433
+ // cache read 0.02× input = $0.006 — exactly 2× this native off-peak card —
1434
+ // but offering a us region needs a molecule-dev US_MODEL_MAP entry
1435
+ // (region-model-maps.ts) plus a regionPricing row here, and that pair is a
1436
+ // deliberate molecule-dev change (it also can't beat this card on the
1437
+ // agentic mix, unlike the legacy 0731 re-host). Single-entry ['cn'] pins
1438
+ // dispatch to the native endpoint regardless of per-model region choice.
1439
+ regions: ['cn'],
1440
+ // Same peak windows as the family — the V4.1 card footnote is verbatim the
1441
+ // 2026-08-31 sentence: 01:00-04:00 and 06:00-10:00 UTC, Mon-Fri, ×2.
1442
+ peakPricing: {
1443
+ windows: [
1444
+ { startMinuteUtc: 60, endMinuteUtc: 240, daysOfWeekUtc: WEEKDAYS_UTC },
1445
+ { startMinuteUtc: 360, endMinuteUtc: 600, daysOfWeekUtc: WEEKDAYS_UTC },
1446
+ ],
1447
+ multiplier: 2,
1448
+ },
1449
+ // Not published by DeepSeek — best-effort estimate (family estimate; the
1450
+ // V4.1 announcement lists benchmarks but no training cutoff).
1451
+ knowledgeCutoff: '2025-07-01',
1307
1452
  },
1308
1453
  // ---------------------------------------------------------------------------
1309
1454
  // Moonshot (Kimi)
@@ -1464,8 +1609,15 @@ export const MODELS = [
1464
1609
  knowledgeCutoff: '2024-04-01',
1465
1610
  // Two generations behind. `supersededBy` names the current selectable Kimi
1466
1611
  // (k3) rather than the also-superseded k2.6, so a saved selection resolves
1467
- // forward in one hop. Still served upstream; stays priceable.
1612
+ // forward in one hop. Moonshot RETIRED kimi-k2.5 on 2026-08-31 —
1613
+ // platform.kimi.ai/docs/models says calls 404, and a live 2026-09-10
1614
+ // dispatch probe confirmed the native host refusing it. The DeepInfra
1615
+ // re-host still answered inference on 2026-09-10 but is DELISTED from its
1616
+ // model catalog (detail page 404, /models/list unpruned) — borrowed time.
1617
+ // Disabled, never deleted: historical usage stays priceable and saved
1618
+ // selections still resolve forward.
1468
1619
  deprecatedAt: '2026-04-01',
1620
+ disabled: true,
1469
1621
  supersededBy: 'kimi-k3',
1470
1622
  },
1471
1623
  // ---------------------------------------------------------------------------
@@ -1830,9 +1982,11 @@ export const MODELS = [
1830
1982
  supportsPromptCaching: true,
1831
1983
  supportsTools: true,
1832
1984
  webSearchToolType: 'web_search',
1833
- // LIST card. A 50%-off launch promo ($0.075/$0.25, cached $0.015) runs to
1834
- // 2026-09-09 24:00 UTC+8; billed at list so metering never under-charges —
1835
- // the promo is a KNOWN_DIVERGENCES entry in check-model-freshness.
1985
+ // List card, re-verified 2026-09-10 on the Z.ai pricing page after the
1986
+ // 50%-off launch promo ($0.075/$0.25, cached $0.015) ended on schedule
1987
+ // 2026-09-09 24:00 UTC+8 — this IS what the provider bills now; models.dev
1988
+ // still carries the expired promo rate (KNOWN_DIVERGENCES entry in
1989
+ // check-model-freshness, re-verify date 2026-12-09).
1836
1990
  inputPricePerMTok: 0.15,
1837
1991
  outputPricePerMTok: 0.5,
1838
1992
  // GLM context cache: read 0.2× input, no write premium.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@molecule/api-resource-ai-models",
3
- "version": "1.5.0",
3
+ "version": "1.6.0",
4
4
  "description": "AI model catalog — server-side source of truth plus an authentication-gated discovery endpoint",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",
@@ -32,7 +32,7 @@
32
32
  "@molecule/api-resource": "1.0.1",
33
33
  "@types/node": "26.1.2",
34
34
  "typescript": "6.0.3",
35
- "vitest": "4.1.10"
35
+ "vitest": "4.1.11"
36
36
  },
37
37
  "peerDependencies": {
38
38
  "@molecule/api-bond": "^1.0.1",