@molecule/api-resource-ai-models 1.5.1 → 1.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -3,7 +3,7 @@ AUTO-GENERATED — DO NOT EDIT THIS FILE.
3
3
  Generated by `mlcl sync-docs` from the package's src/index.ts JSDoc + mlcl/registry.json.
4
4
  Edits here are overwritten on the next commit (molecule's pre-commit hook regenerates).
5
5
  To change this document, edit the module-level JSDoc in src/index.ts.
6
- Generated: 2026-09-06T11:38:29.688Z
6
+ Generated: 2026-09-10T11:48:36.810Z
7
7
  -->
8
8
 
9
9
  # @molecule/api-resource-ai-models
@@ -860,20 +860,26 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
860
860
  not modeled; reasoning_effort low|medium|high default high, image input;
861
861
  grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
862
862
  grok-code-fast-1 no longer listed — retires 2026-08-15)
863
- - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
864
- 2026-09-06; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
865
- never in this catalog. The V4-Pro-GA price RISE effective 2026-08-16T16:00Z
866
- has LANDED and is folded into the base fields, along with the peak-hour 2×
867
- the same card introduced; every rate re-read on the card 2026-08-31 and
868
- unchanged. The card now qualifies the windows BY DAY — "01:00 - 04:00 and
869
- 06:00 - 10:00 UTC, Monday through Friday (all other hours are off-peak)" —
870
- where on 2026-08-18 it carried no day qualifier at all, so the windows are
863
+ - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing +
864
+ /updates/ (verified 2026-09-10; legacy deepseek-chat/-reasoner ids fully
865
+ retired 2026-07-24 — never in this catalog. The 2026-09-10 re-read caught
866
+ the V4.1-Flash release DAY-OF: new evergreen id `deepseek-flash` at
867
+ off-peak miss $0.15 / hit $0.003 / out $0.6 (peak ×2, same Mon-Fri UTC
868
+ windows, 1M ctx / 384K out), the V4 flash + vision-exp ids RETIRED with
869
+ their names "temporarily routed" to V4.1 Flash at Flash prices (the
870
+ `deepseek-v4-flash` entry is repriced to that card), and `deepseek-v4-pro`
871
+ announced to route to V4.1 Flash at Flash prices from 2026-09-14 12:00
872
+ Beijing (staged as `scheduledPricing` there); the pro card itself is
873
+ unchanged (V4-Pro-0813, 0.66/1.98/0.022 off-peak). The card qualifies the
874
+ peak windows BY DAY — "01:00 - 04:00 and 06:00 - 10:00 UTC, Monday through
875
+ Friday (all other hours are off-peak)" — so the windows are
871
876
  `daysOfWeekUtc`-restricted rather than daily. `deepseek-v4-flash-vision-exp`
872
- also appears on the card at flash's rates: EXPERIMENTAL and vision-only-new,
873
- deliberately not catalogued. The 2026-09-06 re-read found every native rate
874
- unchanged; what moved was the US re-host — DeepInfra cut
875
- `DeepSeek-V4-Flash-0731` from $0.08 to $0.06 input, cache read $0.016 to
876
- $0.015, output held at $0.18. See that entry's `regionPricing`.)
877
+ is retired with the rest of V4 flash and stays deliberately uncatalogued.
878
+ The US re-host rates (DeepInfra) are unchanged from the 2026-09-06 read —
879
+ `DeepSeek-V4-Flash-0731` at $0.06/$0.18, cache read $0.015; DeepInfra also
880
+ listed `deepseek-ai/DeepSeek-V4.1-Flash` on 2026-09-10 at $0.30/$1.20
881
+ (cache read 0.02× input) — 2× native, not wired. See each entry's
882
+ `regionPricing`.)
877
883
  - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
878
884
  the US re-host (kimi-k3 flagship 2026-07-16
879
885
  — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
@@ -881,6 +887,9 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
881
887
  constraint that kept kimi-k2.7-code out. BOTH are now in the catalog: the
882
888
  moonshot bond gained preserved thinking (reasoning replayed through tool
883
889
  loops), so kimi-k3 is the Moonshot pick.)
890
+ (re-verified 2026-09-10 on platform.kimi.ai/docs/models: kimi-k2.5 was
891
+ officially discontinued 2026-08-31 — calls 404 — so its entry is now
892
+ `disabled: true`. kimi-k3 / kimi-k2.6 / kimi-k2.7-code all still Active.)
884
893
  - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
885
894
  minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
886
895
  - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
@@ -932,6 +941,11 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
932
941
  DeepInfra re-host zai-org/GLM-5.3-Flash, which bills exactly the list card
933
942
  ($0.15/$0.50, cache read 0.2x = $0.03 — verified live 2026-08-27), mapped
934
943
  in molecule-dev's ZHIPU_US_MODEL_MAP.)
944
+ (re-verified 2026-09-10 on https://docs.z.ai/guides/overview/pricing: the
945
+ promo ended on schedule 2026-09-09 24:00 UTC+8 and the page prints the list
946
+ card again — $0.15/$0.50, cached input $0.03, no promo footnote. models.dev
947
+ still carries the expired promo rate, so the KNOWN_DIVERGENCES entry now
948
+ runs to a fixed 2026-12-09 re-verify date.)
935
949
 
936
950
  Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
937
951
  where the provider doesn't publish one; the provider sources above verify
package/dist/models.d.ts CHANGED
@@ -128,20 +128,26 @@ import type { ModelDefinition } from './types.js';
128
128
  * not modeled; reasoning_effort low|medium|high default high, image input;
129
129
  * grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
130
130
  * grok-code-fast-1 no longer listed — retires 2026-08-15)
131
- * - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
132
- * 2026-09-06; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
133
- * never in this catalog. The V4-Pro-GA price RISE effective 2026-08-16T16:00Z
134
- * has LANDED and is folded into the base fields, along with the peak-hour 2×
135
- * the same card introduced; every rate re-read on the card 2026-08-31 and
136
- * unchanged. The card now qualifies the windows BY DAY — "01:00 - 04:00 and
137
- * 06:00 - 10:00 UTC, Monday through Friday (all other hours are off-peak)" —
138
- * where on 2026-08-18 it carried no day qualifier at all, so the windows are
131
+ * - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing +
132
+ * /updates/ (verified 2026-09-10; legacy deepseek-chat/-reasoner ids fully
133
+ * retired 2026-07-24 — never in this catalog. The 2026-09-10 re-read caught
134
+ * the V4.1-Flash release DAY-OF: new evergreen id `deepseek-flash` at
135
+ * off-peak miss $0.15 / hit $0.003 / out $0.6 (peak ×2, same Mon-Fri UTC
136
+ * windows, 1M ctx / 384K out), the V4 flash + vision-exp ids RETIRED with
137
+ * their names "temporarily routed" to V4.1 Flash at Flash prices (the
138
+ * `deepseek-v4-flash` entry is repriced to that card), and `deepseek-v4-pro`
139
+ * announced to route to V4.1 Flash at Flash prices from 2026-09-14 12:00
140
+ * Beijing (staged as `scheduledPricing` there); the pro card itself is
141
+ * unchanged (V4-Pro-0813, 0.66/1.98/0.022 off-peak). The card qualifies the
142
+ * peak windows BY DAY — "01:00 - 04:00 and 06:00 - 10:00 UTC, Monday through
143
+ * Friday (all other hours are off-peak)" — so the windows are
139
144
  * `daysOfWeekUtc`-restricted rather than daily. `deepseek-v4-flash-vision-exp`
140
- * also appears on the card at flash's rates: EXPERIMENTAL and vision-only-new,
141
- * deliberately not catalogued. The 2026-09-06 re-read found every native rate
142
- * unchanged; what moved was the US re-host — DeepInfra cut
143
- * `DeepSeek-V4-Flash-0731` from $0.08 to $0.06 input, cache read $0.016 to
144
- * $0.015, output held at $0.18. See that entry's `regionPricing`.)
145
+ * is retired with the rest of V4 flash and stays deliberately uncatalogued.
146
+ * The US re-host rates (DeepInfra) are unchanged from the 2026-09-06 read —
147
+ * `DeepSeek-V4-Flash-0731` at $0.06/$0.18, cache read $0.015; DeepInfra also
148
+ * listed `deepseek-ai/DeepSeek-V4.1-Flash` on 2026-09-10 at $0.30/$1.20
149
+ * (cache read 0.02× input) — 2× native, not wired. See each entry's
150
+ * `regionPricing`.)
145
151
  * - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
146
152
  * the US re-host (kimi-k3 flagship 2026-07-16
147
153
  * — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
@@ -149,6 +155,9 @@ import type { ModelDefinition } from './types.js';
149
155
  * constraint that kept kimi-k2.7-code out. BOTH are now in the catalog: the
150
156
  * moonshot bond gained preserved thinking (reasoning replayed through tool
151
157
  * loops), so kimi-k3 is the Moonshot pick.)
158
+ * (re-verified 2026-09-10 on platform.kimi.ai/docs/models: kimi-k2.5 was
159
+ * officially discontinued 2026-08-31 — calls 404 — so its entry is now
160
+ * `disabled: true`. kimi-k3 / kimi-k2.6 / kimi-k2.7-code all still Active.)
152
161
  * - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
153
162
  * minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
154
163
  * - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
@@ -200,6 +209,11 @@ import type { ModelDefinition } from './types.js';
200
209
  * DeepInfra re-host zai-org/GLM-5.3-Flash, which bills exactly the list card
201
210
  * ($0.15/$0.50, cache read 0.2x = $0.03 — verified live 2026-08-27), mapped
202
211
  * in molecule-dev's ZHIPU_US_MODEL_MAP.)
212
+ * (re-verified 2026-09-10 on https://docs.z.ai/guides/overview/pricing: the
213
+ * promo ended on schedule 2026-09-09 24:00 UTC+8 and the page prints the list
214
+ * card again — $0.15/$0.50, cached input $0.03, no promo footnote. models.dev
215
+ * still carries the expired promo rate, so the KNOWN_DIVERGENCES entry now
216
+ * runs to a fixed 2026-12-09 re-verify date.)
203
217
  *
204
218
  * Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
205
219
  * where the provider doesn't publish one; the provider sources above verify
@@ -1 +1 @@
1
- {"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAQjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAoMG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EA6sDnC,CAAA"}
1
+ {"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAQjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkNG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAq1DnC,CAAA"}
package/dist/models.js CHANGED
@@ -132,20 +132,26 @@ const WEEKDAYS_UTC = [1, 2, 3, 4, 5];
132
132
  * not modeled; reasoning_effort low|medium|high default high, image input;
133
133
  * grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
134
134
  * grok-code-fast-1 no longer listed — retires 2026-08-15)
135
- * - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
136
- * 2026-09-06; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
137
- * never in this catalog. The V4-Pro-GA price RISE effective 2026-08-16T16:00Z
138
- * has LANDED and is folded into the base fields, along with the peak-hour 2×
139
- * the same card introduced; every rate re-read on the card 2026-08-31 and
140
- * unchanged. The card now qualifies the windows BY DAY — "01:00 - 04:00 and
141
- * 06:00 - 10:00 UTC, Monday through Friday (all other hours are off-peak)" —
142
- * where on 2026-08-18 it carried no day qualifier at all, so the windows are
135
+ * - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing +
136
+ * /updates/ (verified 2026-09-10; legacy deepseek-chat/-reasoner ids fully
137
+ * retired 2026-07-24 — never in this catalog. The 2026-09-10 re-read caught
138
+ * the V4.1-Flash release DAY-OF: new evergreen id `deepseek-flash` at
139
+ * off-peak miss $0.15 / hit $0.003 / out $0.6 (peak ×2, same Mon-Fri UTC
140
+ * windows, 1M ctx / 384K out), the V4 flash + vision-exp ids RETIRED with
141
+ * their names "temporarily routed" to V4.1 Flash at Flash prices (the
142
+ * `deepseek-v4-flash` entry is repriced to that card), and `deepseek-v4-pro`
143
+ * announced to route to V4.1 Flash at Flash prices from 2026-09-14 12:00
144
+ * Beijing (staged as `scheduledPricing` there); the pro card itself is
145
+ * unchanged (V4-Pro-0813, 0.66/1.98/0.022 off-peak). The card qualifies the
146
+ * peak windows BY DAY — "01:00 - 04:00 and 06:00 - 10:00 UTC, Monday through
147
+ * Friday (all other hours are off-peak)" — so the windows are
143
148
  * `daysOfWeekUtc`-restricted rather than daily. `deepseek-v4-flash-vision-exp`
144
- * also appears on the card at flash's rates: EXPERIMENTAL and vision-only-new,
145
- * deliberately not catalogued. The 2026-09-06 re-read found every native rate
146
- * unchanged; what moved was the US re-host — DeepInfra cut
147
- * `DeepSeek-V4-Flash-0731` from $0.08 to $0.06 input, cache read $0.016 to
148
- * $0.015, output held at $0.18. See that entry's `regionPricing`.)
149
+ * is retired with the rest of V4 flash and stays deliberately uncatalogued.
150
+ * The US re-host rates (DeepInfra) are unchanged from the 2026-09-06 read —
151
+ * `DeepSeek-V4-Flash-0731` at $0.06/$0.18, cache read $0.015; DeepInfra also
152
+ * listed `deepseek-ai/DeepSeek-V4.1-Flash` on 2026-09-10 at $0.30/$1.20
153
+ * (cache read 0.02× input) — 2× native, not wired. See each entry's
154
+ * `regionPricing`.)
149
155
  * - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
150
156
  * the US re-host (kimi-k3 flagship 2026-07-16
151
157
  * — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
@@ -153,6 +159,9 @@ const WEEKDAYS_UTC = [1, 2, 3, 4, 5];
153
159
  * constraint that kept kimi-k2.7-code out. BOTH are now in the catalog: the
154
160
  * moonshot bond gained preserved thinking (reasoning replayed through tool
155
161
  * loops), so kimi-k3 is the Moonshot pick.)
162
+ * (re-verified 2026-09-10 on platform.kimi.ai/docs/models: kimi-k2.5 was
163
+ * officially discontinued 2026-08-31 — calls 404 — so its entry is now
164
+ * `disabled: true`. kimi-k3 / kimi-k2.6 / kimi-k2.7-code all still Active.)
156
165
  * - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
157
166
  * minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
158
167
  * - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
@@ -204,6 +213,11 @@ const WEEKDAYS_UTC = [1, 2, 3, 4, 5];
204
213
  * DeepInfra re-host zai-org/GLM-5.3-Flash, which bills exactly the list card
205
214
  * ($0.15/$0.50, cache read 0.2x = $0.03 — verified live 2026-08-27), mapped
206
215
  * in molecule-dev's ZHIPU_US_MODEL_MAP.)
216
+ * (re-verified 2026-09-10 on https://docs.z.ai/guides/overview/pricing: the
217
+ * promo ended on schedule 2026-09-09 24:00 UTC+8 and the page prints the list
218
+ * card again — $0.15/$0.50, cached input $0.03, no promo footnote. models.dev
219
+ * still carries the expired promo rate, so the KNOWN_DIVERGENCES entry now
220
+ * runs to a fixed 2026-12-09 re-verify date.)
207
221
  *
208
222
  * Knowledge-cutoff dates on non-Anthropic entries are best-effort estimates
209
223
  * where the provider doesn't publish one; the provider sources above verify
@@ -660,11 +674,11 @@ export const MODELS = [
660
674
  // Cached input 0.1× input, cache write 1.25× input (see section note).
661
675
  cacheReadPricePerMTok: 0.02,
662
676
  cacheWritePricePerMTok: 0.25,
663
- // The free-tier PLAN default (2026-08-18) — us-only OpenAI, so this carve-out
664
- // is what lets `freeTierAllows` permit it for plan mode without making it
665
- // outright `freeTier` (which would free it in every mode). Mirrors how
666
- // minimax-m3 was scoped as the prior free planner.
667
- freeTierRegions: ['us'],
677
+ // No freeTierRegions: Luna was the free-tier PLAN default 2026-08-18 →
678
+ // 2026-09-10 via this carve-out; the free planner is now `deepseek-flash`
679
+ // (outright freeTier), so the carve-out is removed — a stale one would
680
+ // claim a free-tier relationship Luna no longer has (see deepseek-v4-pro's
681
+ // note for the same removal pattern).
668
682
  // Not published — best-effort estimate.
669
683
  knowledgeCutoff: '2026-03-01',
670
684
  },
@@ -1143,7 +1157,25 @@ export const MODELS = [
1143
1157
  // DeepSeek
1144
1158
  // Verified: https://api-docs.deepseek.com/quick_start/pricing
1145
1159
  // https://api-docs.deepseek.com/guides/thinking_mode
1146
- // https://api-docs.deepseek.com/updates/ (2026-08-18)
1160
+ // https://api-docs.deepseek.com/updates/ (re-read 2026-09-10)
1161
+ // 2026-09-10: DeepSeek-V4.1-Flash RELEASED (updates page, same day) under the
1162
+ // NEW id `deepseek-flash` — "Change the model name to deepseek-flash to call
1163
+ // the latest V4.1 Flash model." Its card CUT the flash rates to a third of
1164
+ // the 2026-08-16 rise:
1165
+ // flash (V4.1) off-peak miss 0.15 / hit 0.003 / out 0.6 (peak ×2, same
1166
+ // Mon-Fri UTC windows; 1M ctx / 384K out; concurrency 2500)
1167
+ // and the announcement claims "native multimodal visual understanding", so
1168
+ // the new entry carries supportsVision. It also "outperforms V4 Pro on
1169
+ // performance, cost, and speed", and two retirement facts landed with it:
1170
+ // 1. V4 Flash + V4 Flash Vision Exp are RETIRED; the legacy ids
1171
+ // `deepseek-v4-flash` / `deepseek-v4-flash-vision-exp` are "temporarily
1172
+ // routed" to V4.1 Flash and billed at Flash prices — so the v4-flash
1173
+ // entry below is repriced to the V4.1 card (its id still answers), and
1174
+ // leans on that "temporary" routing until molecule-dev's default ids
1175
+ // move to `deepseek-flash`.
1176
+ // 2. From 2026-09-14 12:00 Beijing (04:00 UTC), until V4.1 Pro ships, ALL
1177
+ // `deepseek-v4-pro` requests route to V4.1 Flash at Flash prices —
1178
+ // staged as `scheduledPricing` on the pro entry below.
1147
1179
  // 2026-08-13: V4-Pro GA — and with it the price rise that the "coming soon"
1148
1180
  // note below had been waiting on. It was staged as `scheduledPricing`
1149
1181
  // effective 2026-08-16T16:00Z; that instant has PASSED and the rates are now
@@ -1245,8 +1277,32 @@ export const MODELS = [
1245
1277
  ],
1246
1278
  multiplier: 2,
1247
1279
  },
1280
+ // ANNOUNCED 2026-09-10 (updates page): from 2026-09-14 12:00 Beijing time
1281
+ // (04:00 UTC), and until V4.1 Pro ships, every `deepseek-v4-pro` request is
1282
+ // routed to V4.1 Flash and billed at the V4.1 FLASH price — so the base
1283
+ // rates become the flash card below (the peak windows are identical, so
1284
+ // they carry through unchanged). The US `regionPricing` above is DeepInfra's
1285
+ // own card for its V4-Pro copy and is NOT touched by the native routing.
1286
+ // Fold into the base fields once the instant has passed (the freshness
1287
+ // gate gives models.dev its scheduled-landing grace meanwhile).
1288
+ scheduledPricing: {
1289
+ effectiveFrom: '2026-09-14T04:00:00Z',
1290
+ inputPricePerMTok: 0.15,
1291
+ outputPricePerMTok: 0.6,
1292
+ cacheReadPricePerMTok: 0.003,
1293
+ cacheWritePricePerMTok: 0.15,
1294
+ source: 'https://api-docs.deepseek.com/updates/ (2026-09-10): deepseek-v4-pro → V4.1 Flash at Flash prices from 2026-09-14 12:00 Beijing',
1295
+ },
1248
1296
  // Not published by DeepSeek — best-effort estimate.
1249
1297
  knowledgeCutoff: '2025-07-01',
1298
+ // Deprecated the day it stops being itself: DeepSeek's own testing has
1299
+ // V4.1 Flash "comprehensively surpassing" Pro on performance/cost/speed,
1300
+ // and from 2026-09-14 12:00 Beijing every `deepseek-v4-pro` request routes
1301
+ // to V4.1 Flash at Flash prices (see scheduledPricing above) until V4.1 Pro
1302
+ // ships — at which point this becomes a new entry's succession problem, not
1303
+ // this one's. Until the 14th it still serves real V4-Pro-0813 weights at
1304
+ // the Pro card, so it stays selectable for existing selections.
1305
+ deprecatedAt: '2026-09-14',
1250
1306
  },
1251
1307
  {
1252
1308
  id: 'deepseek-v4-flash',
@@ -1263,17 +1319,22 @@ export const MODELS = [
1263
1319
  supportsVision: false,
1264
1320
  supportsPromptCaching: true,
1265
1321
  supportsTools: true,
1266
- // Free-tier default: cheapest model + a fast, non-thinking tool-calling
1267
- // executor — the model the IDE picks when none is chosen. Exactly one model
1268
- // in this catalog may carry freeTier (enforced by lookup.test.ts).
1269
- freeTier: true,
1270
- // Off-peak rates; peak is `peakPricing.multiplier` × these (see below).
1271
- inputPricePerMTok: 0.22,
1272
- outputPricePerMTok: 0.66,
1322
+ // freeTier moved to `deepseek-flash` (2026-09-10) together with
1323
+ // molecule-dev's default ids — the free tier follows the go-forward
1324
+ // evergreen id, not a retired name whose routing DeepSeek calls
1325
+ // "temporary". Exactly one model in this catalog may carry freeTier
1326
+ // (enforced by lookup.test.ts).
1327
+ // The V4.1 Flash card (2026-09-10). This legacy id is RETIRED and
1328
+ // "temporarily routed" to V4.1 Flash, billed at Flash prices — so these ARE
1329
+ // the rates the id answers with today (they replace the 0.22/0.66/0.007
1330
+ // card the same id served from 2026-08-16; same in-place succession as the
1331
+ // 2026-07-31 0731 re-post-train). Peak is `peakPricing.multiplier` × these.
1332
+ inputPricePerMTok: 0.15,
1333
+ outputPricePerMTok: 0.6,
1273
1334
  // DeepSeek automatic context cache: absolute cache-hit price ($/M).
1274
- cacheReadPricePerMTok: 0.007,
1335
+ cacheReadPricePerMTok: 0.003,
1275
1336
  // DeepSeek charges no cache-write premium — write bills at input.
1276
- cacheWritePricePerMTok: 0.22,
1337
+ cacheWritePricePerMTok: 0.15,
1277
1338
  // US (DeepInfra) DEFAULT as of 2026-08-16 — flipped from CN when DeepSeek's
1278
1339
  // rise landed (owner decision 2026-08-14). CN was cheaper on real traffic
1279
1340
  // only because of its cache reads; the rise takes those from $0.0028 to
@@ -1287,6 +1348,20 @@ export const MODELS = [
1287
1348
  // where CN's cheaper reads start winning again. This deliberately splits
1288
1349
  // the plan/execute pair across regions — Pro stays CN because its US
1289
1350
  // re-host is ~2.3x its own native rate even after the rise.
1351
+ // Re-derived 2026-09-10 against the repriced CN card: CN's cache hits are
1352
+ // now 5× cheaper than the re-host's ($0.003 vs $0.015), so CN wins on
1353
+ // blended input (0.0118 vs 0.0177 $/MTok off-peak) — but US wins on output
1354
+ // at 0.18 vs 0.6/1.2, and CN only takes a turn whose output is under ~1.4%
1355
+ // of its input. At real agentic output ratios (5–25%) US still wins every
1356
+ // hour, so the US default holds.
1357
+ // MIGRATED 2026-09-10: molecule-dev's default literals (EXECUTOR_MODEL /
1358
+ // EXECUTOR_MODEL_CUSTOM / DEFAULT_CHAT_MODEL / COMPACTION_MODEL /
1359
+ // COMMIT_MESSAGE_MODEL / FREE_TIER_MODELS.execute) now point at
1360
+ // `deepseek-flash` (cn-native). This entry stays selectable — not
1361
+ // `supersededBy` — because its US leg (DeepInfra 0731 at $0.06/$0.18 flat,
1362
+ // older-but-solid weights DeepInfra keeps serving) is still the cheapest
1363
+ // flash serving there is and saved selections should keep working; it is
1364
+ // merely hidden from offering via `deprecatedAt`.
1290
1365
  regions: ['us', 'cn'],
1291
1366
  // US = DeepInfra, verified 2026-09-06 against the id the bond actually
1292
1367
  // sends: `deepseek-ai/DeepSeek-V4-Flash-0731`, the official release that
@@ -1313,6 +1388,71 @@ export const MODELS = [
1313
1388
  },
1314
1389
  // Not published by DeepSeek — best-effort estimate.
1315
1390
  knowledgeCutoff: '2025-07-01',
1391
+ // Retired upstream 2026-09-10 (V4 flash ids; legacy name "temporarily
1392
+ // routed" to V4.1 Flash). See the regions note above for why this is
1393
+ // deprecatedAt — hidden from offering, still selectable — rather than
1394
+ // supersededBy.
1395
+ deprecatedAt: '2026-09-10',
1396
+ },
1397
+ {
1398
+ // Released 2026-09-10 (updates page) — the go-forward EVERGREEN flash id.
1399
+ // Same rates the repriced legacy entry above carries natively, but under
1400
+ // the id DeepSeek actually wants called, with the multimodal surface the
1401
+ // announcement claims ("native multimodal visual understanding") and none
1402
+ // of the retirement ambiguity of a "temporarily routed" legacy name.
1403
+ id: 'deepseek-flash',
1404
+ provider: 'deepseek',
1405
+ label: 'DeepSeek V4.1 Flash',
1406
+ description: 'Newest ultra-cheap & fast flash — vision-capable agentic coding',
1407
+ contextWindow: 1_000_000,
1408
+ maxOutputTokens: 384_000,
1409
+ // Non-thinking (see section note) — same bond treatment as the family:
1410
+ // the executor tool loop does not replay reasoning_content.
1411
+ supportsThinking: false,
1412
+ thinkingBudgetTokens: 0,
1413
+ thinkingConfigurable: false,
1414
+ supportsVision: true,
1415
+ supportsPromptCaching: true,
1416
+ supportsTools: true,
1417
+ // Free-tier default (moved from deepseek-v4-flash 2026-09-10): the
1418
+ // cheapest capable executor on the go-forward id — fast, non-thinking,
1419
+ // tool-calling, and now vision-capable. Serving cost is ~2× the retired
1420
+ // US leg on output-heavy turns (CN 0.6/1.2 output vs DeepInfra's flat
1421
+ // 0.18) — tiers.ts's free-turn budget math derives from this entry.
1422
+ // Exactly one model in this catalog may carry freeTier (lookup.test.ts).
1423
+ freeTier: true,
1424
+ // Off-peak V4.1 Flash card; peak is `peakPricing.multiplier` × these.
1425
+ inputPricePerMTok: 0.15,
1426
+ outputPricePerMTok: 0.6,
1427
+ // DeepSeek automatic context cache: absolute cache-hit price ($/M).
1428
+ cacheReadPricePerMTok: 0.003,
1429
+ // DeepSeek charges no cache-write premium — write bills at input.
1430
+ cacheWritePricePerMTok: 0.15,
1431
+ // Native first (the default region); the DeepInfra US re-host
1432
+ // (`deepseek-ai/DeepSeek-V4.1-Flash`, molecule-dev region-model-maps.ts)
1433
+ // is offered as a SECOND region since 2026-09-14, when the native endpoint
1434
+ // queued every completion for hours (SSE keep-alives, no token, three
1435
+ // Synthase turns lost) while DeepInfra answered at once. Its card, read
1436
+ // live from api.deepinfra.com/models on 2026-09-14: $0.20/$0.60, cache
1437
+ // read 3% of input = $0.006 — roughly 2× this native off-peak card on
1438
+ // cache reads, so native stays the default and `us` is the fallback a
1439
+ // project picks when native is unavailable.
1440
+ regions: ['cn', 'us'],
1441
+ regionPricing: {
1442
+ us: { inputPricePerMTok: 0.2, outputPricePerMTok: 0.6, cacheReadPricePerMTok: 0.006 },
1443
+ },
1444
+ // Same peak windows as the family — the V4.1 card footnote is verbatim the
1445
+ // 2026-08-31 sentence: 01:00-04:00 and 06:00-10:00 UTC, Mon-Fri, ×2.
1446
+ peakPricing: {
1447
+ windows: [
1448
+ { startMinuteUtc: 60, endMinuteUtc: 240, daysOfWeekUtc: WEEKDAYS_UTC },
1449
+ { startMinuteUtc: 360, endMinuteUtc: 600, daysOfWeekUtc: WEEKDAYS_UTC },
1450
+ ],
1451
+ multiplier: 2,
1452
+ },
1453
+ // Not published by DeepSeek — best-effort estimate (family estimate; the
1454
+ // V4.1 announcement lists benchmarks but no training cutoff).
1455
+ knowledgeCutoff: '2025-07-01',
1316
1456
  },
1317
1457
  // ---------------------------------------------------------------------------
1318
1458
  // Moonshot (Kimi)
@@ -1473,8 +1613,15 @@ export const MODELS = [
1473
1613
  knowledgeCutoff: '2024-04-01',
1474
1614
  // Two generations behind. `supersededBy` names the current selectable Kimi
1475
1615
  // (k3) rather than the also-superseded k2.6, so a saved selection resolves
1476
- // forward in one hop. Still served upstream; stays priceable.
1616
+ // forward in one hop. Moonshot RETIRED kimi-k2.5 on 2026-08-31 —
1617
+ // platform.kimi.ai/docs/models says calls 404, and a live 2026-09-10
1618
+ // dispatch probe confirmed the native host refusing it. The DeepInfra
1619
+ // re-host still answered inference on 2026-09-10 but is DELISTED from its
1620
+ // model catalog (detail page 404, /models/list unpruned) — borrowed time.
1621
+ // Disabled, never deleted: historical usage stays priceable and saved
1622
+ // selections still resolve forward.
1477
1623
  deprecatedAt: '2026-04-01',
1624
+ disabled: true,
1478
1625
  supersededBy: 'kimi-k3',
1479
1626
  },
1480
1627
  // ---------------------------------------------------------------------------
@@ -1839,9 +1986,11 @@ export const MODELS = [
1839
1986
  supportsPromptCaching: true,
1840
1987
  supportsTools: true,
1841
1988
  webSearchToolType: 'web_search',
1842
- // LIST card. A 50%-off launch promo ($0.075/$0.25, cached $0.015) runs to
1843
- // 2026-09-09 24:00 UTC+8; billed at list so metering never under-charges —
1844
- // the promo is a KNOWN_DIVERGENCES entry in check-model-freshness.
1989
+ // List card, re-verified 2026-09-10 on the Z.ai pricing page after the
1990
+ // 50%-off launch promo ($0.075/$0.25, cached $0.015) ended on schedule
1991
+ // 2026-09-09 24:00 UTC+8 — this IS what the provider bills now; models.dev
1992
+ // still carries the expired promo rate (KNOWN_DIVERGENCES entry in
1993
+ // check-model-freshness, re-verify date 2026-12-09).
1845
1994
  inputPricePerMTok: 0.15,
1846
1995
  outputPricePerMTok: 0.5,
1847
1996
  // GLM context cache: read 0.2× input, no write premium.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@molecule/api-resource-ai-models",
3
- "version": "1.5.1",
3
+ "version": "1.6.1",
4
4
  "description": "AI model catalog — server-side source of truth plus an authentication-gated discovery endpoint",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",
@@ -32,7 +32,7 @@
32
32
  "@molecule/api-resource": "1.0.1",
33
33
  "@types/node": "26.1.2",
34
34
  "typescript": "6.0.3",
35
- "vitest": "4.1.10"
35
+ "vitest": "4.1.11"
36
36
  },
37
37
  "peerDependencies": {
38
38
  "@molecule/api-bond": "^1.0.1",