@molecule/api-resource-ai-models 1.2.2 → 1.2.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -3,7 +3,7 @@ AUTO-GENERATED — DO NOT EDIT THIS FILE.
3
3
  Generated by `mlcl sync-docs` from the package's src/index.ts JSDoc + mlcl/registry.json.
4
4
  Edits here are overwritten on the next commit (molecule's pre-commit hook regenerates).
5
5
  To change this document, edit the module-level JSDoc in src/index.ts.
6
- Generated: 2026-08-15T19:37:59.137Z
6
+ Generated: 2026-08-18T03:40:46.175Z
7
7
  -->
8
8
 
9
9
  # @molecule/api-resource-ai-models
@@ -789,12 +789,12 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
789
789
  grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
790
790
  grok-code-fast-1 no longer listed — retires 2026-08-15)
791
791
  - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
792
- 2026-08-14; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
793
- never in this catalog. V4-Pro GA on 2026-08-13 came with a price RISE
794
- effective 2026-08-16T16:00Z plus the long-announced peak-hour 2×: both
795
- entries carry it as `scheduledPricing`, so today's rates bill until that
796
- instant and the new ones after. Re-verify weekday-vs-daily peak windows and
797
- the CN/US region default once it lands — see the entries.)
792
+ 2026-08-18; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
793
+ never in this catalog. The V4-Pro-GA price RISE effective 2026-08-16T16:00Z
794
+ has LANDED and is folded into the base fields, along with the peak-hour 2×
795
+ the same card introduced. The rate card gives the peak windows as
796
+ "01:00 - 04:00 and 06:00 - 10:00 UTC (all other hours are off-peak)" with
797
+ no day qualifier — DAILY, as `peakPricing` models them.)
798
798
  - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
799
799
  the US re-host (kimi-k3 flagship 2026-07-16
800
800
  — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
package/dist/models.d.ts CHANGED
@@ -73,12 +73,12 @@ import type { ModelDefinition } from './types.js';
73
73
  * grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
74
74
  * grok-code-fast-1 no longer listed — retires 2026-08-15)
75
75
  * - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
76
- * 2026-08-14; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
77
- * never in this catalog. V4-Pro GA on 2026-08-13 came with a price RISE
78
- * effective 2026-08-16T16:00Z plus the long-announced peak-hour 2×: both
79
- * entries carry it as `scheduledPricing`, so today's rates bill until that
80
- * instant and the new ones after. Re-verify weekday-vs-daily peak windows and
81
- * the CN/US region default once it lands — see the entries.)
76
+ * 2026-08-18; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
77
+ * never in this catalog. The V4-Pro-GA price RISE effective 2026-08-16T16:00Z
78
+ * has LANDED and is folded into the base fields, along with the peak-hour 2×
79
+ * the same card introduced. The rate card gives the peak windows as
80
+ * "01:00 - 04:00 and 06:00 - 10:00 UTC (all other hours are off-peak)" with
81
+ * no day qualifier — DAILY, as `peakPricing` models them.)
82
82
  * - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
83
83
  * the US re-host (kimi-k3 flagship 2026-07-16
84
84
  * — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
@@ -1 +1 @@
1
- {"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA6GG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAk6CnC,CAAA"}
1
+ {"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA6GG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAo5CnC,CAAA"}
package/dist/models.js CHANGED
@@ -72,12 +72,12 @@
72
72
  * grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
73
73
  * grok-code-fast-1 no longer listed — retires 2026-08-15)
74
74
  * - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
75
- * 2026-08-14; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
76
- * never in this catalog. V4-Pro GA on 2026-08-13 came with a price RISE
77
- * effective 2026-08-16T16:00Z plus the long-announced peak-hour 2×: both
78
- * entries carry it as `scheduledPricing`, so today's rates bill until that
79
- * instant and the new ones after. Re-verify weekday-vs-daily peak windows and
80
- * the CN/US region default once it lands — see the entries.)
75
+ * 2026-08-18; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
76
+ * never in this catalog. The V4-Pro-GA price RISE effective 2026-08-16T16:00Z
77
+ * has LANDED and is folded into the base fields, along with the peak-hour 2×
78
+ * the same card introduced. The rate card gives the peak windows as
79
+ * "01:00 - 04:00 and 06:00 - 10:00 UTC (all other hours are off-peak)" with
80
+ * no day qualifier — DAILY, as `peakPricing` models them.)
81
81
  * - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
82
82
  * the US re-host (kimi-k3 flagship 2026-07-16
83
83
  * — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
@@ -480,6 +480,11 @@ export const MODELS = [
480
480
  // Cached input 0.1× input; write premium reported 1.25× (see section note).
481
481
  cacheReadPricePerMTok: 0.02,
482
482
  cacheWritePricePerMTok: 0.25,
483
+ // The free-tier PLAN default (2026-08-18) — us-only OpenAI, so this carve-out
484
+ // is what lets `freeTierAllows` permit it for plan mode without making it
485
+ // outright `freeTier` (which would free it in every mode). Mirrors how
486
+ // minimax-m3 was scoped as the prior free planner.
487
+ freeTierRegions: ['us'],
483
488
  // Not published — best-effort estimate.
484
489
  knowledgeCutoff: '2026-03-01',
485
490
  },
@@ -896,35 +901,31 @@ export const MODELS = [
896
901
  // DeepSeek
897
902
  // Verified: https://api-docs.deepseek.com/quick_start/pricing
898
903
  // https://api-docs.deepseek.com/guides/thinking_mode
899
- // https://api-docs.deepseek.com/updates/ (2026-08-14)
904
+ // https://api-docs.deepseek.com/updates/ (2026-08-18)
900
905
  // 2026-08-13: V4-Pro GA — and with it the price rise that the "coming soon"
901
- // note below had been waiting on. It is STAGED, not applied: both models
902
- // carry `scheduledPricing` effective 2026-08-16T16:00Z, so the catalog bills
903
- // today's verified rates until that instant and the new ones after it, with
904
- // nobody landing an edit at 16:00 UTC on a Sunday. The new card is
905
- // off-peak/peak (peak = exactly 2× off-peak), so it maps onto base rates +
906
- // `peakPricing` multiplier 2 — which is why the peak windows removed below
907
- // come back here rather than as flat rates.
906
+ // note below had been waiting on. It was staged as `scheduledPricing`
907
+ // effective 2026-08-16T16:00Z; that instant has PASSED and the rates are now
908
+ // folded into the base fields, re-verified 2026-08-18 against the live rate
909
+ // card. The card is off-peak/peak with peak exactly 2× off-peak, so it maps
910
+ // onto base rates + `peakPricing` multiplier 2 — which is why the peak
911
+ // windows removed in July come back here rather than as flat rates.
908
912
  // pro off-peak 0.66 / 1.98, cache hit 0.022 (peak 1.32 / 3.96 / 0.044)
909
913
  // flash off-peak 0.22 / 0.66, cache hit 0.007 (peak 0.44 / 1.32 / 0.014)
910
- // Cache HITS are the real move — pro 0.003625 → 0.022 (6.1×) off-peak, 0.044
911
- // (12.1×) at peak — and agentic input is ~94% cache hits, so effective input
912
- // cost rises far more than the list prices suggest. Both are free-tier models
913
- // (flash is `freeTier`, pro is the free-tier planner) on the CN default.
914
- // TWO things to re-verify once it lands (2026-08-17):
915
- // 1. WEEKDAYS OR DAILY. The rate card says only "Peak hours are 01:00 -
916
- // 04:00 and 06:00 - 10:00 UTC (all other hours are off-peak)" with no
917
- // day qualifier, so the windows below are DAILY per the provider's own
918
- // doc; press coverage described them as weekday-only. `peakPricing` has
919
- // no day-of-week concept, so if it is weekday-only this over-bills every
920
- // weekend peak window and needs the field extended, not the numbers
921
- // nudged.
922
- // 2. THE CN-VS-US DEFAULT. `regions: ['cn', 'us']` defaults to CN on an
923
- // owner decision (2026-08-01) taken when CN ran ~5.7× cheaper on real
924
- // traffic. Post-change DeepInfra's US flash rates (0.08/0.18/0.016) are
925
- // BELOW CN's new off-peak on both input and output — CN wins only on
926
- // cache reads. Re-derive against measured cache-hit ratios before
927
- // leaving the default where it is.
914
+ // Cache HITS were the real move — pro 0.003625 → 0.022 (6.1×) off-peak,
915
+ // 0.044 (12.1×) at peak — and agentic input is ~94% cache hits, so effective
916
+ // input cost rose far more than the list prices suggest. Both are free-tier
917
+ // models (flash is `freeTier`, pro is the free-tier planner).
918
+ // The two post-landing re-verifications, both settled 2026-08-18:
919
+ // 1. WEEKDAYS OR DAILY — DAILY. The rate card still says only "Peak hours
920
+ // are 01:00 - 04:00 and 06:00 - 10:00 UTC (all other hours are
921
+ // off-peak)" with no day qualifier (press coverage had described them as
922
+ // weekday-only). `peakPricing` has no day-of-week concept, so if the
923
+ // provider ever qualifies these by day this over-bills every weekend
924
+ // peak window and needs the FIELD extended, not the numbers nudged.
925
+ // 2. THE CN-VS-US DEFAULT — settled per model, and they differ. Post-rise
926
+ // DeepInfra's US flash rates (0.08/0.18/0.016) are BELOW CN's off-peak
927
+ // on both input and output, so Flash defaults US; Pro's US re-host still
928
+ // bills ~2.3× its native card, so Pro stays CN. See each entry.
928
929
  // 2026-07-31: DeepSeek-V4-Flash OFFICIAL API launched in public beta — the
929
930
  // SAME `deepseek-v4-flash` id now serves the re-post-trained 0731 build
930
931
  // (same architecture/size; much stronger agent benchmarks — beats
@@ -954,11 +955,13 @@ export const MODELS = [
954
955
  supportsVision: false,
955
956
  supportsPromptCaching: true,
956
957
  supportsTools: true,
957
- inputPricePerMTok: 0.435,
958
- outputPricePerMTok: 0.87,
958
+ // Off-peak rates; peak is `peakPricing.multiplier` × these (see below).
959
+ inputPricePerMTok: 0.66,
960
+ outputPricePerMTok: 1.98,
959
961
  // DeepSeek automatic context cache: absolute cache-hit price ($/M).
960
- cacheReadPricePerMTok: 0.003625,
961
- cacheWritePricePerMTok: 0.435,
962
+ cacheReadPricePerMTok: 0.022,
963
+ // DeepSeek charges no cache-write premium — write bills at input.
964
+ cacheWritePricePerMTok: 0.66,
962
965
  // Native-China DEFAULT (owner decision 2026-08-01, re-derived 2026-08-14):
963
966
  // the US re-host (DeepInfra) bills ~3× list and ~28× cache reads, and
964
967
  // agentic input is ~94% cache hits, so US processing ran ~5.7× native on
@@ -982,28 +985,18 @@ export const MODELS = [
982
985
  regionPricing: {
983
986
  us: { inputPricePerMTok: 1.3, outputPricePerMTok: 2.6, cacheReadPricePerMTok: 0.1 },
984
987
  },
985
- // The peak-hour 2× surcharge is now ON the rate card with a dated switch
986
- // (2026-08-13 announcement, effective 2026-08-16T16:00Z) — so it is staged
987
- // below rather than live. The previously pre-wired windows had been REMOVED
988
- // for over-billing every peak-window turn 2× for weeks against a rate card
989
- // that showed a single flat rate; staging is what keeps this from repeating
990
- // in the other direction. Peak = 01:00-04:00 and 06:00-10:00 UTC (Beijing
991
- // business hours), which is 2× the off-peak rates exactly.
992
- scheduledPricing: {
993
- effectiveFrom: '2026-08-16T16:00:00Z',
994
- inputPricePerMTok: 0.66,
995
- outputPricePerMTok: 1.98,
996
- cacheReadPricePerMTok: 0.022,
997
- // DeepSeek charges no cache-write premium — write bills at input.
998
- cacheWritePricePerMTok: 0.66,
999
- peakPricing: {
1000
- windows: [
1001
- { startMinuteUtc: 60, endMinuteUtc: 240 },
1002
- { startMinuteUtc: 360, endMinuteUtc: 600 },
1003
- ],
1004
- multiplier: 2,
1005
- },
1006
- source: 'https://api-docs.deepseek.com/quick_start/pricing/',
988
+ // The peak-hour 2× surcharge is ON the rate card and LIVE since
989
+ // 2026-08-16T16:00Z. Peak = 01:00-04:00 and 06:00-10:00 UTC (Beijing
990
+ // business hours), daily, at exactly 2× the off-peak rates above. Windows
991
+ // like these were once pre-wired ahead of the card and over-billed every
992
+ // peak-window turn for weeks — hence the rule that they only exist here
993
+ // once the provider's own page shows them, which it now does.
994
+ peakPricing: {
995
+ windows: [
996
+ { startMinuteUtc: 60, endMinuteUtc: 240 },
997
+ { startMinuteUtc: 360, endMinuteUtc: 600 },
998
+ ],
999
+ multiplier: 2,
1007
1000
  },
1008
1001
  // Not published by DeepSeek — best-effort estimate.
1009
1002
  knowledgeCutoff: '2025-07-01',
@@ -1027,11 +1020,13 @@ export const MODELS = [
1027
1020
  // executor — the model the IDE picks when none is chosen. Exactly one model
1028
1021
  // in this catalog may carry freeTier (enforced by lookup.test.ts).
1029
1022
  freeTier: true,
1030
- inputPricePerMTok: 0.14,
1031
- outputPricePerMTok: 0.28,
1023
+ // Off-peak rates; peak is `peakPricing.multiplier` × these (see below).
1024
+ inputPricePerMTok: 0.22,
1025
+ outputPricePerMTok: 0.66,
1032
1026
  // DeepSeek automatic context cache: absolute cache-hit price ($/M).
1033
- cacheReadPricePerMTok: 0.0028,
1034
- cacheWritePricePerMTok: 0.14,
1027
+ cacheReadPricePerMTok: 0.007,
1028
+ // DeepSeek charges no cache-write premium — write bills at input.
1029
+ cacheWritePricePerMTok: 0.22,
1035
1030
  // US (DeepInfra) DEFAULT as of 2026-08-16 — flipped from CN when DeepSeek's
1036
1031
  // rise landed (owner decision 2026-08-14). CN was cheaper on real traffic
1037
1032
  // only because of its cache reads; the rise takes those from $0.0028 to
@@ -1053,22 +1048,15 @@ export const MODELS = [
1053
1048
  regionPricing: {
1054
1049
  us: { inputPricePerMTok: 0.08, outputPricePerMTok: 0.18, cacheReadPricePerMTok: 0.016 },
1055
1050
  },
1056
- // Peak-hour surcharge staged, not live (see deepseek-v4-pro).
1057
- scheduledPricing: {
1058
- effectiveFrom: '2026-08-16T16:00:00Z',
1059
- inputPricePerMTok: 0.22,
1060
- outputPricePerMTok: 0.66,
1061
- cacheReadPricePerMTok: 0.007,
1062
- // DeepSeek charges no cache-write premium — write bills at input.
1063
- cacheWritePricePerMTok: 0.22,
1064
- peakPricing: {
1065
- windows: [
1066
- { startMinuteUtc: 60, endMinuteUtc: 240 },
1067
- { startMinuteUtc: 360, endMinuteUtc: 600 },
1068
- ],
1069
- multiplier: 2,
1070
- },
1071
- source: 'https://api-docs.deepseek.com/quick_start/pricing/',
1051
+ // Peak-hour surcharge live since 2026-08-16T16:00Z (see deepseek-v4-pro).
1052
+ // It applies to the NATIVE CN card only — this model defaults to the US
1053
+ // re-host, which is flat, so most turns never take it.
1054
+ peakPricing: {
1055
+ windows: [
1056
+ { startMinuteUtc: 60, endMinuteUtc: 240 },
1057
+ { startMinuteUtc: 360, endMinuteUtc: 600 },
1058
+ ],
1059
+ multiplier: 2,
1072
1060
  },
1073
1061
  // Not published by DeepSeek — best-effort estimate.
1074
1062
  knowledgeCutoff: '2025-07-01',
@@ -1271,15 +1259,13 @@ export const MODELS = [
1271
1259
  // US default. DeepInfra list matches native; only the cache write differs
1272
1260
  // (no premium → region input rate). Verified 2026-08-01.
1273
1261
  regions: ['us', 'cn'],
1274
- // The free tier PLANS with this model (molecule-dev FREE_TIER_MODELS.plan,
1275
- // 2026-08-14), so its default US region must be free-tier selectable. It
1276
- // took over from deepseek-v4-pro@cn: measured on the real starting-point
1277
- // selection it scored 8/8 against Pro's 7/8 — including the case Pro failed
1278
- // — at 1.28c/plan-turn flat versus Pro's 2.62c off-peak and 5.24c inside
1279
- // DeepSeek's Beijing-hours windows, and it adds vision, which Pro (text
1280
- // only) could not offer discovery. CN is NOT listed: it is dearer than US
1281
- // here, so free planning stays on the cheaper host.
1282
- freeTierRegions: ['us'],
1262
+ // No freeTierRegions: the free tier stopped planning with this model on
1263
+ // 2026-08-18 (gpt-5.6-luna took over as FREE_TIER_MODELS.plan). It had been
1264
+ // the free planner since 2026-08-14, taking over from deepseek-v4-pro@cn.
1265
+ // The carve-out only ever existed to keep the free tier's OWN plan default
1266
+ // usable, and `freeTierAllows` checks `FREE_TIER_MODELS[mode] === modelId`
1267
+ // before it looks at regions, so leaving it here would widen nothing — it
1268
+ // would just claim a free-tier relationship that no longer exists.
1283
1269
  regionPricing: {
1284
1270
  // Verified 2026-08-14 against api.deepinfra.com/models/MiniMaxAI/MiniMax-M3
1285
1271
  // (cache read = 0.2 × input). Was 0.3/1.2/0.06 — DeepInfra had repriced
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@molecule/api-resource-ai-models",
3
- "version": "1.2.2",
3
+ "version": "1.2.4",
4
4
  "description": "AI model catalog — server-side source of truth plus an authentication-gated discovery endpoint",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",