@molecule/api-resource-ai-models 1.2.1 → 1.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -3,7 +3,7 @@ AUTO-GENERATED — DO NOT EDIT THIS FILE.
3
3
  Generated by `mlcl sync-docs` from the package's src/index.ts JSDoc + mlcl/registry.json.
4
4
  Edits here are overwritten on the next commit (molecule's pre-commit hook regenerates).
5
5
  To change this document, edit the module-level JSDoc in src/index.ts.
6
- Generated: 2026-08-13T22:10:11.926Z
6
+ Generated: 2026-08-18T03:40:46.175Z
7
7
  -->
8
8
 
9
9
  # @molecule/api-resource-ai-models
@@ -789,12 +789,12 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
789
789
  grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
790
790
  grok-code-fast-1 no longer listed — retires 2026-08-15)
791
791
  - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
792
- 2026-08-14; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
793
- never in this catalog. V4-Pro GA on 2026-08-13 came with a price RISE
794
- effective 2026-08-16T16:00Z plus the long-announced peak-hour 2×: both
795
- entries carry it as `scheduledPricing`, so today's rates bill until that
796
- instant and the new ones after. Re-verify weekday-vs-daily peak windows and
797
- the CN/US region default once it lands — see the entries.)
792
+ 2026-08-18; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
793
+ never in this catalog. The V4-Pro-GA price RISE effective 2026-08-16T16:00Z
794
+ has LANDED and is folded into the base fields, along with the peak-hour 2×
795
+ the same card introduced. The rate card gives the peak windows as
796
+ "01:00 - 04:00 and 06:00 - 10:00 UTC (all other hours are off-peak)" with
797
+ no day qualifier — DAILY, as `peakPricing` models them.)
798
798
  - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
799
799
  the US re-host (kimi-k3 flagship 2026-07-16
800
800
  — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
@@ -817,6 +817,16 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
817
817
  125%) — the EN edition of that doc simply lags (zero qwen3.8 mentions),
818
818
  which an earlier pass misread as "excepted/console-only". qwen3.7-max
819
819
  still runs its 50%-off promo — billed here at list, $2.50/$7.50)
820
+ (re-verified 2026-08-15 against api.deepinfra.com/models/: qwen3.8-max's US
821
+ re-host rates are unchanged — $1.65/$4.951, cache read 0.1248× input =
822
+ $0.206. DeepInfra also began serving `Qwen/Qwen3.8-2.4T-A95B` on
823
+ 2026-08-12; by DeepInfra's own description it is the OPEN-WEIGHT variant of
824
+ Qwen3.8 Max (2.4T MoE, 95B active), 262,144 ctx / 131,072 out, $2/$6 with
825
+ cache read 0.1× input. NOT added: it is the same tier as qwen3.8-max, which
826
+ the catalog already offers, and it costs MORE on that very host ($2/$6 vs
827
+ $1.65/$4.951), so nothing would ever select it — and carrying both would
828
+ put two selectable Alibaba flagships in one family. Revisit only if Alibaba
829
+ publishes it as a distinct first-party DashScope model id.)
820
830
  - Zhipu: https://docs.z.ai/guides/overview/pricing (unchanged; glm-5.2 is
821
831
  the newest — "GLM-5.3/5.5" rumors have no released ids as of 2026-07-28)
822
832
 
package/dist/models.d.ts CHANGED
@@ -73,12 +73,12 @@ import type { ModelDefinition } from './types.js';
73
73
  * grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
74
74
  * grok-code-fast-1 no longer listed — retires 2026-08-15)
75
75
  * - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
76
- * 2026-08-14; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
77
- * never in this catalog. V4-Pro GA on 2026-08-13 came with a price RISE
78
- * effective 2026-08-16T16:00Z plus the long-announced peak-hour 2×: both
79
- * entries carry it as `scheduledPricing`, so today's rates bill until that
80
- * instant and the new ones after. Re-verify weekday-vs-daily peak windows and
81
- * the CN/US region default once it lands — see the entries.)
76
+ * 2026-08-18; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
77
+ * never in this catalog. The V4-Pro-GA price RISE effective 2026-08-16T16:00Z
78
+ * has LANDED and is folded into the base fields, along with the peak-hour 2×
79
+ * the same card introduced. The rate card gives the peak windows as
80
+ * "01:00 - 04:00 and 06:00 - 10:00 UTC (all other hours are off-peak)" with
81
+ * no day qualifier — DAILY, as `peakPricing` models them.)
82
82
  * - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
83
83
  * the US re-host (kimi-k3 flagship 2026-07-16
84
84
  * — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
@@ -101,6 +101,16 @@ import type { ModelDefinition } from './types.js';
101
101
  * 125%) — the EN edition of that doc simply lags (zero qwen3.8 mentions),
102
102
  * which an earlier pass misread as "excepted/console-only". qwen3.7-max
103
103
  * still runs its 50%-off promo — billed here at list, $2.50/$7.50)
104
+ * (re-verified 2026-08-15 against api.deepinfra.com/models/: qwen3.8-max's US
105
+ * re-host rates are unchanged — $1.65/$4.951, cache read 0.1248× input =
106
+ * $0.206. DeepInfra also began serving `Qwen/Qwen3.8-2.4T-A95B` on
107
+ * 2026-08-12; by DeepInfra's own description it is the OPEN-WEIGHT variant of
108
+ * Qwen3.8 Max (2.4T MoE, 95B active), 262,144 ctx / 131,072 out, $2/$6 with
109
+ * cache read 0.1× input. NOT added: it is the same tier as qwen3.8-max, which
110
+ * the catalog already offers, and it costs MORE on that very host ($2/$6 vs
111
+ * $1.65/$4.951), so nothing would ever select it — and carrying both would
112
+ * put two selectable Alibaba flagships in one family. Revisit only if Alibaba
113
+ * publishes it as a distinct first-party DashScope model id.)
104
114
  * - Zhipu: https://docs.z.ai/guides/overview/pricing (unchanged; glm-5.2 is
105
115
  * the newest — "GLM-5.3/5.5" rumors have no released ids as of 2026-07-28)
106
116
  *
@@ -1 +1 @@
1
- {"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAmGG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAk6CnC,CAAA"}
1
+ {"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA6GG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EAi5CnC,CAAA"}
package/dist/models.js CHANGED
@@ -72,12 +72,12 @@
72
72
  * grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
73
73
  * grok-code-fast-1 no longer listed — retires 2026-08-15)
74
74
  * - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
75
- * 2026-08-14; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
76
- * never in this catalog. V4-Pro GA on 2026-08-13 came with a price RISE
77
- * effective 2026-08-16T16:00Z plus the long-announced peak-hour 2×: both
78
- * entries carry it as `scheduledPricing`, so today's rates bill until that
79
- * instant and the new ones after. Re-verify weekday-vs-daily peak windows and
80
- * the CN/US region default once it lands — see the entries.)
75
+ * 2026-08-18; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
76
+ * never in this catalog. The V4-Pro-GA price RISE effective 2026-08-16T16:00Z
77
+ * has LANDED and is folded into the base fields, along with the peak-hour 2×
78
+ * the same card introduced. The rate card gives the peak windows as
79
+ * "01:00 - 04:00 and 06:00 - 10:00 UTC (all other hours are off-peak)" with
80
+ * no day qualifier — DAILY, as `peakPricing` models them.)
81
81
  * - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
82
82
  * the US re-host (kimi-k3 flagship 2026-07-16
83
83
  * — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
@@ -100,6 +100,16 @@
100
100
  * 125%) — the EN edition of that doc simply lags (zero qwen3.8 mentions),
101
101
  * which an earlier pass misread as "excepted/console-only". qwen3.7-max
102
102
  * still runs its 50%-off promo — billed here at list, $2.50/$7.50)
103
+ * (re-verified 2026-08-15 against api.deepinfra.com/models/: qwen3.8-max's US
104
+ * re-host rates are unchanged — $1.65/$4.951, cache read 0.1248× input =
105
+ * $0.206. DeepInfra also began serving `Qwen/Qwen3.8-2.4T-A95B` on
106
+ * 2026-08-12; by DeepInfra's own description it is the OPEN-WEIGHT variant of
107
+ * Qwen3.8 Max (2.4T MoE, 95B active), 262,144 ctx / 131,072 out, $2/$6 with
108
+ * cache read 0.1× input. NOT added: it is the same tier as qwen3.8-max, which
109
+ * the catalog already offers, and it costs MORE on that very host ($2/$6 vs
110
+ * $1.65/$4.951), so nothing would ever select it — and carrying both would
111
+ * put two selectable Alibaba flagships in one family. Revisit only if Alibaba
112
+ * publishes it as a distinct first-party DashScope model id.)
103
113
  * - Zhipu: https://docs.z.ai/guides/overview/pricing (unchanged; glm-5.2 is
104
114
  * the newest — "GLM-5.3/5.5" rumors have no released ids as of 2026-07-28)
105
115
  *
@@ -886,35 +896,31 @@ export const MODELS = [
886
896
  // DeepSeek
887
897
  // Verified: https://api-docs.deepseek.com/quick_start/pricing
888
898
  // https://api-docs.deepseek.com/guides/thinking_mode
889
- // https://api-docs.deepseek.com/updates/ (2026-08-14)
899
+ // https://api-docs.deepseek.com/updates/ (2026-08-18)
890
900
  // 2026-08-13: V4-Pro GA — and with it the price rise that the "coming soon"
891
- // note below had been waiting on. It is STAGED, not applied: both models
892
- // carry `scheduledPricing` effective 2026-08-16T16:00Z, so the catalog bills
893
- // today's verified rates until that instant and the new ones after it, with
894
- // nobody landing an edit at 16:00 UTC on a Sunday. The new card is
895
- // off-peak/peak (peak = exactly 2× off-peak), so it maps onto base rates +
896
- // `peakPricing` multiplier 2 — which is why the peak windows removed below
897
- // come back here rather than as flat rates.
901
+ // note below had been waiting on. It was staged as `scheduledPricing`
902
+ // effective 2026-08-16T16:00Z; that instant has PASSED and the rates are now
903
+ // folded into the base fields, re-verified 2026-08-18 against the live rate
904
+ // card. The card is off-peak/peak with peak exactly 2× off-peak, so it maps
905
+ // onto base rates + `peakPricing` multiplier 2 — which is why the peak
906
+ // windows removed in July come back here rather than as flat rates.
898
907
  // pro off-peak 0.66 / 1.98, cache hit 0.022 (peak 1.32 / 3.96 / 0.044)
899
908
  // flash off-peak 0.22 / 0.66, cache hit 0.007 (peak 0.44 / 1.32 / 0.014)
900
- // Cache HITS are the real move — pro 0.003625 → 0.022 (6.1×) off-peak, 0.044
901
- // (12.1×) at peak — and agentic input is ~94% cache hits, so effective input
902
- // cost rises far more than the list prices suggest. Both are free-tier models
903
- // (flash is `freeTier`, pro is the free-tier planner) on the CN default.
904
- // TWO things to re-verify once it lands (2026-08-17):
905
- // 1. WEEKDAYS OR DAILY. The rate card says only "Peak hours are 01:00 -
906
- // 04:00 and 06:00 - 10:00 UTC (all other hours are off-peak)" with no
907
- // day qualifier, so the windows below are DAILY per the provider's own
908
- // doc; press coverage described them as weekday-only. `peakPricing` has
909
- // no day-of-week concept, so if it is weekday-only this over-bills every
910
- // weekend peak window and needs the field extended, not the numbers
911
- // nudged.
912
- // 2. THE CN-VS-US DEFAULT. `regions: ['cn', 'us']` defaults to CN on an
913
- // owner decision (2026-08-01) taken when CN ran ~5.7× cheaper on real
914
- // traffic. Post-change DeepInfra's US flash rates (0.08/0.18/0.016) are
915
- // BELOW CN's new off-peak on both input and output — CN wins only on
916
- // cache reads. Re-derive against measured cache-hit ratios before
917
- // leaving the default where it is.
909
+ // Cache HITS were the real move — pro 0.003625 → 0.022 (6.1×) off-peak,
910
+ // 0.044 (12.1×) at peak — and agentic input is ~94% cache hits, so effective
911
+ // input cost rose far more than the list prices suggest. Both are free-tier
912
+ // models (flash is `freeTier`, pro is the free-tier planner).
913
+ // The two post-landing re-verifications, both settled 2026-08-18:
914
+ // 1. WEEKDAYS OR DAILY — DAILY. The rate card still says only "Peak hours
915
+ // are 01:00 - 04:00 and 06:00 - 10:00 UTC (all other hours are
916
+ // off-peak)" with no day qualifier (press coverage had described them as
917
+ // weekday-only). `peakPricing` has no day-of-week concept, so if the
918
+ // provider ever qualifies these by day this over-bills every weekend
919
+ // peak window and needs the FIELD extended, not the numbers nudged.
920
+ // 2. THE CN-VS-US DEFAULT — settled per model, and they differ. Post-rise
921
+ // DeepInfra's US flash rates (0.08/0.18/0.016) are BELOW CN's off-peak
922
+ // on both input and output, so Flash defaults US; Pro's US re-host still
923
+ // bills ~2.3× its native card, so Pro stays CN. See each entry.
918
924
  // 2026-07-31: DeepSeek-V4-Flash OFFICIAL API launched in public beta — the
919
925
  // SAME `deepseek-v4-flash` id now serves the re-post-trained 0731 build
920
926
  // (same architecture/size; much stronger agent benchmarks — beats
@@ -944,11 +950,13 @@ export const MODELS = [
944
950
  supportsVision: false,
945
951
  supportsPromptCaching: true,
946
952
  supportsTools: true,
947
- inputPricePerMTok: 0.435,
948
- outputPricePerMTok: 0.87,
953
+ // Off-peak rates; peak is `peakPricing.multiplier` × these (see below).
954
+ inputPricePerMTok: 0.66,
955
+ outputPricePerMTok: 1.98,
949
956
  // DeepSeek automatic context cache: absolute cache-hit price ($/M).
950
- cacheReadPricePerMTok: 0.003625,
951
- cacheWritePricePerMTok: 0.435,
957
+ cacheReadPricePerMTok: 0.022,
958
+ // DeepSeek charges no cache-write premium — write bills at input.
959
+ cacheWritePricePerMTok: 0.66,
952
960
  // Native-China DEFAULT (owner decision 2026-08-01, re-derived 2026-08-14):
953
961
  // the US re-host (DeepInfra) bills ~3× list and ~28× cache reads, and
954
962
  // agentic input is ~94% cache hits, so US processing ran ~5.7× native on
@@ -972,28 +980,18 @@ export const MODELS = [
972
980
  regionPricing: {
973
981
  us: { inputPricePerMTok: 1.3, outputPricePerMTok: 2.6, cacheReadPricePerMTok: 0.1 },
974
982
  },
975
- // The peak-hour 2× surcharge is now ON the rate card with a dated switch
976
- // (2026-08-13 announcement, effective 2026-08-16T16:00Z) — so it is staged
977
- // below rather than live. The previously pre-wired windows had been REMOVED
978
- // for over-billing every peak-window turn 2× for weeks against a rate card
979
- // that showed a single flat rate; staging is what keeps this from repeating
980
- // in the other direction. Peak = 01:00-04:00 and 06:00-10:00 UTC (Beijing
981
- // business hours), which is 2× the off-peak rates exactly.
982
- scheduledPricing: {
983
- effectiveFrom: '2026-08-16T16:00:00Z',
984
- inputPricePerMTok: 0.66,
985
- outputPricePerMTok: 1.98,
986
- cacheReadPricePerMTok: 0.022,
987
- // DeepSeek charges no cache-write premium — write bills at input.
988
- cacheWritePricePerMTok: 0.66,
989
- peakPricing: {
990
- windows: [
991
- { startMinuteUtc: 60, endMinuteUtc: 240 },
992
- { startMinuteUtc: 360, endMinuteUtc: 600 },
993
- ],
994
- multiplier: 2,
995
- },
996
- source: 'https://api-docs.deepseek.com/quick_start/pricing/',
983
+ // The peak-hour 2× surcharge is ON the rate card and LIVE since
984
+ // 2026-08-16T16:00Z. Peak = 01:00-04:00 and 06:00-10:00 UTC (Beijing
985
+ // business hours), daily, at exactly 2× the off-peak rates above. Windows
986
+ // like these were once pre-wired ahead of the card and over-billed every
987
+ // peak-window turn for weeks — hence the rule that they only exist here
988
+ // once the provider's own page shows them, which it now does.
989
+ peakPricing: {
990
+ windows: [
991
+ { startMinuteUtc: 60, endMinuteUtc: 240 },
992
+ { startMinuteUtc: 360, endMinuteUtc: 600 },
993
+ ],
994
+ multiplier: 2,
997
995
  },
998
996
  // Not published by DeepSeek — best-effort estimate.
999
997
  knowledgeCutoff: '2025-07-01',
@@ -1017,11 +1015,13 @@ export const MODELS = [
1017
1015
  // executor — the model the IDE picks when none is chosen. Exactly one model
1018
1016
  // in this catalog may carry freeTier (enforced by lookup.test.ts).
1019
1017
  freeTier: true,
1020
- inputPricePerMTok: 0.14,
1021
- outputPricePerMTok: 0.28,
1018
+ // Off-peak rates; peak is `peakPricing.multiplier` × these (see below).
1019
+ inputPricePerMTok: 0.22,
1020
+ outputPricePerMTok: 0.66,
1022
1021
  // DeepSeek automatic context cache: absolute cache-hit price ($/M).
1023
- cacheReadPricePerMTok: 0.0028,
1024
- cacheWritePricePerMTok: 0.14,
1022
+ cacheReadPricePerMTok: 0.007,
1023
+ // DeepSeek charges no cache-write premium — write bills at input.
1024
+ cacheWritePricePerMTok: 0.22,
1025
1025
  // US (DeepInfra) DEFAULT as of 2026-08-16 — flipped from CN when DeepSeek's
1026
1026
  // rise landed (owner decision 2026-08-14). CN was cheaper on real traffic
1027
1027
  // only because of its cache reads; the rise takes those from $0.0028 to
@@ -1043,22 +1043,15 @@ export const MODELS = [
1043
1043
  regionPricing: {
1044
1044
  us: { inputPricePerMTok: 0.08, outputPricePerMTok: 0.18, cacheReadPricePerMTok: 0.016 },
1045
1045
  },
1046
- // Peak-hour surcharge staged, not live (see deepseek-v4-pro).
1047
- scheduledPricing: {
1048
- effectiveFrom: '2026-08-16T16:00:00Z',
1049
- inputPricePerMTok: 0.22,
1050
- outputPricePerMTok: 0.66,
1051
- cacheReadPricePerMTok: 0.007,
1052
- // DeepSeek charges no cache-write premium — write bills at input.
1053
- cacheWritePricePerMTok: 0.22,
1054
- peakPricing: {
1055
- windows: [
1056
- { startMinuteUtc: 60, endMinuteUtc: 240 },
1057
- { startMinuteUtc: 360, endMinuteUtc: 600 },
1058
- ],
1059
- multiplier: 2,
1060
- },
1061
- source: 'https://api-docs.deepseek.com/quick_start/pricing/',
1046
+ // Peak-hour surcharge live since 2026-08-16T16:00Z (see deepseek-v4-pro).
1047
+ // It applies to the NATIVE CN card only — this model defaults to the US
1048
+ // re-host, which is flat, so most turns never take it.
1049
+ peakPricing: {
1050
+ windows: [
1051
+ { startMinuteUtc: 60, endMinuteUtc: 240 },
1052
+ { startMinuteUtc: 360, endMinuteUtc: 600 },
1053
+ ],
1054
+ multiplier: 2,
1062
1055
  },
1063
1056
  // Not published by DeepSeek — best-effort estimate.
1064
1057
  knowledgeCutoff: '2025-07-01',
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@molecule/api-resource-ai-models",
3
- "version": "1.2.1",
3
+ "version": "1.2.3",
4
4
  "description": "AI model catalog — server-side source of truth plus an authentication-gated discovery endpoint",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",