@molecule/api-resource-ai-models 1.0.2 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/models.js CHANGED
@@ -27,6 +27,18 @@
27
27
  * control) carries `thinkingConfigurable: false` and OMITS both fields —
28
28
  * there is nothing to tune.
29
29
  *
30
+ * ONE GENERATION PER FAMILY. When a provider ships a newer generation of a
31
+ * model line, the older entry gets `supersededBy: '<newer id>'` and stops being
32
+ * offered — the picker never shows both `qwen3.7-max` and `qwen3.8-max`. The
33
+ * entry is NEVER deleted: `getModel()` still resolves it so saved selections and
34
+ * historical usage stay priceable, and a persisted id resolves forward to the
35
+ * successor. Supersede only within the same TIER: a cheaper or specialist model
36
+ * with no newer equivalent (`gemini-3.1-pro-preview`, `qwen3-coder-plus`,
37
+ * `kimi-k2.7-code`, `grok-build-0.1`) keeps at most `deprecatedAt`, so every
38
+ * provider keeps a real choice. `__tests__/lookup.test.ts` fails on any two
39
+ * selectable models of one family at different versions that aren't a
40
+ * documented exception.
41
+ *
30
42
  * Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
31
43
  * 2026-07-30 GPT-5.6 repricing — cross-check prices against models.dev with
32
44
  * `npm run check:model-freshness` from the workspace root):
@@ -48,21 +60,31 @@
48
60
  * 2026-07-21 $1.50/$7.50 supersedes 3.5-flash as the agentic flagship;
49
61
  * gemini-3.1-pro-preview still the pro tier — "3.5 Pro" has NOT shipped as
50
62
  * of 2026-07-28 despite the coming-soon badge; do not add until it has an id)
63
+ * (re-verified 2026-08-13: gemini-3.7-flash "New Stable" — supersedes
64
+ * 3.6-flash as the flash flagship at the SAME list price ($1.50/$7.50, cache
65
+ * read $0.15), with a launch promo ($0.75/$3.75, cache read $0.075) through
66
+ * 2026-12-31 billed here at list; specs from /docs/models/gemini-3.7-flash:
67
+ * 1M ctx / 65,536 out, thinking low|medium|high (no minimal), vision, tools,
68
+ * caching, search grounding, code execution, url context)
51
69
  * - xAI: https://docs.x.ai/developers/models + /developers/grok-4-5
52
70
  * (grok-4.5 flagship 2026-07-08: $2/$6, 500K ctx, ≥200K prompts bill 2× —
53
71
  * not modeled; reasoning_effort low|medium|high default high, image input;
54
72
  * grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
55
73
  * grok-code-fast-1 no longer listed — retires 2026-08-15)
56
- * - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (unchanged V4
57
- * Pro/Flash pricing; legacy deepseek-chat/-reasoner ids fully retired
58
- * 2026-07-24 — never in this catalog; the announced peak-hour 2× surcharge is
59
- * still NOT active as of 2026-07-28, see the entries)
60
- * - Moonshot: https://platform.kimi.ai/docs/models (kimi-k3 flagship 2026-07-16
74
+ * - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
75
+ * 2026-08-14; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
76
+ * never in this catalog. V4-Pro GA on 2026-08-13 came with a price RISE
77
+ * effective 2026-08-16T16:00Z plus the long-announced peak-hour 2×: both
78
+ * entries carry it as `scheduledPricing`, so today's rates bill until that
79
+ * instant and the new ones after. Re-verify weekday-vs-daily peak windows and
80
+ * the CN/US region default once it lands — see the entries.)
81
+ * - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
82
+ * the US re-host (kimi-k3 flagship 2026-07-16
61
83
  * — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
62
84
  * reasoning_content that must be replayed through tool loops, the same
63
- * constraint that keeps kimi-k2.7-code out; add BOTH once the moonshot bond
64
- * supports preserved thinking + reasoning_effort low|high|max. kimi-k2.6
65
- * remains the newest model the bond can run correctly.)
85
+ * constraint that kept kimi-k2.7-code out. BOTH are now in the catalog: the
86
+ * moonshot bond gained preserved thinking (reasoning replayed through tool
87
+ * loops), so kimi-k3 is the Moonshot pick.)
66
88
  * - MiniMax: https://platform.minimax.io/docs/guides/pricing-paygo (unchanged;
67
89
  * minimax-m3 $0.30/$1.20 is a "permanent 50% off" list rate)
68
90
  * - Alibaba: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
@@ -191,9 +213,11 @@ export const MODELS = [
191
213
  cacheReadPricePerMTok: 0.5,
192
214
  cacheWritePricePerMTok: 6.25,
193
215
  knowledgeCutoff: '2026-01-01',
194
- // Superseded by claude-opus-5 (same price); still served upstream and the
195
- // recommended refusal-fallback target. Selectable under "Older models".
216
+ // Superseded by claude-opus-5 (same price, same tier); still served upstream
217
+ // and the recommended refusal-fallback target, so it stays priceable and
218
+ // callable by id — it is just not OFFERED, since opus-5 is a drop-in.
196
219
  deprecatedAt: '2026-07-28',
220
+ supersededBy: 'claude-opus-5',
197
221
  },
198
222
  {
199
223
  id: 'claude-sonnet-5',
@@ -255,9 +279,12 @@ export const MODELS = [
255
279
  knowledgeCutoff: '2026-01-01',
256
280
  // Superseded by claude-opus-4-8 (launched 2026-05-28) at identical pricing;
257
281
  // still Active upstream (deprecations page 2026-08-06: retires no sooner
258
- // than 2027-04-16). Selectable under "Older models". NO fast mode —
259
- // speed:"fast" on 4.7 returns an error (pricing page, fast-mode section).
282
+ // than 2027-04-16). NO fast mode — speed:"fast" on 4.7 returns an error
283
+ // (pricing page, fast-mode section). `supersededBy` names the CURRENT
284
+ // selectable Opus (opus-5), not the also-superseded 4.8, so a saved
285
+ // selection resolves forward in one hop.
260
286
  deprecatedAt: '2026-05-28',
287
+ supersededBy: 'claude-opus-5',
261
288
  },
262
289
  {
263
290
  id: 'claude-opus-4-6',
@@ -285,8 +312,9 @@ export const MODELS = [
285
312
  cacheReadPricePerMTok: 0.5,
286
313
  cacheWritePricePerMTok: 6.25,
287
314
  knowledgeCutoff: '2025-05-01',
288
- // Superseded by claude-opus-4-8; kept selectable (Older models) + priceable.
315
+ // Superseded by the current Opus (opus-5) — kept priceable, not offered.
289
316
  deprecatedAt: '2026-06-16',
317
+ supersededBy: 'claude-opus-5',
290
318
  },
291
319
  {
292
320
  id: 'claude-sonnet-4-6',
@@ -315,8 +343,9 @@ export const MODELS = [
315
343
  cacheReadPricePerMTok: 0.3,
316
344
  cacheWritePricePerMTok: 3.75,
317
345
  knowledgeCutoff: '2025-08-01',
318
- // Superseded by claude-sonnet-5; kept selectable (Older models) + priceable.
346
+ // Superseded by claude-sonnet-5 (same tier) — kept priceable, not offered.
319
347
  deprecatedAt: '2026-07-07',
348
+ supersededBy: 'claude-sonnet-5',
320
349
  },
321
350
  {
322
351
  id: 'claude-haiku-4-5-20251001',
@@ -471,9 +500,10 @@ export const MODELS = [
471
500
  cacheReadPricePerMTok: 0.5,
472
501
  cacheWritePricePerMTok: 5,
473
502
  knowledgeCutoff: '2025-12-01',
474
- // Superseded by gpt-5.6-sol (same price); still listed as current by
475
- // OpenAI. Selectable under "Older models".
503
+ // Superseded by gpt-5.6-sol (same frontier tier, same $5/$30); still listed
504
+ // as current by OpenAI, so it stays priceable — it is just not offered.
476
505
  deprecatedAt: '2026-07-09',
506
+ supersededBy: 'gpt-5.6-sol',
477
507
  },
478
508
  {
479
509
  id: 'gpt-5.4',
@@ -501,9 +531,11 @@ export const MODELS = [
501
531
  cacheWritePricePerMTok: 2.5,
502
532
  knowledgeCutoff: '2025-08-31',
503
533
  // OpenAI still lists gpt-5.4 as current, but gpt-5.6-terra covers this
504
- // tier at the same price — moved to "Older models" (deprecatedAt is OUR
505
- // picker taxonomy, not OpenAI's deprecations page).
534
+ // balanced tier for LESS ($2/$12 vs $2.50/$15) — superseded, so the picker
535
+ // offers only the 5.6 generation (this is OUR taxonomy, not OpenAI's
536
+ // deprecations page; the model stays priceable).
506
537
  deprecatedAt: '2026-07-28',
538
+ supersededBy: 'gpt-5.6-terra',
507
539
  },
508
540
  {
509
541
  id: 'gpt-5.4-mini',
@@ -530,11 +562,12 @@ export const MODELS = [
530
562
  cacheReadPricePerMTok: 0.075,
531
563
  cacheWritePricePerMTok: 0.75,
532
564
  knowledgeCutoff: '2025-08-31',
533
- // OpenAI still lists gpt-5.4-mini as current, but like gpt-5.4 above it's
534
- // superseded in our lineup (cheap/fast tier is better served by the newer
535
- // models) — moved to "Older models" (deprecatedAt is OUR picker taxonomy,
536
- // not OpenAI's deprecations page).
565
+ // Superseded by gpt-5.6-luna, which IS the newer cheap/fast tier and is
566
+ // strictly better on every axis that made this the budget pick: $0.20/$1.20
567
+ // vs $0.75/$4.50 after the 2026-07-30 repricing, and a 1M window vs 400K.
568
+ // Hiding it therefore costs OpenAI no cheap option. Stays priceable.
537
569
  deprecatedAt: '2026-08-01',
570
+ supersededBy: 'gpt-5.6-luna',
538
571
  },
539
572
  // ---------------------------------------------------------------------------
540
573
  // Google
@@ -548,11 +581,46 @@ export const MODELS = [
548
581
  // replace outright: the google bond has never been implemented/wired, so no
549
582
  // historical usage can reference the old ids.
550
583
  // ---------------------------------------------------------------------------
584
+ {
585
+ id: 'gemini-3.7-flash',
586
+ provider: 'google',
587
+ label: 'Gemini 3.7 Flash',
588
+ description: 'Google agentic flagship — complex coding & multi-step execution',
589
+ // Verified against /docs/models/gemini-3.7-flash (2026-08-13).
590
+ contextWindow: 1_048_576,
591
+ maxOutputTokens: 65_536,
592
+ supportsThinking: true,
593
+ thinkingBudgetTokens: 10_000,
594
+ thinkingConfigurable: true,
595
+ // thinking_level low|medium|high — minimal NOT supported on this model.
596
+ supportedEffortLevels: ['low', 'medium', 'high'],
597
+ defaultEffortLevel: 'medium',
598
+ supportsVision: true,
599
+ supportsPromptCaching: true,
600
+ supportsTools: true,
601
+ webSearchToolType: 'google_search',
602
+ codeExecutionToolType: 'code_execution',
603
+ webFetchToolType: 'url_context',
604
+ // "New Stable" 2026-08-13. LIST price $1.50/$7.50 — same as 3.6-flash.
605
+ // Google runs a launch promo ($0.75/$3.75, cache read $0.075) through
606
+ // 2026-12-31; billed here at standard list so metering never under-charges
607
+ // (same policy as claude-sonnet-5's intro pricing — see the matching
608
+ // KNOWN_DIVERGENCES entry in scripts/check-model-freshness.mjs, expiring
609
+ // 2026-12-31).
610
+ inputPricePerMTok: 1.5,
611
+ outputPricePerMTok: 7.5,
612
+ // Gemini context cache: read $0.15/M (0.1× input), no write premium
613
+ // (storage billed separately per hour — not modeled).
614
+ cacheReadPricePerMTok: 0.15,
615
+ cacheWritePricePerMTok: 1.5,
616
+ // Not on Google's docs — models.dev reports 2026-03 (lead, not authority).
617
+ knowledgeCutoff: '2026-03-01',
618
+ },
551
619
  {
552
620
  id: 'gemini-3.6-flash',
553
621
  provider: 'google',
554
622
  label: 'Gemini 3.6 Flash',
555
- description: 'Google agentic flagship — frontier intelligence + grounding',
623
+ description: 'Previous Google agentic flagship — frontier intelligence + grounding',
556
624
  // Window/output not on the pricing page — carried over from 3.5-flash;
557
625
  // re-verify against /docs/models.
558
626
  contextWindow: 1_048_576,
@@ -579,6 +647,11 @@ export const MODELS = [
579
647
  cacheWritePricePerMTok: 1.5,
580
648
  // Not published — best-effort estimate.
581
649
  knowledgeCutoff: '2026-01-01',
650
+ // Superseded by gemini-3.7-flash (2026-08-13) — same flash tier, same list
651
+ // price; Google's own models page now calls 3.6 "previous-generation".
652
+ // Still served upstream, so it stays priceable.
653
+ deprecatedAt: '2026-08-13',
654
+ supersededBy: 'gemini-3.7-flash',
582
655
  },
583
656
  {
584
657
  id: 'gemini-3.5-flash',
@@ -608,9 +681,12 @@ export const MODELS = [
608
681
  cacheReadPricePerMTok: 0.15,
609
682
  cacheWritePricePerMTok: 1.5,
610
683
  knowledgeCutoff: '2025-01-01',
611
- // Superseded by gemini-3.6-flash (2026-07-21); still served upstream.
612
- // Selectable under "Older models".
684
+ // Superseded within the flash tier (first by 3.6-flash on 2026-07-21, now
685
+ // pointed one hop to gemini-3.7-flash — supersededBy must target a
686
+ // SELECTABLE model, never a chain). Still served upstream, so it stays
687
+ // priceable.
613
688
  deprecatedAt: '2026-07-21',
689
+ supersededBy: 'gemini-3.7-flash',
614
690
  },
615
691
  {
616
692
  id: 'gemini-3.1-pro-preview',
@@ -640,6 +716,10 @@ export const MODELS = [
640
716
  cacheReadPricePerMTok: 0.2,
641
717
  cacheWritePricePerMTok: 2,
642
718
  knowledgeCutoff: '2025-01-01',
719
+ // NOT superseded despite the lower version number: this is Google's only
720
+ // PRO-tier id (no GA "3.5/3.6 Pro" exists), and the 3.6 flash flagship is a
721
+ // different tier. Superseding it would leave Google with no deep-reasoning
722
+ // option at all — see `ModelDefinition.supersededBy` (same-tier rule).
643
723
  },
644
724
  // ---------------------------------------------------------------------------
645
725
  // xAI (Grok)
@@ -703,9 +783,13 @@ export const MODELS = [
703
783
  cacheReadPricePerMTok: 0.2,
704
784
  cacheWritePricePerMTok: 1.25,
705
785
  knowledgeCutoff: '2025-12-01',
706
- // Superseded by grok-4.5 as the xAI pick (4.3 keeps the bigger 1M window
707
- // — the reason it stays selectable under "Older models").
786
+ // Superseded by grok-4.5: the previous version of the same general-purpose
787
+ // Grok line, not a separately-named tier. It keeps a bigger window (1M vs
788
+ // 500K) and a lower price, which is why it was previously left selectable —
789
+ // but offering two generations of one family is exactly what the picker no
790
+ // longer does, and grok-4.5 is xAI's own recommendation. Stays priceable.
708
791
  deprecatedAt: '2026-07-28',
792
+ supersededBy: 'grok-4.5',
709
793
  },
710
794
  {
711
795
  id: 'grok-build-0.1',
@@ -732,6 +816,9 @@ export const MODELS = [
732
816
  // Not published by xAI — best-effort estimate (grok-4-generation base).
733
817
  knowledgeCutoff: '2025-06-01',
734
818
  // Niche coding beta; grok-4.5 is the xAI pick — kept out of the main list.
819
+ // NOT superseded: its own family (grok-build) has no newer version, and it
820
+ // is xAI's cheapest tool-capable model, so it stays selectable under
821
+ // "Older models" (and is the deliberately-weak live selftest target).
735
822
  deprecatedAt: '2026-07-28',
736
823
  },
737
824
  {
@@ -799,20 +886,44 @@ export const MODELS = [
799
886
  // DeepSeek
800
887
  // Verified: https://api-docs.deepseek.com/quick_start/pricing
801
888
  // https://api-docs.deepseek.com/guides/thinking_mode
802
- // https://api-docs.deepseek.com/updates/ (2026-07-31)
889
+ // https://api-docs.deepseek.com/updates/ (2026-08-14)
890
+ // 2026-08-13: V4-Pro GA — and with it the price rise that the "coming soon"
891
+ // note below had been waiting on. It is STAGED, not applied: both models
892
+ // carry `scheduledPricing` effective 2026-08-16T16:00Z, so the catalog bills
893
+ // today's verified rates until that instant and the new ones after it, with
894
+ // nobody landing an edit at 16:00 UTC on a Sunday. The new card is
895
+ // off-peak/peak (peak = exactly 2× off-peak), so it maps onto base rates +
896
+ // `peakPricing` multiplier 2 — which is why the peak windows removed below
897
+ // come back here rather than as flat rates.
898
+ // pro off-peak 0.66 / 1.98, cache hit 0.022 (peak 1.32 / 3.96 / 0.044)
899
+ // flash off-peak 0.22 / 0.66, cache hit 0.007 (peak 0.44 / 1.32 / 0.014)
900
+ // Cache HITS are the real move — pro 0.003625 → 0.022 (6.1×) off-peak, 0.044
901
+ // (12.1×) at peak — and agentic input is ~94% cache hits, so effective input
902
+ // cost rises far more than the list prices suggest. Both are free-tier models
903
+ // (flash is `freeTier`, pro is the free-tier planner) on the CN default.
904
+ // TWO things to re-verify once it lands (2026-08-17):
905
+ // 1. WEEKDAYS OR DAILY. The rate card says only "Peak hours are 01:00 -
906
+ // 04:00 and 06:00 - 10:00 UTC (all other hours are off-peak)" with no
907
+ // day qualifier, so the windows below are DAILY per the provider's own
908
+ // doc; press coverage described them as weekday-only. `peakPricing` has
909
+ // no day-of-week concept, so if it is weekday-only this over-bills every
910
+ // weekend peak window and needs the field extended, not the numbers
911
+ // nudged.
912
+ // 2. THE CN-VS-US DEFAULT. `regions: ['cn', 'us']` defaults to CN on an
913
+ // owner decision (2026-08-01) taken when CN ran ~5.7× cheaper on real
914
+ // traffic. Post-change DeepInfra's US flash rates (0.08/0.18/0.016) are
915
+ // BELOW CN's new off-peak on both input and output — CN wins only on
916
+ // cache reads. Re-derive against measured cache-hit ratios before
917
+ // leaving the default where it is.
803
918
  // 2026-07-31: DeepSeek-V4-Flash OFFICIAL API launched in public beta — the
804
919
  // SAME `deepseek-v4-flash` id now serves the re-post-trained 0731 build
805
920
  // (same architecture/size; much stronger agent benchmarks — beats
806
- // V4-Pro-Preview on Terminal Bench 2.1 / DeepSWE). No pricing/limit/
807
- // capability changes. V4-Pro official release "coming soon" — re-verify
808
- // pricing THEN (the announced peak-hour 2× was tied to the V4 official
809
- // rollout and is still not on the rate card).
921
+ // V4-Pro-Preview on Terminal Bench 2.1 / DeepSWE).
810
922
  // OpenAI/Anthropic-compatible API; text/code only (no vision); 1M context,
811
923
  // 384K max output, automatic context (prompt) caching with ABSOLUTE cache-hit
812
924
  // prices (~1/50–1/120 of miss — not the old 0.1× rule). Launch discount made
813
- // PERMANENT 2026-05-23 (Pro $1.74/$3.48 → $0.435/$0.87). Peak-hour 2×
814
- // pricing announced for the mid-Jul 2026 "V4 official" release — re-verify
815
- // then. Thinking now defaults ENABLED upstream and supports tool calling
925
+ // PERMANENT 2026-05-23 (Pro $1.74/$3.48 → $0.435/$0.87), and ENDED by the
926
+ // 2026-08-16 rise above. Thinking now defaults ENABLED upstream and supports tool calling
816
927
  // (reasoning_effort: high|max), BUT tool loops in thinking mode must replay
817
928
  // assistant reasoning_content on every subsequent request (400 on omission).
818
929
  // The bond explicitly sends thinking:{type:"disabled"} — Synthase runs
@@ -838,28 +949,49 @@ export const MODELS = [
838
949
  // DeepSeek automatic context cache: absolute cache-hit price ($/M).
839
950
  cacheReadPricePerMTok: 0.003625,
840
951
  cacheWritePricePerMTok: 0.435,
841
- // Native-China DEFAULT (owner decision 2026-08-01): the US re-host
842
- // (DeepInfra) bills ~3× list and ~28× cache reads, and agentic input is
843
- // ~94% cache hits, so US processing ran ~5.7× native on real traffic.
844
- // Users opt into US per model via the picker's region control.
952
+ // Native-China DEFAULT (owner decision 2026-08-01, re-derived 2026-08-14):
953
+ // the US re-host (DeepInfra) bills ~3× list and ~28× cache reads, and
954
+ // agentic input is ~94% cache hits, so US processing ran ~5.7× native on
955
+ // real traffic. The 2026-08-16 rise narrows that to ~2.3× — still decisive,
956
+ // so Pro stays CN while Flash flipped to US (see its note). Users opt into
957
+ // US per model via the picker's region control.
845
958
  regions: ['cn', 'us'],
846
- // The free tier PLANS with this model on the cheap native host (it is the
847
- // molecule-dev FREE_TIER_MODELS.plan), so CN is free-tier selectable; the
848
- // ~3× US re-host stays paid-only (free users switch to Flash for US).
849
- freeTierRegions: ['cn'],
959
+ // No freeTierRegions: the free tier stopped planning with this model on
960
+ // 2026-08-14 (minimax-m3 took over — cheaper, and it beat this model on the
961
+ // selection self-test). The carve-out only ever existed to keep the free
962
+ // tier's OWN plan default usable, and `freeTierAllows` checks
963
+ // `FREE_TIER_MODELS[mode] === modelId` before it looks at regions, so
964
+ // leaving it here would widen nothing — it would just claim a free-tier
965
+ // relationship that no longer exists.
850
966
  // US = DeepInfra, verified 2026-08-01 via api.deepinfra.com/models/
851
967
  // deepseek-ai/DeepSeek-V4-Pro. No cache-write premium (omitted → region
852
968
  // input rate).
853
969
  regionPricing: {
854
970
  us: { inputPricePerMTok: 1.3, outputPricePerMTok: 2.6, cacheReadPricePerMTok: 0.1 },
855
971
  },
856
- // The announced peak-hour 2× surcharge (Beijing business hours) is STILL
857
- // NOT ACTIVE as of 2026-07-28 — the official rate card lists a single flat
858
- // rate, and no switch-over date is published. The pre-wired windows were
859
- // REMOVED: they had been over-billing every peak-window turn 2× for weeks
860
- // (this is the free-tier default model, so that directly shrank free
861
- // users' allowances). Re-add via `peakPricing` the day DeepSeek's rate
862
- // card actually shows the surcharge.
972
+ // The peak-hour 2× surcharge is now ON the rate card with a dated switch
973
+ // (2026-08-13 announcement, effective 2026-08-16T16:00Z) — so it is staged
974
+ // below rather than live. The previously pre-wired windows had been REMOVED
975
+ // for over-billing every peak-window turn 2× for weeks against a rate card
976
+ // that showed a single flat rate; staging is what keeps this from repeating
977
+ // in the other direction. Peak = 01:00-04:00 and 06:00-10:00 UTC (Beijing
978
+ // business hours), which is 2× the off-peak rates exactly.
979
+ scheduledPricing: {
980
+ effectiveFrom: '2026-08-16T16:00:00Z',
981
+ inputPricePerMTok: 0.66,
982
+ outputPricePerMTok: 1.98,
983
+ cacheReadPricePerMTok: 0.022,
984
+ // DeepSeek charges no cache-write premium — write bills at input.
985
+ cacheWritePricePerMTok: 0.66,
986
+ peakPricing: {
987
+ windows: [
988
+ { startMinuteUtc: 60, endMinuteUtc: 240 },
989
+ { startMinuteUtc: 360, endMinuteUtc: 600 },
990
+ ],
991
+ multiplier: 2,
992
+ },
993
+ source: 'https://api-docs.deepseek.com/quick_start/pricing/',
994
+ },
863
995
  // Not published by DeepSeek — best-effort estimate.
864
996
  knowledgeCutoff: '2025-07-01',
865
997
  },
@@ -887,15 +1019,44 @@ export const MODELS = [
887
1019
  // DeepSeek automatic context cache: absolute cache-hit price ($/M).
888
1020
  cacheReadPricePerMTok: 0.0028,
889
1021
  cacheWritePricePerMTok: 0.14,
890
- // Native-China default, matching deepseek-v4-pro (see its note) — even
891
- // though Flash's US list price is BELOW native, its cache reads are 6.4×,
892
- // and the plan/execute pair defaults to one region deliberately.
893
- regions: ['cn', 'us'],
894
- // US = DeepInfra, verified 2026-08-01 (api.deepinfra.com/models/…V4-Flash).
1022
+ // US (DeepInfra) DEFAULT as of 2026-08-16 — flipped from CN when DeepSeek's
1023
+ // rise landed (owner decision 2026-08-14). CN was cheaper on real traffic
1024
+ // only because of its cache reads; the rise takes those from $0.0028 to
1025
+ // $0.007 (peak $0.014) against DeepInfra's flat $0.016, which is no longer
1026
+ // enough to carry the 1.6-3.1x it now loses on fresh input and output. On
1027
+ // the agentic mix this model actually serves (~94% cache hits) US is
1028
+ // cheaper at EVERY hour: 0.114c/turn flat vs 0.152c off-peak and 0.303c at
1029
+ // peak. It is also flat-rate, so free-tier cost stops varying by Beijing
1030
+ // business hours. Re-derive if the cache-hit ratio drops much below ~90%,
1031
+ // where CN's cheaper reads start winning again. This deliberately splits
1032
+ // the plan/execute pair across regions — Pro stays CN because its US
1033
+ // re-host is ~2.3x its own native rate even after the rise.
1034
+ regions: ['us', 'cn'],
1035
+ // US = DeepInfra, verified 2026-08-13 against the id the bond actually
1036
+ // sends: `deepseek-ai/DeepSeek-V4-Flash-0731`, the official release that
1037
+ // supersedes the preview weights still served under the un-dated id
1038
+ // (cents_per_input_token 0.000008, cents_per_output_token 0.000018,
1039
+ // rate_per_input_token_cached 0.2 → cache read = 0.2 × input).
895
1040
  regionPricing: {
896
- us: { inputPricePerMTok: 0.09, outputPricePerMTok: 0.18, cacheReadPricePerMTok: 0.018 },
1041
+ us: { inputPricePerMTok: 0.08, outputPricePerMTok: 0.18, cacheReadPricePerMTok: 0.016 },
1042
+ },
1043
+ // Peak-hour surcharge staged, not live (see deepseek-v4-pro).
1044
+ scheduledPricing: {
1045
+ effectiveFrom: '2026-08-16T16:00:00Z',
1046
+ inputPricePerMTok: 0.22,
1047
+ outputPricePerMTok: 0.66,
1048
+ cacheReadPricePerMTok: 0.007,
1049
+ // DeepSeek charges no cache-write premium — write bills at input.
1050
+ cacheWritePricePerMTok: 0.22,
1051
+ peakPricing: {
1052
+ windows: [
1053
+ { startMinuteUtc: 60, endMinuteUtc: 240 },
1054
+ { startMinuteUtc: 360, endMinuteUtc: 600 },
1055
+ ],
1056
+ multiplier: 2,
1057
+ },
1058
+ source: 'https://api-docs.deepseek.com/quick_start/pricing/',
897
1059
  },
898
- // Peak-hour surcharge NOT active (see deepseek-v4-pro) — windows removed.
899
1060
  // Not published by DeepSeek — best-effort estimate.
900
1061
  knowledgeCutoff: '2025-07-01',
901
1062
  },
@@ -910,6 +1071,12 @@ export const MODELS = [
910
1071
  // and kimi-k2.7-code (coding flagship — forced thinking, no depth knob).
911
1072
  // kimi-k2.x thinking stays on/off only; the bond disables it for those by
912
1073
  // default (KIMI_REASONING_EFFORT env tunes it).
1074
+ // EVERY moonshot entry declares `regions` explicitly. A model that omits the
1075
+ // field defaults to `['us']` (effectiveModelRegion), which would route it to
1076
+ // the bare `moonshot` bond — DeepInfra when its key is set — with an id that
1077
+ // host has never heard of, i.e. a 404 at dispatch. The freshness gate's
1078
+ // region-re-host coverage check fails on exactly that (a us-region moonshot
1079
+ // model missing from the bond's modelMap).
913
1080
  // ---------------------------------------------------------------------------
914
1081
  {
915
1082
  id: 'kimi-k3',
@@ -936,8 +1103,21 @@ export const MODELS = [
936
1103
  // Automatic context cache: absolute cache-hit price ($0.30/M = 0.1× input).
937
1104
  cacheReadPricePerMTok: 0.3,
938
1105
  cacheWritePricePerMTok: 3,
939
- // No US re-host exists (not on DeepInfra) — pinned to native China.
940
- regions: ['cn'],
1106
+ // US default = DeepInfra, verified 2026-08-13 against
1107
+ // api.deepinfra.com/models/moonshotai/Kimi-K3: cents_per_input_token
1108
+ // 0.000285 → $2.85/MTok, cents_per_output_token 0.001425 → $14.25/MTok,
1109
+ // rate_per_input_token_cached 0.1 → cache read $0.285/MTok, and
1110
+ // rate_per_input_token_cache_write null → no write premium (the omitted
1111
+ // cache-write field falls back to the region's input rate). Cheaper than
1112
+ // Moonshot native on every axis, which is why US leads (see the
1113
+ // cheapest-default-region invariant in __tests__/lookup.test.ts).
1114
+ // The host serves the full 1M context, unquantized, and returns
1115
+ // reasoning_content while accepting reasoning_effort — probed live
1116
+ // 2026-08-13 — so the preserved-thinking tool loop works there unchanged.
1117
+ regions: ['us', 'cn'],
1118
+ regionPricing: {
1119
+ us: { inputPricePerMTok: 2.85, outputPricePerMTok: 14.25, cacheReadPricePerMTok: 0.285 },
1120
+ },
941
1121
  // Not published — best-effort estimate.
942
1122
  knowledgeCutoff: '2026-01-01',
943
1123
  },
@@ -962,15 +1142,21 @@ export const MODELS = [
962
1142
  // Automatic context cache: absolute cache-hit price ($0.19/M = 0.2× input).
963
1143
  cacheReadPricePerMTok: 0.19,
964
1144
  cacheWritePricePerMTok: 0.95,
965
- // US default (DeepInfra bills below native here). Verified 2026-08-01.
1145
+ // US default (DeepInfra bills below native here). Re-verified 2026-08-13
1146
+ // against api.deepinfra.com/models/moonshotai/Kimi-K2.7-Code — the host
1147
+ // repriced since 2026-08-01 ($0.74/$3.50/$0.15): cents_per_input_token
1148
+ // 0.000068, cents_per_output_token 0.00034, rate_per_input_token_cached
1149
+ // 0.2 → cache read $0.136, no write premium.
966
1150
  regions: ['us', 'cn'],
967
1151
  regionPricing: {
968
- us: { inputPricePerMTok: 0.74, outputPricePerMTok: 3.5, cacheReadPricePerMTok: 0.15 },
1152
+ us: { inputPricePerMTok: 0.68, outputPricePerMTok: 3.4, cacheReadPricePerMTok: 0.136 },
969
1153
  },
970
1154
  // Not published — best-effort estimate.
971
1155
  knowledgeCutoff: '2025-10-01',
972
- // kimi-k3 is the Moonshot pick; the coding specialist stays selectable
973
- // under "Older models" for anyone who wants the cheaper tier.
1156
+ // kimi-k3 is the Moonshot pick, but this is NOT superseded: the coding
1157
+ // specialist is a distinct, much cheaper tier ($0.95/$4 vs $3/$15) with no
1158
+ // K3 equivalent, so it stays selectable under "Older models" — superseding
1159
+ // it would leave Moonshot with only the flagship.
974
1160
  deprecatedAt: '2026-07-28',
975
1161
  },
976
1162
  {
@@ -1002,8 +1188,9 @@ export const MODELS = [
1002
1188
  us: { inputPricePerMTok: 0.75, outputPricePerMTok: 3.5, cacheReadPricePerMTok: 0.15 },
1003
1189
  },
1004
1190
  knowledgeCutoff: '2025-04-01',
1005
- // Superseded by kimi-k3; moved to "Older models".
1191
+ // Superseded by kimi-k3 (same general-purpose line) — kept priceable.
1006
1192
  deprecatedAt: '2026-07-28',
1193
+ supersededBy: 'kimi-k3',
1007
1194
  },
1008
1195
  {
1009
1196
  id: 'kimi-k2.5',
@@ -1030,9 +1217,11 @@ export const MODELS = [
1030
1217
  us: { inputPricePerMTok: 0.45, outputPricePerMTok: 2.25, cacheReadPricePerMTok: 0.07 },
1031
1218
  },
1032
1219
  knowledgeCutoff: '2024-04-01',
1033
- // Superseded by kimi-k2.6 (still served upstream, no announced retirement);
1034
- // kept selectable (Older models) + priceable.
1220
+ // Two generations behind. `supersededBy` names the current selectable Kimi
1221
+ // (k3) rather than the also-superseded k2.6, so a saved selection resolves
1222
+ // forward in one hop. Still served upstream; stays priceable.
1035
1223
  deprecatedAt: '2026-04-01',
1224
+ supersededBy: 'kimi-k3',
1036
1225
  },
1037
1226
  // ---------------------------------------------------------------------------
1038
1227
  // MiniMax
@@ -1069,8 +1258,21 @@ export const MODELS = [
1069
1258
  // US default. DeepInfra list matches native; only the cache write differs
1070
1259
  // (no premium → region input rate). Verified 2026-08-01.
1071
1260
  regions: ['us', 'cn'],
1261
+ // The free tier PLANS with this model (molecule-dev FREE_TIER_MODELS.plan,
1262
+ // 2026-08-14), so its default US region must be free-tier selectable. It
1263
+ // took over from deepseek-v4-pro@cn: measured on the real starting-point
1264
+ // selection it scored 8/8 against Pro's 7/8 — including the case Pro failed
1265
+ // — at 1.28c/plan-turn flat versus Pro's 2.62c off-peak and 5.24c inside
1266
+ // DeepSeek's Beijing-hours windows, and it adds vision, which Pro (text
1267
+ // only) could not offer discovery. CN is NOT listed: it is dearer than US
1268
+ // here, so free planning stays on the cheaper host.
1269
+ freeTierRegions: ['us'],
1072
1270
  regionPricing: {
1073
- us: { inputPricePerMTok: 0.3, outputPricePerMTok: 1.2, cacheReadPricePerMTok: 0.06 },
1271
+ // Verified 2026-08-14 against api.deepinfra.com/models/MiniMaxAI/MiniMax-M3
1272
+ // (cache read = 0.2 × input). Was 0.3/1.2/0.06 — DeepInfra had repriced
1273
+ // and nothing noticed, because the freshness gate's re-host check only
1274
+ // covered deepseek and moonshot until this date.
1275
+ us: { inputPricePerMTok: 0.28, outputPricePerMTok: 1.1, cacheReadPricePerMTok: 0.056 },
1074
1276
  },
1075
1277
  // From the official HF chat template ("Knowledge cutoff: January 2026").
1076
1278
  knowledgeCutoff: '2026-01-01',
@@ -1101,9 +1303,9 @@ export const MODELS = [
1101
1303
  us: { inputPricePerMTok: 0.25, outputPricePerMTok: 1, cacheReadPricePerMTok: 0.05 },
1102
1304
  },
1103
1305
  knowledgeCutoff: '2025-09-01',
1104
- // Superseded by minimax-m3 (same price, 1M ctx, multimodal); moved to
1105
- // "Older models".
1306
+ // Superseded by minimax-m3 (same price, 1M ctx, multimodal) — kept priceable.
1106
1307
  deprecatedAt: '2026-07-28',
1308
+ supersededBy: 'minimax-m3',
1107
1309
  },
1108
1310
  {
1109
1311
  id: 'minimax-m2.5',
@@ -1126,17 +1328,18 @@ export const MODELS = [
1126
1328
  // No US re-host exists (not on DeepInfra) — pinned to native China.
1127
1329
  regions: ['cn'],
1128
1330
  knowledgeCutoff: '2025-01-01',
1129
- // Superseded by minimax-m3 (legacy upstream, still served); kept selectable
1130
- // (Older models) + priceable.
1331
+ // Superseded by minimax-m3 (legacy upstream, still served) — kept priceable.
1131
1332
  deprecatedAt: '2026-03-18',
1333
+ supersededBy: 'minimax-m3',
1132
1334
  },
1133
1335
  // ---------------------------------------------------------------------------
1134
1336
  // Alibaba (Qwen)
1135
1337
  // Verified: https://www.alibabacloud.com/help/en/model-studio/deep-thinking
1136
1338
  // https://www.alibabacloud.com/help/en/model-studio/qwen-coder
1137
1339
  // https://openrouter.ai/qwen/qwen3.7-max
1138
- // qwen3.7-max (2026-05-21) is the agentic flagship — Alibaba's own Qwen-Coder
1139
- // docs now recommend the general-purpose models over Qwen-Coder. Its thinking
1340
+ // qwen3.8-max (2026-08-03) is the agentic flagship, succeeding qwen3.7-max —
1341
+ // Alibaba's own Qwen-Coder docs now recommend the general-purpose models over
1342
+ // Qwen-Coder. Their thinking
1140
1343
  // uses enable_thinking (default ON for the 3.7 series) + thinking_budget
1141
1344
  // (token cap) — a real budget param, so effort scales the budget. The
1142
1345
  // qwen3-coder models are NON-thinking (previous catalog entry was wrong).
@@ -1176,6 +1379,14 @@ export const MODELS = [
1176
1379
  cacheReadPricePerMTok: 0.4,
1177
1380
  cacheWritePricePerMTok: 2,
1178
1381
  regions: ['us', 'cn'],
1382
+ // US = DeepInfra (Qwen/Qwen3.8-Max), verified 2026-08-14 against
1383
+ // api.deepinfra.com/models/ (cache read = 0.1248 x input). ABSENT until then:
1384
+ // every US turn was metered at Alibaba's native rates while running on
1385
+ // DeepInfra, and the model 404'd outright because the bond's modelMap had
1386
+ // never been updated past qwen3.7-max.
1387
+ regionPricing: {
1388
+ us: { inputPricePerMTok: 1.65, outputPricePerMTok: 4.951, cacheReadPricePerMTok: 0.206 },
1389
+ },
1179
1390
  // Not published by Alibaba — best-effort estimate.
1180
1391
  knowledgeCutoff: '2026-04-01',
1181
1392
  },
@@ -1205,8 +1416,18 @@ export const MODELS = [
1205
1416
  // US default. DeepInfra bills identical rates (no regionPricing needed).
1206
1417
  // Verified 2026-08-01.
1207
1418
  regions: ['us', 'cn'],
1419
+ // US = DeepInfra, verified 2026-08-14 (cache read = 0.2 x input). Superseded,
1420
+ // but still priceable for historical usage, so its region rates must be real.
1421
+ regionPricing: {
1422
+ us: { inputPricePerMTok: 2.5, outputPricePerMTok: 7.5, cacheReadPricePerMTok: 0.5 },
1423
+ },
1208
1424
  // Not published by Alibaba — best-effort estimate.
1209
1425
  knowledgeCutoff: '2026-01-01',
1426
+ // Superseded by qwen3.8-max (GA 2026-08-03): same tier and mechanism, and
1427
+ // CHEAPER at list ($2/$6 vs $2.50/$7.50). Still served upstream (the 50%-off
1428
+ // promo runs on this id), so it stays priceable — it is just not offered.
1429
+ deprecatedAt: '2026-08-03',
1430
+ supersededBy: 'qwen3.8-max',
1210
1431
  },
1211
1432
  {
1212
1433
  id: 'qwen3-coder-plus',
@@ -1236,8 +1457,11 @@ export const MODELS = [
1236
1457
  us: { inputPricePerMTok: 0.3, outputPricePerMTok: 1, cacheReadPricePerMTok: 0.1 },
1237
1458
  },
1238
1459
  knowledgeCutoff: '2025-06-01',
1239
- // Alibaba itself recommends the general-purpose models over Qwen-Coder;
1240
- // qwen3.7-max is the pick — moved to "Older models".
1460
+ // Alibaba itself recommends the general-purpose models over Qwen-Coder, so
1461
+ // this sits in "Older models" — but it is NOT superseded: it is a distinct
1462
+ // coding specialist and Alibaba's cheap tier (US $0.30/$1 vs qwen3.8-max's
1463
+ // $2/$6), with no newer coder id. Superseding it would leave Alibaba with
1464
+ // only the flagship.
1241
1465
  deprecatedAt: '2026-07-28',
1242
1466
  },
1243
1467
  // ---------------------------------------------------------------------------
@@ -1309,7 +1533,9 @@ export const MODELS = [
1309
1533
  us: { inputPricePerMTok: 0.6, outputPricePerMTok: 2.08, cacheReadPricePerMTok: 0.12 },
1310
1534
  },
1311
1535
  knowledgeCutoff: '2025-01-01',
1312
- // Superseded by glm-5.2; moved to "Older models".
1536
+ // Superseded by glm-5.2 (same line, bigger window, reasoning_effort) — kept
1537
+ // priceable.
1313
1538
  deprecatedAt: '2026-07-28',
1539
+ supersededBy: 'glm-5.2',
1314
1540
  },
1315
1541
  ];