@molecule/api-resource-ai-models 1.1.0 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +151 -15
- package/dist/handlers/list.d.ts +8 -1
- package/dist/handlers/list.d.ts.map +1 -1
- package/dist/handlers/list.js +12 -4
- package/dist/lookup.d.ts +68 -8
- package/dist/lookup.d.ts.map +1 -1
- package/dist/lookup.js +125 -15
- package/dist/models.d.ts +13 -4
- package/dist/models.d.ts.map +1 -1
- package/dist/models.js +184 -37
- package/dist/types.d.ts +54 -0
- package/dist/types.d.ts.map +1 -1
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -3,7 +3,7 @@ AUTO-GENERATED — DO NOT EDIT THIS FILE.
|
|
|
3
3
|
Generated by `mlcl sync-docs` from the package's src/index.ts JSDoc + mlcl/registry.json.
|
|
4
4
|
Edits here are overwritten on the next commit (molecule's pre-commit hook regenerates).
|
|
5
5
|
To change this document, edit the module-level JSDoc in src/index.ts.
|
|
6
|
-
Generated: 2026-08-
|
|
6
|
+
Generated: 2026-08-13T22:10:11.926Z
|
|
7
7
|
-->
|
|
8
8
|
|
|
9
9
|
# @molecule/api-resource-ai-models
|
|
@@ -260,6 +260,57 @@ interface ModelDefinition {
|
|
|
260
260
|
windows: { startMinuteUtc: number; endMinuteUtc: number }[]
|
|
261
261
|
multiplier: number
|
|
262
262
|
}
|
|
263
|
+
/**
|
|
264
|
+
* A price change the provider has ANNOUNCED with a dated effective instant,
|
|
265
|
+
* staged ahead of time. Until `effectiveFrom` the model bills at the base
|
|
266
|
+
* rates above; from that instant on, these rates replace them.
|
|
267
|
+
*
|
|
268
|
+
* This exists because the freshness gate can only ever see prices that are
|
|
269
|
+
* ALREADY live: it diffs the catalog against models.dev's *current* rates, so
|
|
270
|
+
* a change announced today and effective in three days is invisible to it
|
|
271
|
+
* until after it lands — and the cron runs every 8h, so the catalog would
|
|
272
|
+
* under-meter for up to a third of a day at whatever the new rate is. Landing
|
|
273
|
+
* the new numbers early is not an option either: that over-bills every turn
|
|
274
|
+
* until the switch (the exact mistake the removed DeepSeek peak windows made
|
|
275
|
+
* for weeks against the free-tier default model). Staging with a timestamp is
|
|
276
|
+
* the only form that is correct on BOTH sides of the instant, and it needs no
|
|
277
|
+
* one awake at the switch.
|
|
278
|
+
*
|
|
279
|
+
* Applies to the BASE rates only — a `regionPricing` entry is a different
|
|
280
|
+
* host's rate card (a US re-host does not reprice because the native provider
|
|
281
|
+
* did) and is never touched by a scheduled change.
|
|
282
|
+
*
|
|
283
|
+
* `peakPricing` here, when declared, replaces the model's peak windows from
|
|
284
|
+
* the same instant; when omitted, the model's existing windows carry through
|
|
285
|
+
* unchanged. To schedule the END of peak pricing, declare an explicit
|
|
286
|
+
* `{ windows: [], multiplier: 1 }`.
|
|
287
|
+
*
|
|
288
|
+
* Resolution is `effectiveBaseRates()` / `effectivePeakPricing()`, and every
|
|
289
|
+
* consumer reaches it through `modelRegionRates()` / `priceMultiplierAt()` /
|
|
290
|
+
* the `withEffectivePricing()` projection the list handler serves — so a
|
|
291
|
+
* scheduled change lands everywhere at once with no follow-up edit. Once the
|
|
292
|
+
* instant has passed, fold the rates into the base fields and delete this
|
|
293
|
+
* (the freshness gate now verifies them against models.dev normally).
|
|
294
|
+
*/
|
|
295
|
+
scheduledPricing?: {
|
|
296
|
+
/** ISO-8601 UTC instant the new rates take effect. */
|
|
297
|
+
effectiveFrom: string
|
|
298
|
+
/** Input price per million *uncached* tokens in USD, from `effectiveFrom`. */
|
|
299
|
+
inputPricePerMTok: number
|
|
300
|
+
/** Output price per million tokens in USD, from `effectiveFrom`. */
|
|
301
|
+
outputPricePerMTok: number
|
|
302
|
+
/** Prompt-cache *read* price per million tokens in USD, from `effectiveFrom`. */
|
|
303
|
+
cacheReadPricePerMTok: number
|
|
304
|
+
/** Prompt-cache *write* price per million tokens in USD, from `effectiveFrom`. */
|
|
305
|
+
cacheWritePricePerMTok: number
|
|
306
|
+
/** Peak-hour pricing from `effectiveFrom` (omitted → existing windows carry through). */
|
|
307
|
+
peakPricing?: {
|
|
308
|
+
windows: { startMinuteUtc: number; endMinuteUtc: number }[]
|
|
309
|
+
multiplier: number
|
|
310
|
+
}
|
|
311
|
+
/** Where the change was announced, for the re-verify pass after it lands. */
|
|
312
|
+
source?: string
|
|
313
|
+
}
|
|
263
314
|
/**
|
|
264
315
|
* Fast-mode ("priority speed") pricing — the per-MTok rates billed when a
|
|
265
316
|
* request runs with the provider's fast/priority tier (e.g. Anthropic's
|
|
@@ -431,6 +482,25 @@ type EffortLevel = string
|
|
|
431
482
|
|
|
432
483
|
### Functions
|
|
433
484
|
|
|
485
|
+
#### `effectiveBaseRates(modelDef, at)`
|
|
486
|
+
|
|
487
|
+
A model's BASE token rates in effect at a given instant — the staged
|
|
488
|
+
{@link ModelDefinition.scheduledPricing} rates once their `effectiveFrom` has
|
|
489
|
+
passed, else the base fields.
|
|
490
|
+
|
|
491
|
+
These are the native provider's rates. A `regionPricing` override is a
|
|
492
|
+
different host's rate card and is resolved separately by
|
|
493
|
+
{@link modelRegionRates}.
|
|
494
|
+
|
|
495
|
+
```typescript
|
|
496
|
+
function effectiveBaseRates(modelDef: ModelDefinition, at?: Date): ModelTokenRates
|
|
497
|
+
```
|
|
498
|
+
|
|
499
|
+
- `modelDef` — The model definition.
|
|
500
|
+
- `at` — The instant to price at (defaults to now).
|
|
501
|
+
|
|
502
|
+
**Returns:** The base rates in effect at that instant.
|
|
503
|
+
|
|
434
504
|
#### `effectiveModelRegion(modelDef, requested)`
|
|
435
505
|
|
|
436
506
|
Resolve a model's effective processing region: the requested region when the
|
|
@@ -448,6 +518,25 @@ function effectiveModelRegion(modelDef: ModelDefinition | undefined, requested?:
|
|
|
448
518
|
|
|
449
519
|
**Returns:** The effective region code.
|
|
450
520
|
|
|
521
|
+
#### `effectivePeakPricing(modelDef, at)`
|
|
522
|
+
|
|
523
|
+
A model's peak-hour pricing in effect at a given instant: the staged
|
|
524
|
+
{@link ModelDefinition.scheduledPricing} `peakPricing` once its
|
|
525
|
+
`effectiveFrom` has passed (when that entry declares one — an omitted one
|
|
526
|
+
leaves the existing windows in force), else the model's own `peakPricing`.
|
|
527
|
+
|
|
528
|
+
```typescript
|
|
529
|
+
function effectivePeakPricing(
|
|
530
|
+
modelDef: ModelDefinition,
|
|
531
|
+
at?: Date,
|
|
532
|
+
): { windows: { startMinuteUtc: number; endMinuteUtc: number }[]; multiplier: number } | undefined
|
|
533
|
+
```
|
|
534
|
+
|
|
535
|
+
- `modelDef` — The model definition.
|
|
536
|
+
- `at` — The instant to evaluate (defaults to now).
|
|
537
|
+
|
|
538
|
+
**Returns:** The peak-pricing config in effect, or `undefined` when none is.
|
|
539
|
+
|
|
451
540
|
#### `getAvailableModels(availableProviders)`
|
|
452
541
|
|
|
453
542
|
Get models that are currently usable — filtered to only providers that are available.
|
|
@@ -527,41 +616,58 @@ function list(_req: MoleculeRequest, res: MoleculeResponse): Promise<void>
|
|
|
527
616
|
- `_req` — The request object (unused).
|
|
528
617
|
- `res` — The response object.
|
|
529
618
|
|
|
530
|
-
#### `modelRegionRates(modelDef, requested)`
|
|
619
|
+
#### `modelRegionRates(modelDef, requested, at)`
|
|
531
620
|
|
|
532
621
|
The token rates for a model in a given processing region: the model's
|
|
533
622
|
{@link ModelDefinition.regionPricing} override for the region when one
|
|
534
|
-
exists, else the base rates (the native provider's list
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
|
|
623
|
+
exists, else the base rates in effect at `at` (the native provider's list
|
|
624
|
+
prices, including any staged {@link ModelDefinition.scheduledPricing} change
|
|
625
|
+
that has landed). Omitted cache fields in an override fall back to the
|
|
626
|
+
override's input price (hosts with no cache discount / no write premium). The
|
|
627
|
+
region is resolved via {@link effectiveModelRegion}, so callers may pass the
|
|
628
|
+
raw user choice.
|
|
629
|
+
|
|
630
|
+
`at` defaults to NOW rather than being required, so an existing caller cannot
|
|
631
|
+
keep billing a superseded rate by omitting it — metering should still pass
|
|
632
|
+
each request's own timestamp, the same way it must for
|
|
633
|
+
{@link priceMultiplierAt}.
|
|
538
634
|
|
|
539
635
|
```typescript
|
|
540
|
-
function modelRegionRates(modelDef: ModelDefinition, requested?: string): ModelTokenRates
|
|
636
|
+
function modelRegionRates(modelDef: ModelDefinition, requested?: string, at?: Date): ModelTokenRates
|
|
541
637
|
```
|
|
542
638
|
|
|
543
639
|
- `modelDef` — The model definition.
|
|
544
640
|
- `requested` — The user's per-model region choice, if any.
|
|
641
|
+
- `at` — The instant to price at (defaults to now).
|
|
545
642
|
|
|
546
643
|
**Returns:** The region-effective rates.
|
|
547
644
|
|
|
548
|
-
#### `priceMultiplierAt(modelDef, at)`
|
|
645
|
+
#### `priceMultiplierAt(modelDef, at, region)`
|
|
549
646
|
|
|
550
|
-
The price multiplier in effect for a model at a given instant.
|
|
647
|
+
The price multiplier in effect for a model at a given instant, in a region.
|
|
551
648
|
|
|
552
649
|
Consults the model's {@link ModelDefinition.peakPricing} windows (UTC,
|
|
553
650
|
half-open, may wrap midnight). Metering MUST call this with each request's
|
|
554
651
|
own timestamp so peak-hour usage bills at the provider's real rate — pricing
|
|
555
652
|
everything at the flat rate silently under-meters peak traffic.
|
|
556
653
|
|
|
654
|
+
Peak windows belong to the NATIVE provider, so they apply only where the base
|
|
655
|
+
rates do. A region with a {@link ModelDefinition.regionPricing} override is a
|
|
656
|
+
different host billing its own complete rate card, including whether it has
|
|
657
|
+
time-of-day pricing at all — and re-hosts generally do not. Applying the
|
|
658
|
+
native provider's surcharge on top of a re-host's flat rates would over-bill
|
|
659
|
+
every turn in its windows (DeepSeek's 2× Beijing-hours pricing charged
|
|
660
|
+
against DeepInfra, which has no peak pricing).
|
|
661
|
+
|
|
557
662
|
```typescript
|
|
558
|
-
function priceMultiplierAt(modelDef: ModelDefinition | undefined, at: Date): number
|
|
663
|
+
function priceMultiplierAt(modelDef: ModelDefinition | undefined, at: Date, region?: string): number
|
|
559
664
|
```
|
|
560
665
|
|
|
561
666
|
- `modelDef` — The model definition (or undefined).
|
|
562
667
|
- `at` — The instant the request was made.
|
|
668
|
+
- `region` — The user's per-model region choice, if any (omitted → the model's default region).
|
|
563
669
|
|
|
564
|
-
**Returns:** The multiplier (`1` outside peak windows
|
|
670
|
+
**Returns:** The multiplier (`1` outside peak windows, when none are declared, or in a region that prices off its own override).
|
|
565
671
|
|
|
566
672
|
#### `resolveSelectableModelId(id)`
|
|
567
673
|
|
|
@@ -579,6 +685,27 @@ function resolveSelectableModelId(id: string): string | undefined
|
|
|
579
685
|
|
|
580
686
|
**Returns:** The selectable successor's id, the id itself when it is already selectable, or `undefined` for an unknown or `disabled` model (nothing to forward to).
|
|
581
687
|
|
|
688
|
+
#### `withEffectivePricing(modelDef, at)`
|
|
689
|
+
|
|
690
|
+
A model projected onto the pricing in effect at a given instant: the staged
|
|
691
|
+
{@link ModelDefinition.scheduledPricing} rates folded into the base fields
|
|
692
|
+
(and its peak windows into `peakPricing`) once effective, with the staged
|
|
693
|
+
entry stripped.
|
|
694
|
+
|
|
695
|
+
This is what the `GET /ai/models` handler serves, so a client renders the
|
|
696
|
+
rates that are actually billing right now without needing to resolve a
|
|
697
|
+
schedule against its own clock — the server's clock is the only one that
|
|
698
|
+
decides when a price change lands.
|
|
699
|
+
|
|
700
|
+
```typescript
|
|
701
|
+
function withEffectivePricing(modelDef: ModelDefinition, at?: Date): ModelDefinition
|
|
702
|
+
```
|
|
703
|
+
|
|
704
|
+
- `modelDef` — The model definition.
|
|
705
|
+
- `at` — The instant to project at (defaults to now).
|
|
706
|
+
|
|
707
|
+
**Returns:** The model with effective pricing and no `scheduledPricing`.
|
|
708
|
+
|
|
582
709
|
### Constants
|
|
583
710
|
|
|
584
711
|
#### `MODEL_IDS`
|
|
@@ -650,15 +777,24 @@ Sources (verified 2026-07-28; OpenAI re-verified 2026-07-31 after the
|
|
|
650
777
|
2026-07-21 $1.50/$7.50 supersedes 3.5-flash as the agentic flagship;
|
|
651
778
|
gemini-3.1-pro-preview still the pro tier — "3.5 Pro" has NOT shipped as
|
|
652
779
|
of 2026-07-28 despite the coming-soon badge; do not add until it has an id)
|
|
780
|
+
(re-verified 2026-08-13: gemini-3.7-flash "New Stable" — supersedes
|
|
781
|
+
3.6-flash as the flash flagship at the SAME list price ($1.50/$7.50, cache
|
|
782
|
+
read $0.15), with a launch promo ($0.75/$3.75, cache read $0.075) through
|
|
783
|
+
2026-12-31 billed here at list; specs from /docs/models/gemini-3.7-flash:
|
|
784
|
+
1M ctx / 65,536 out, thinking low|medium|high (no minimal), vision, tools,
|
|
785
|
+
caching, search grounding, code execution, url context)
|
|
653
786
|
- xAI: https://docs.x.ai/developers/models + /developers/grok-4-5
|
|
654
787
|
(grok-4.5 flagship 2026-07-08: $2/$6, 500K ctx, ≥200K prompts bill 2× —
|
|
655
788
|
not modeled; reasoning_effort low|medium|high default high, image input;
|
|
656
789
|
grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
|
|
657
790
|
grok-code-fast-1 no longer listed — retires 2026-08-15)
|
|
658
|
-
- DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
|
|
791
|
+
- DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
|
|
792
|
+
2026-08-14; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
|
|
793
|
+
never in this catalog. V4-Pro GA on 2026-08-13 came with a price RISE
|
|
794
|
+
effective 2026-08-16T16:00Z plus the long-announced peak-hour 2×: both
|
|
795
|
+
entries carry it as `scheduledPricing`, so today's rates bill until that
|
|
796
|
+
instant and the new ones after. Re-verify weekday-vs-daily peak windows and
|
|
797
|
+
the CN/US region default once it lands — see the entries.)
|
|
662
798
|
- Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
|
|
663
799
|
the US re-host (kimi-k3 flagship 2026-07-16
|
|
664
800
|
— 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
|
package/dist/handlers/list.d.ts
CHANGED
|
@@ -5,7 +5,14 @@
|
|
|
5
5
|
* bonded under the `'ai'` category AND are selectable — neither `disabled` (a
|
|
6
6
|
* model the provider retired) nor superseded by a newer generation of the same
|
|
7
7
|
* family, so the picker offers exactly one generation per family. Both kinds
|
|
8
|
-
* stay priceable via `getModel`.
|
|
8
|
+
* stay priceable via `getModel`.
|
|
9
|
+
*
|
|
10
|
+
* The one projection applied is `withEffectivePricing`: a model carrying a
|
|
11
|
+
* staged `scheduledPricing` change is served at whichever rates are billing at
|
|
12
|
+
* request time, with the schedule stripped. The server's clock decides when an
|
|
13
|
+
* announced price change lands, so a client never renders a rate that is not
|
|
14
|
+
* yet in force (nor keeps rendering one that has been superseded), and no
|
|
15
|
+
* client needs to know that scheduled pricing exists. Every other
|
|
9
16
|
* `ModelDefinition` field is fine to expose to authenticated clients today.
|
|
10
17
|
*
|
|
11
18
|
* Secure-by-default: this handler enforces authentication IN the handler
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"list.d.ts","sourceRoot":"","sources":["../../src/handlers/list.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"list.d.ts","sourceRoot":"","sources":["../../src/handlers/list.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;;;;;;GA0BG;AAIH,OAAO,KAAK,EAAE,eAAe,EAAE,gBAAgB,EAAE,MAAM,wBAAwB,CAAA;AAM/E;;;;;;;;;;;GAWG;AACH,wBAAsB,IAAI,CAAC,IAAI,EAAE,eAAe,EAAE,GAAG,EAAE,gBAAgB,GAAG,OAAO,CAAC,IAAI,CAAC,CAiBtF"}
|
package/dist/handlers/list.js
CHANGED
|
@@ -5,7 +5,14 @@
|
|
|
5
5
|
* bonded under the `'ai'` category AND are selectable — neither `disabled` (a
|
|
6
6
|
* model the provider retired) nor superseded by a newer generation of the same
|
|
7
7
|
* family, so the picker offers exactly one generation per family. Both kinds
|
|
8
|
-
* stay priceable via `getModel`.
|
|
8
|
+
* stay priceable via `getModel`.
|
|
9
|
+
*
|
|
10
|
+
* The one projection applied is `withEffectivePricing`: a model carrying a
|
|
11
|
+
* staged `scheduledPricing` change is served at whichever rates are billing at
|
|
12
|
+
* request time, with the schedule stripped. The server's clock decides when an
|
|
13
|
+
* announced price change lands, so a client never renders a rate that is not
|
|
14
|
+
* yet in force (nor keeps rendering one that has been superseded), and no
|
|
15
|
+
* client needs to know that scheduled pricing exists. Every other
|
|
9
16
|
* `ModelDefinition` field is fine to expose to authenticated clients today.
|
|
10
17
|
*
|
|
11
18
|
* Secure-by-default: this handler enforces authentication IN the handler
|
|
@@ -20,7 +27,7 @@
|
|
|
20
27
|
*/
|
|
21
28
|
import { getAll } from '@molecule/api-bond';
|
|
22
29
|
import { t } from '@molecule/api-i18n';
|
|
23
|
-
import { isSelectableModel } from '../lookup.js';
|
|
30
|
+
import { isSelectableModel, withEffectivePricing } from '../lookup.js';
|
|
24
31
|
import { MODELS } from '../models.js';
|
|
25
32
|
/**
|
|
26
33
|
* Returns models whose `provider` has a bond registered under the `'ai'`
|
|
@@ -44,7 +51,8 @@ export async function list(_req, res) {
|
|
|
44
51
|
return;
|
|
45
52
|
}
|
|
46
53
|
const bondedProviders = new Set(getAll('ai').keys());
|
|
47
|
-
const
|
|
48
|
-
const
|
|
54
|
+
const now = new Date();
|
|
55
|
+
const models = MODELS.filter((m) => bondedProviders.has(m.provider) && isSelectableModel(m)).map((m) => withEffectivePricing(m, now));
|
|
56
|
+
const response = { models };
|
|
49
57
|
res.json(response);
|
|
50
58
|
}
|
package/dist/lookup.d.ts
CHANGED
|
@@ -67,18 +67,70 @@ export declare function getModelsByProvider(provider: AIProviderID): readonly Mo
|
|
|
67
67
|
*/
|
|
68
68
|
export declare function getAvailableModels(availableProviders: ReadonlySet<AIProviderID> | readonly AIProviderID[]): readonly ModelDefinition[];
|
|
69
69
|
/**
|
|
70
|
-
*
|
|
70
|
+
* A model's BASE token rates in effect at a given instant — the staged
|
|
71
|
+
* {@link ModelDefinition.scheduledPricing} rates once their `effectiveFrom` has
|
|
72
|
+
* passed, else the base fields.
|
|
73
|
+
*
|
|
74
|
+
* These are the native provider's rates. A `regionPricing` override is a
|
|
75
|
+
* different host's rate card and is resolved separately by
|
|
76
|
+
* {@link modelRegionRates}.
|
|
77
|
+
*
|
|
78
|
+
* @param modelDef - The model definition.
|
|
79
|
+
* @param at - The instant to price at (defaults to now).
|
|
80
|
+
* @returns The base rates in effect at that instant.
|
|
81
|
+
*/
|
|
82
|
+
export declare function effectiveBaseRates(modelDef: ModelDefinition, at?: Date): ModelTokenRates;
|
|
83
|
+
/**
|
|
84
|
+
* A model's peak-hour pricing in effect at a given instant: the staged
|
|
85
|
+
* {@link ModelDefinition.scheduledPricing} `peakPricing` once its
|
|
86
|
+
* `effectiveFrom` has passed (when that entry declares one — an omitted one
|
|
87
|
+
* leaves the existing windows in force), else the model's own `peakPricing`.
|
|
88
|
+
*
|
|
89
|
+
* @param modelDef - The model definition.
|
|
90
|
+
* @param at - The instant to evaluate (defaults to now).
|
|
91
|
+
* @returns The peak-pricing config in effect, or `undefined` when none is.
|
|
92
|
+
*/
|
|
93
|
+
export declare function effectivePeakPricing(modelDef: ModelDefinition, at?: Date): ModelDefinition['peakPricing'];
|
|
94
|
+
/**
|
|
95
|
+
* A model projected onto the pricing in effect at a given instant: the staged
|
|
96
|
+
* {@link ModelDefinition.scheduledPricing} rates folded into the base fields
|
|
97
|
+
* (and its peak windows into `peakPricing`) once effective, with the staged
|
|
98
|
+
* entry stripped.
|
|
99
|
+
*
|
|
100
|
+
* This is what the `GET /ai/models` handler serves, so a client renders the
|
|
101
|
+
* rates that are actually billing right now without needing to resolve a
|
|
102
|
+
* schedule against its own clock — the server's clock is the only one that
|
|
103
|
+
* decides when a price change lands.
|
|
104
|
+
*
|
|
105
|
+
* @param modelDef - The model definition.
|
|
106
|
+
* @param at - The instant to project at (defaults to now).
|
|
107
|
+
* @returns The model with effective pricing and no `scheduledPricing`.
|
|
108
|
+
*/
|
|
109
|
+
export declare function withEffectivePricing(modelDef: ModelDefinition, at?: Date): ModelDefinition;
|
|
110
|
+
/**
|
|
111
|
+
* The price multiplier in effect for a model at a given instant, in a region.
|
|
71
112
|
*
|
|
72
113
|
* Consults the model's {@link ModelDefinition.peakPricing} windows (UTC,
|
|
73
114
|
* half-open, may wrap midnight). Metering MUST call this with each request's
|
|
74
115
|
* own timestamp so peak-hour usage bills at the provider's real rate — pricing
|
|
75
116
|
* everything at the flat rate silently under-meters peak traffic.
|
|
76
117
|
*
|
|
118
|
+
* Peak windows belong to the NATIVE provider, so they apply only where the base
|
|
119
|
+
* rates do. A region with a {@link ModelDefinition.regionPricing} override is a
|
|
120
|
+
* different host billing its own complete rate card, including whether it has
|
|
121
|
+
* time-of-day pricing at all — and re-hosts generally do not. Applying the
|
|
122
|
+
* native provider's surcharge on top of a re-host's flat rates would over-bill
|
|
123
|
+
* every turn in its windows (DeepSeek's 2× Beijing-hours pricing charged
|
|
124
|
+
* against DeepInfra, which has no peak pricing).
|
|
125
|
+
*
|
|
77
126
|
* @param modelDef - The model definition (or undefined).
|
|
78
127
|
* @param at - The instant the request was made.
|
|
79
|
-
* @
|
|
128
|
+
* @param region - The user's per-model region choice, if any (omitted → the
|
|
129
|
+
* model's default region).
|
|
130
|
+
* @returns The multiplier (`1` outside peak windows, when none are declared, or
|
|
131
|
+
* in a region that prices off its own override).
|
|
80
132
|
*/
|
|
81
|
-
export declare function priceMultiplierAt(modelDef: ModelDefinition | undefined, at: Date): number;
|
|
133
|
+
export declare function priceMultiplierAt(modelDef: ModelDefinition | undefined, at: Date, region?: string): number;
|
|
82
134
|
/**
|
|
83
135
|
* Resolve a model's effective processing region: the requested region when the
|
|
84
136
|
* model's {@link ModelDefinition.regions} list offers it, else the model's
|
|
@@ -105,14 +157,22 @@ export interface ModelTokenRates {
|
|
|
105
157
|
/**
|
|
106
158
|
* The token rates for a model in a given processing region: the model's
|
|
107
159
|
* {@link ModelDefinition.regionPricing} override for the region when one
|
|
108
|
-
* exists, else the base rates (the native provider's list
|
|
109
|
-
*
|
|
110
|
-
*
|
|
111
|
-
*
|
|
160
|
+
* exists, else the base rates in effect at `at` (the native provider's list
|
|
161
|
+
* prices, including any staged {@link ModelDefinition.scheduledPricing} change
|
|
162
|
+
* that has landed). Omitted cache fields in an override fall back to the
|
|
163
|
+
* override's input price (hosts with no cache discount / no write premium). The
|
|
164
|
+
* region is resolved via {@link effectiveModelRegion}, so callers may pass the
|
|
165
|
+
* raw user choice.
|
|
166
|
+
*
|
|
167
|
+
* `at` defaults to NOW rather than being required, so an existing caller cannot
|
|
168
|
+
* keep billing a superseded rate by omitting it — metering should still pass
|
|
169
|
+
* each request's own timestamp, the same way it must for
|
|
170
|
+
* {@link priceMultiplierAt}.
|
|
112
171
|
*
|
|
113
172
|
* @param modelDef - The model definition.
|
|
114
173
|
* @param requested - The user's per-model region choice, if any.
|
|
174
|
+
* @param at - The instant to price at (defaults to now).
|
|
115
175
|
* @returns The region-effective rates.
|
|
116
176
|
*/
|
|
117
|
-
export declare function modelRegionRates(modelDef: ModelDefinition, requested?: string): ModelTokenRates;
|
|
177
|
+
export declare function modelRegionRates(modelDef: ModelDefinition, requested?: string, at?: Date): ModelTokenRates;
|
|
118
178
|
//# sourceMappingURL=lookup.d.ts.map
|
package/dist/lookup.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"lookup.d.ts","sourceRoot":"","sources":["../src/lookup.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAGH,OAAO,KAAK,EAAE,YAAY,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAE/D;;;;;;;;GAQG;AACH,wBAAgB,iBAAiB,CAC/B,KAAK,EAAE,IAAI,CAAC,eAAe,EAAE,UAAU,GAAG,cAAc,CAAC,GACxD,OAAO,CAET;AAED;;;;;;;;;;;GAWG;AACH,wBAAgB,wBAAwB,CAAC,EAAE,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CASvE;AAED;;;;;;;GAOG;AACH,eAAO,MAAM,SAAS,EAAE,WAAW,CAAC,MAAM,CAEzC,CAAA;AAED;;;;;;;;;;GAUG;AACH,wBAAgB,QAAQ,CAAC,EAAE,EAAE,MAAM,GAAG,eAAe,GAAG,SAAS,CAEhE;AAED;;;;;GAKG;AACH,wBAAgB,mBAAmB,CAAC,QAAQ,EAAE,YAAY,GAAG,SAAS,eAAe,EAAE,CAEtF;AAED;;;;;;;;;GASG;AACH,wBAAgB,kBAAkB,CAChC,kBAAkB,EAAE,WAAW,CAAC,YAAY,CAAC,GAAG,SAAS,YAAY,EAAE,GACtE,SAAS,eAAe,EAAE,CAI5B;AAED
|
|
1
|
+
{"version":3,"file":"lookup.d.ts","sourceRoot":"","sources":["../src/lookup.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAGH,OAAO,KAAK,EAAE,YAAY,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAE/D;;;;;;;;GAQG;AACH,wBAAgB,iBAAiB,CAC/B,KAAK,EAAE,IAAI,CAAC,eAAe,EAAE,UAAU,GAAG,cAAc,CAAC,GACxD,OAAO,CAET;AAED;;;;;;;;;;;GAWG;AACH,wBAAgB,wBAAwB,CAAC,EAAE,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CASvE;AAED;;;;;;;GAOG;AACH,eAAO,MAAM,SAAS,EAAE,WAAW,CAAC,MAAM,CAEzC,CAAA;AAED;;;;;;;;;;GAUG;AACH,wBAAgB,QAAQ,CAAC,EAAE,EAAE,MAAM,GAAG,eAAe,GAAG,SAAS,CAEhE;AAED;;;;;GAKG;AACH,wBAAgB,mBAAmB,CAAC,QAAQ,EAAE,YAAY,GAAG,SAAS,eAAe,EAAE,CAEtF;AAED;;;;;;;;;GASG;AACH,wBAAgB,kBAAkB,CAChC,kBAAkB,EAAE,WAAW,CAAC,YAAY,CAAC,GAAG,SAAS,YAAY,EAAE,GACtE,SAAS,eAAe,EAAE,CAI5B;AAoBD;;;;;;;;;;;;GAYG;AACH,wBAAgB,kBAAkB,CAChC,QAAQ,EAAE,eAAe,EACzB,EAAE,GAAE,IAAiB,GACpB,eAAe,CAgBjB;AAED;;;;;;;;;GASG;AACH,wBAAgB,oBAAoB,CAClC,QAAQ,EAAE,eAAe,EACzB,EAAE,GAAE,IAAiB,GACpB,eAAe,CAAC,aAAa,CAAC,CAMhC;AAED;;;;;;;;;;;;;;GAcG;AACH,wBAAgB,oBAAoB,CAClC,QAAQ,EAAE,eAAe,EACzB,EAAE,GAAE,IAAiB,GACpB,eAAe,CASjB;AAED;;;;;;;;;;;;;;;;;;;;;;GAsBG;AACH,wBAAgB,iBAAiB,CAC/B,QAAQ,EAAE,eAAe,GAAG,SAAS,EACrC,EAAE,EAAE,IAAI,EACR,MAAM,CAAC,EAAE,MAAM,GACd,MAAM,CAcR;AAED;;;;;;;;;;GAUG;AACH,wBAAgB,oBAAoB,CAClC,QAAQ,EAAE,eAAe,GAAG,SAAS,EACrC,SAAS,CAAC,EAAE,MAAM,GACjB,MAAM,CAGR;AAED,2DAA2D;AAC3D,MAAM,WAAW,eAAe;IAC9B,sDAAsD;IACtD,iBAAiB,EAAE,MAAM,CAAA;IACzB,8CAA8C;IAC9C,kBAAkB,EAAE,MAAM,CAAA;IAC1B,yDAAyD;IACzD,qBAAqB,EAAE,MAAM,CAAA;IAC7B,0DAA0D;IAC1D,sBAAsB,EAAE,MAAM,CAAA;CAC/B;AAED;;;;;;;;;;;;;;;;;;;GAmBG;AACH,wBAAgB,gBAAgB,CAC9B,QAAQ,EAAE,eAAe,EACzB,SAAS,CAAC,EAAE,MAAM,EAClB,EAAE,GAAE,IAAiB,GACpB,eAAe,CAYjB"}
|
package/dist/lookup.js
CHANGED
|
@@ -86,19 +86,126 @@ export function getAvailableModels(availableProviders) {
|
|
|
86
86
|
return MODELS.filter((m) => providerSet.has(m.provider) && isSelectableModel(m));
|
|
87
87
|
}
|
|
88
88
|
/**
|
|
89
|
-
*
|
|
89
|
+
* Whether a model's staged {@link ModelDefinition.scheduledPricing} change has
|
|
90
|
+
* taken effect at a given instant.
|
|
91
|
+
*
|
|
92
|
+
* @param modelDef - The model definition.
|
|
93
|
+
* @param at - The instant to evaluate.
|
|
94
|
+
* @returns `true` once `at` is at or past the scheduled `effectiveFrom`.
|
|
95
|
+
*/
|
|
96
|
+
function scheduledPricingApplies(modelDef, at) {
|
|
97
|
+
const scheduled = modelDef.scheduledPricing;
|
|
98
|
+
if (!scheduled)
|
|
99
|
+
return false;
|
|
100
|
+
const effectiveFrom = Date.parse(scheduled.effectiveFrom);
|
|
101
|
+
// An unparseable date must never silently reprice a model. Ignoring the
|
|
102
|
+
// staged entry keeps the current, verified rates in force.
|
|
103
|
+
if (Number.isNaN(effectiveFrom))
|
|
104
|
+
return false;
|
|
105
|
+
return at.getTime() >= effectiveFrom;
|
|
106
|
+
}
|
|
107
|
+
/**
|
|
108
|
+
* A model's BASE token rates in effect at a given instant — the staged
|
|
109
|
+
* {@link ModelDefinition.scheduledPricing} rates once their `effectiveFrom` has
|
|
110
|
+
* passed, else the base fields.
|
|
111
|
+
*
|
|
112
|
+
* These are the native provider's rates. A `regionPricing` override is a
|
|
113
|
+
* different host's rate card and is resolved separately by
|
|
114
|
+
* {@link modelRegionRates}.
|
|
115
|
+
*
|
|
116
|
+
* @param modelDef - The model definition.
|
|
117
|
+
* @param at - The instant to price at (defaults to now).
|
|
118
|
+
* @returns The base rates in effect at that instant.
|
|
119
|
+
*/
|
|
120
|
+
export function effectiveBaseRates(modelDef, at = new Date()) {
|
|
121
|
+
const scheduled = modelDef.scheduledPricing;
|
|
122
|
+
if (scheduled && scheduledPricingApplies(modelDef, at)) {
|
|
123
|
+
return {
|
|
124
|
+
inputPricePerMTok: scheduled.inputPricePerMTok,
|
|
125
|
+
outputPricePerMTok: scheduled.outputPricePerMTok,
|
|
126
|
+
cacheReadPricePerMTok: scheduled.cacheReadPricePerMTok,
|
|
127
|
+
cacheWritePricePerMTok: scheduled.cacheWritePricePerMTok,
|
|
128
|
+
};
|
|
129
|
+
}
|
|
130
|
+
return {
|
|
131
|
+
inputPricePerMTok: modelDef.inputPricePerMTok,
|
|
132
|
+
outputPricePerMTok: modelDef.outputPricePerMTok,
|
|
133
|
+
cacheReadPricePerMTok: modelDef.cacheReadPricePerMTok,
|
|
134
|
+
cacheWritePricePerMTok: modelDef.cacheWritePricePerMTok,
|
|
135
|
+
};
|
|
136
|
+
}
|
|
137
|
+
/**
|
|
138
|
+
* A model's peak-hour pricing in effect at a given instant: the staged
|
|
139
|
+
* {@link ModelDefinition.scheduledPricing} `peakPricing` once its
|
|
140
|
+
* `effectiveFrom` has passed (when that entry declares one — an omitted one
|
|
141
|
+
* leaves the existing windows in force), else the model's own `peakPricing`.
|
|
142
|
+
*
|
|
143
|
+
* @param modelDef - The model definition.
|
|
144
|
+
* @param at - The instant to evaluate (defaults to now).
|
|
145
|
+
* @returns The peak-pricing config in effect, or `undefined` when none is.
|
|
146
|
+
*/
|
|
147
|
+
export function effectivePeakPricing(modelDef, at = new Date()) {
|
|
148
|
+
const scheduled = modelDef.scheduledPricing;
|
|
149
|
+
if (scheduled?.peakPricing && scheduledPricingApplies(modelDef, at)) {
|
|
150
|
+
return scheduled.peakPricing;
|
|
151
|
+
}
|
|
152
|
+
return modelDef.peakPricing;
|
|
153
|
+
}
|
|
154
|
+
/**
|
|
155
|
+
* A model projected onto the pricing in effect at a given instant: the staged
|
|
156
|
+
* {@link ModelDefinition.scheduledPricing} rates folded into the base fields
|
|
157
|
+
* (and its peak windows into `peakPricing`) once effective, with the staged
|
|
158
|
+
* entry stripped.
|
|
159
|
+
*
|
|
160
|
+
* This is what the `GET /ai/models` handler serves, so a client renders the
|
|
161
|
+
* rates that are actually billing right now without needing to resolve a
|
|
162
|
+
* schedule against its own clock — the server's clock is the only one that
|
|
163
|
+
* decides when a price change lands.
|
|
164
|
+
*
|
|
165
|
+
* @param modelDef - The model definition.
|
|
166
|
+
* @param at - The instant to project at (defaults to now).
|
|
167
|
+
* @returns The model with effective pricing and no `scheduledPricing`.
|
|
168
|
+
*/
|
|
169
|
+
export function withEffectivePricing(modelDef, at = new Date()) {
|
|
170
|
+
if (!modelDef.scheduledPricing)
|
|
171
|
+
return modelDef;
|
|
172
|
+
const { scheduledPricing: _scheduledPricing, ...rest } = modelDef;
|
|
173
|
+
const peakPricing = effectivePeakPricing(modelDef, at);
|
|
174
|
+
return {
|
|
175
|
+
...rest,
|
|
176
|
+
...effectiveBaseRates(modelDef, at),
|
|
177
|
+
...(peakPricing ? { peakPricing } : {}),
|
|
178
|
+
};
|
|
179
|
+
}
|
|
180
|
+
/**
|
|
181
|
+
* The price multiplier in effect for a model at a given instant, in a region.
|
|
90
182
|
*
|
|
91
183
|
* Consults the model's {@link ModelDefinition.peakPricing} windows (UTC,
|
|
92
184
|
* half-open, may wrap midnight). Metering MUST call this with each request's
|
|
93
185
|
* own timestamp so peak-hour usage bills at the provider's real rate — pricing
|
|
94
186
|
* everything at the flat rate silently under-meters peak traffic.
|
|
95
187
|
*
|
|
188
|
+
* Peak windows belong to the NATIVE provider, so they apply only where the base
|
|
189
|
+
* rates do. A region with a {@link ModelDefinition.regionPricing} override is a
|
|
190
|
+
* different host billing its own complete rate card, including whether it has
|
|
191
|
+
* time-of-day pricing at all — and re-hosts generally do not. Applying the
|
|
192
|
+
* native provider's surcharge on top of a re-host's flat rates would over-bill
|
|
193
|
+
* every turn in its windows (DeepSeek's 2× Beijing-hours pricing charged
|
|
194
|
+
* against DeepInfra, which has no peak pricing).
|
|
195
|
+
*
|
|
96
196
|
* @param modelDef - The model definition (or undefined).
|
|
97
197
|
* @param at - The instant the request was made.
|
|
98
|
-
* @
|
|
198
|
+
* @param region - The user's per-model region choice, if any (omitted → the
|
|
199
|
+
* model's default region).
|
|
200
|
+
* @returns The multiplier (`1` outside peak windows, when none are declared, or
|
|
201
|
+
* in a region that prices off its own override).
|
|
99
202
|
*/
|
|
100
|
-
export function priceMultiplierAt(modelDef, at) {
|
|
101
|
-
|
|
203
|
+
export function priceMultiplierAt(modelDef, at, region) {
|
|
204
|
+
if (!modelDef)
|
|
205
|
+
return 1;
|
|
206
|
+
if (modelDef.regionPricing?.[effectiveModelRegion(modelDef, region)])
|
|
207
|
+
return 1;
|
|
208
|
+
const peak = effectivePeakPricing(modelDef, at);
|
|
102
209
|
if (!peak || peak.windows.length === 0)
|
|
103
210
|
return 1;
|
|
104
211
|
const minute = at.getUTCHours() * 60 + at.getUTCMinutes();
|
|
@@ -129,25 +236,28 @@ export function effectiveModelRegion(modelDef, requested) {
|
|
|
129
236
|
/**
|
|
130
237
|
* The token rates for a model in a given processing region: the model's
|
|
131
238
|
* {@link ModelDefinition.regionPricing} override for the region when one
|
|
132
|
-
* exists, else the base rates (the native provider's list
|
|
133
|
-
*
|
|
134
|
-
*
|
|
135
|
-
*
|
|
239
|
+
* exists, else the base rates in effect at `at` (the native provider's list
|
|
240
|
+
* prices, including any staged {@link ModelDefinition.scheduledPricing} change
|
|
241
|
+
* that has landed). Omitted cache fields in an override fall back to the
|
|
242
|
+
* override's input price (hosts with no cache discount / no write premium). The
|
|
243
|
+
* region is resolved via {@link effectiveModelRegion}, so callers may pass the
|
|
244
|
+
* raw user choice.
|
|
245
|
+
*
|
|
246
|
+
* `at` defaults to NOW rather than being required, so an existing caller cannot
|
|
247
|
+
* keep billing a superseded rate by omitting it — metering should still pass
|
|
248
|
+
* each request's own timestamp, the same way it must for
|
|
249
|
+
* {@link priceMultiplierAt}.
|
|
136
250
|
*
|
|
137
251
|
* @param modelDef - The model definition.
|
|
138
252
|
* @param requested - The user's per-model region choice, if any.
|
|
253
|
+
* @param at - The instant to price at (defaults to now).
|
|
139
254
|
* @returns The region-effective rates.
|
|
140
255
|
*/
|
|
141
|
-
export function modelRegionRates(modelDef, requested) {
|
|
256
|
+
export function modelRegionRates(modelDef, requested, at = new Date()) {
|
|
142
257
|
const region = effectiveModelRegion(modelDef, requested);
|
|
143
258
|
const override = modelDef.regionPricing?.[region];
|
|
144
259
|
if (!override) {
|
|
145
|
-
return
|
|
146
|
-
inputPricePerMTok: modelDef.inputPricePerMTok,
|
|
147
|
-
outputPricePerMTok: modelDef.outputPricePerMTok,
|
|
148
|
-
cacheReadPricePerMTok: modelDef.cacheReadPricePerMTok,
|
|
149
|
-
cacheWritePricePerMTok: modelDef.cacheWritePricePerMTok,
|
|
150
|
-
};
|
|
260
|
+
return effectiveBaseRates(modelDef, at);
|
|
151
261
|
}
|
|
152
262
|
return {
|
|
153
263
|
inputPricePerMTok: override.inputPricePerMTok,
|
package/dist/models.d.ts
CHANGED
|
@@ -61,15 +61,24 @@ import type { ModelDefinition } from './types.js';
|
|
|
61
61
|
* 2026-07-21 $1.50/$7.50 supersedes 3.5-flash as the agentic flagship;
|
|
62
62
|
* gemini-3.1-pro-preview still the pro tier — "3.5 Pro" has NOT shipped as
|
|
63
63
|
* of 2026-07-28 despite the coming-soon badge; do not add until it has an id)
|
|
64
|
+
* (re-verified 2026-08-13: gemini-3.7-flash "New Stable" — supersedes
|
|
65
|
+
* 3.6-flash as the flash flagship at the SAME list price ($1.50/$7.50, cache
|
|
66
|
+
* read $0.15), with a launch promo ($0.75/$3.75, cache read $0.075) through
|
|
67
|
+
* 2026-12-31 billed here at list; specs from /docs/models/gemini-3.7-flash:
|
|
68
|
+
* 1M ctx / 65,536 out, thinking low|medium|high (no minimal), vision, tools,
|
|
69
|
+
* caching, search grounding, code execution, url context)
|
|
64
70
|
* - xAI: https://docs.x.ai/developers/models + /developers/grok-4-5
|
|
65
71
|
* (grok-4.5 flagship 2026-07-08: $2/$6, 500K ctx, ≥200K prompts bill 2× —
|
|
66
72
|
* not modeled; reasoning_effort low|medium|high default high, image input;
|
|
67
73
|
* grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
|
|
68
74
|
* grok-code-fast-1 no longer listed — retires 2026-08-15)
|
|
69
|
-
* - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (
|
|
70
|
-
*
|
|
71
|
-
*
|
|
72
|
-
*
|
|
75
|
+
* - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
|
|
76
|
+
* 2026-08-14; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
|
|
77
|
+
* never in this catalog. V4-Pro GA on 2026-08-13 came with a price RISE
|
|
78
|
+
* effective 2026-08-16T16:00Z plus the long-announced peak-hour 2×: both
|
|
79
|
+
* entries carry it as `scheduledPricing`, so today's rates bill until that
|
|
80
|
+
* instant and the new ones after. Re-verify weekday-vs-daily peak windows and
|
|
81
|
+
* the CN/US region default once it lands — see the entries.)
|
|
73
82
|
* - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
|
|
74
83
|
* the US re-host (kimi-k3 flagship 2026-07-16
|
|
75
84
|
* — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
|
package/dist/models.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD
|
|
1
|
+
{"version":3,"file":"models.d.ts","sourceRoot":"","sources":["../src/models.ts"],"names":[],"mappings":"AAAA;;;;;;;;GAQG;AAEH,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,YAAY,CAAA;AAEjD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAmGG;AACH,eAAO,MAAM,MAAM,EAAE,SAAS,eAAe,EA+5CnC,CAAA"}
|
package/dist/models.js
CHANGED
|
@@ -60,15 +60,24 @@
|
|
|
60
60
|
* 2026-07-21 $1.50/$7.50 supersedes 3.5-flash as the agentic flagship;
|
|
61
61
|
* gemini-3.1-pro-preview still the pro tier — "3.5 Pro" has NOT shipped as
|
|
62
62
|
* of 2026-07-28 despite the coming-soon badge; do not add until it has an id)
|
|
63
|
+
* (re-verified 2026-08-13: gemini-3.7-flash "New Stable" — supersedes
|
|
64
|
+
* 3.6-flash as the flash flagship at the SAME list price ($1.50/$7.50, cache
|
|
65
|
+
* read $0.15), with a launch promo ($0.75/$3.75, cache read $0.075) through
|
|
66
|
+
* 2026-12-31 billed here at list; specs from /docs/models/gemini-3.7-flash:
|
|
67
|
+
* 1M ctx / 65,536 out, thinking low|medium|high (no minimal), vision, tools,
|
|
68
|
+
* caching, search grounding, code execution, url context)
|
|
63
69
|
* - xAI: https://docs.x.ai/developers/models + /developers/grok-4-5
|
|
64
70
|
* (grok-4.5 flagship 2026-07-08: $2/$6, 500K ctx, ≥200K prompts bill 2× —
|
|
65
71
|
* not modeled; reasoning_effort low|medium|high default high, image input;
|
|
66
72
|
* grok-4.3 still served at $1.25/$2.50 with the bigger 1M window;
|
|
67
73
|
* grok-code-fast-1 no longer listed — retires 2026-08-15)
|
|
68
|
-
* - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (
|
|
69
|
-
*
|
|
70
|
-
*
|
|
71
|
-
*
|
|
74
|
+
* - DeepSeek: https://api-docs.deepseek.com/quick_start/pricing (verified
|
|
75
|
+
* 2026-08-14; legacy deepseek-chat/-reasoner ids fully retired 2026-07-24 —
|
|
76
|
+
* never in this catalog. V4-Pro GA on 2026-08-13 came with a price RISE
|
|
77
|
+
* effective 2026-08-16T16:00Z plus the long-announced peak-hour 2×: both
|
|
78
|
+
* entries carry it as `scheduledPricing`, so today's rates bill until that
|
|
79
|
+
* instant and the new ones after. Re-verify weekday-vs-daily peak windows and
|
|
80
|
+
* the CN/US region default once it lands — see the entries.)
|
|
72
81
|
* - Moonshot: https://platform.kimi.ai/docs/models + DeepInfra's model API for
|
|
73
82
|
* the US re-host (kimi-k3 flagship 2026-07-16
|
|
74
83
|
* — 2.8T MoE, 1M ctx, $3/$15 — NOT added: thinking is forced-on with
|
|
@@ -572,11 +581,46 @@ export const MODELS = [
|
|
|
572
581
|
// replace outright: the google bond has never been implemented/wired, so no
|
|
573
582
|
// historical usage can reference the old ids.
|
|
574
583
|
// ---------------------------------------------------------------------------
|
|
584
|
+
{
|
|
585
|
+
id: 'gemini-3.7-flash',
|
|
586
|
+
provider: 'google',
|
|
587
|
+
label: 'Gemini 3.7 Flash',
|
|
588
|
+
description: 'Google agentic flagship — complex coding & multi-step execution',
|
|
589
|
+
// Verified against /docs/models/gemini-3.7-flash (2026-08-13).
|
|
590
|
+
contextWindow: 1_048_576,
|
|
591
|
+
maxOutputTokens: 65_536,
|
|
592
|
+
supportsThinking: true,
|
|
593
|
+
thinkingBudgetTokens: 10_000,
|
|
594
|
+
thinkingConfigurable: true,
|
|
595
|
+
// thinking_level low|medium|high — minimal NOT supported on this model.
|
|
596
|
+
supportedEffortLevels: ['low', 'medium', 'high'],
|
|
597
|
+
defaultEffortLevel: 'medium',
|
|
598
|
+
supportsVision: true,
|
|
599
|
+
supportsPromptCaching: true,
|
|
600
|
+
supportsTools: true,
|
|
601
|
+
webSearchToolType: 'google_search',
|
|
602
|
+
codeExecutionToolType: 'code_execution',
|
|
603
|
+
webFetchToolType: 'url_context',
|
|
604
|
+
// "New Stable" 2026-08-13. LIST price $1.50/$7.50 — same as 3.6-flash.
|
|
605
|
+
// Google runs a launch promo ($0.75/$3.75, cache read $0.075) through
|
|
606
|
+
// 2026-12-31; billed here at standard list so metering never under-charges
|
|
607
|
+
// (same policy as claude-sonnet-5's intro pricing — see the matching
|
|
608
|
+
// KNOWN_DIVERGENCES entry in scripts/check-model-freshness.mjs, expiring
|
|
609
|
+
// 2026-12-31).
|
|
610
|
+
inputPricePerMTok: 1.5,
|
|
611
|
+
outputPricePerMTok: 7.5,
|
|
612
|
+
// Gemini context cache: read $0.15/M (0.1× input), no write premium
|
|
613
|
+
// (storage billed separately per hour — not modeled).
|
|
614
|
+
cacheReadPricePerMTok: 0.15,
|
|
615
|
+
cacheWritePricePerMTok: 1.5,
|
|
616
|
+
// Not on Google's docs — models.dev reports 2026-03 (lead, not authority).
|
|
617
|
+
knowledgeCutoff: '2026-03-01',
|
|
618
|
+
},
|
|
575
619
|
{
|
|
576
620
|
id: 'gemini-3.6-flash',
|
|
577
621
|
provider: 'google',
|
|
578
622
|
label: 'Gemini 3.6 Flash',
|
|
579
|
-
description: 'Google agentic flagship — frontier intelligence + grounding',
|
|
623
|
+
description: 'Previous Google agentic flagship — frontier intelligence + grounding',
|
|
580
624
|
// Window/output not on the pricing page — carried over from 3.5-flash;
|
|
581
625
|
// re-verify against /docs/models.
|
|
582
626
|
contextWindow: 1_048_576,
|
|
@@ -603,6 +647,11 @@ export const MODELS = [
|
|
|
603
647
|
cacheWritePricePerMTok: 1.5,
|
|
604
648
|
// Not published — best-effort estimate.
|
|
605
649
|
knowledgeCutoff: '2026-01-01',
|
|
650
|
+
// Superseded by gemini-3.7-flash (2026-08-13) — same flash tier, same list
|
|
651
|
+
// price; Google's own models page now calls 3.6 "previous-generation".
|
|
652
|
+
// Still served upstream, so it stays priceable.
|
|
653
|
+
deprecatedAt: '2026-08-13',
|
|
654
|
+
supersededBy: 'gemini-3.7-flash',
|
|
606
655
|
},
|
|
607
656
|
{
|
|
608
657
|
id: 'gemini-3.5-flash',
|
|
@@ -632,10 +681,12 @@ export const MODELS = [
|
|
|
632
681
|
cacheReadPricePerMTok: 0.15,
|
|
633
682
|
cacheWritePricePerMTok: 1.5,
|
|
634
683
|
knowledgeCutoff: '2025-01-01',
|
|
635
|
-
// Superseded by
|
|
636
|
-
//
|
|
684
|
+
// Superseded within the flash tier (first by 3.6-flash on 2026-07-21, now
|
|
685
|
+
// pointed one hop to gemini-3.7-flash — supersededBy must target a
|
|
686
|
+
// SELECTABLE model, never a chain). Still served upstream, so it stays
|
|
687
|
+
// priceable.
|
|
637
688
|
deprecatedAt: '2026-07-21',
|
|
638
|
-
supersededBy: 'gemini-3.
|
|
689
|
+
supersededBy: 'gemini-3.7-flash',
|
|
639
690
|
},
|
|
640
691
|
{
|
|
641
692
|
id: 'gemini-3.1-pro-preview',
|
|
@@ -835,20 +886,44 @@ export const MODELS = [
|
|
|
835
886
|
// DeepSeek
|
|
836
887
|
// Verified: https://api-docs.deepseek.com/quick_start/pricing
|
|
837
888
|
// https://api-docs.deepseek.com/guides/thinking_mode
|
|
838
|
-
// https://api-docs.deepseek.com/updates/ (2026-
|
|
889
|
+
// https://api-docs.deepseek.com/updates/ (2026-08-14)
|
|
890
|
+
// 2026-08-13: V4-Pro GA — and with it the price rise that the "coming soon"
|
|
891
|
+
// note below had been waiting on. It is STAGED, not applied: both models
|
|
892
|
+
// carry `scheduledPricing` effective 2026-08-16T16:00Z, so the catalog bills
|
|
893
|
+
// today's verified rates until that instant and the new ones after it, with
|
|
894
|
+
// nobody landing an edit at 16:00 UTC on a Sunday. The new card is
|
|
895
|
+
// off-peak/peak (peak = exactly 2× off-peak), so it maps onto base rates +
|
|
896
|
+
// `peakPricing` multiplier 2 — which is why the peak windows removed below
|
|
897
|
+
// come back here rather than as flat rates.
|
|
898
|
+
// pro off-peak 0.66 / 1.98, cache hit 0.022 (peak 1.32 / 3.96 / 0.044)
|
|
899
|
+
// flash off-peak 0.22 / 0.66, cache hit 0.007 (peak 0.44 / 1.32 / 0.014)
|
|
900
|
+
// Cache HITS are the real move — pro 0.003625 → 0.022 (6.1×) off-peak, 0.044
|
|
901
|
+
// (12.1×) at peak — and agentic input is ~94% cache hits, so effective input
|
|
902
|
+
// cost rises far more than the list prices suggest. Both are free-tier models
|
|
903
|
+
// (flash is `freeTier`, pro is the free-tier planner) on the CN default.
|
|
904
|
+
// TWO things to re-verify once it lands (2026-08-17):
|
|
905
|
+
// 1. WEEKDAYS OR DAILY. The rate card says only "Peak hours are 01:00 -
|
|
906
|
+
// 04:00 and 06:00 - 10:00 UTC (all other hours are off-peak)" with no
|
|
907
|
+
// day qualifier, so the windows below are DAILY per the provider's own
|
|
908
|
+
// doc; press coverage described them as weekday-only. `peakPricing` has
|
|
909
|
+
// no day-of-week concept, so if it is weekday-only this over-bills every
|
|
910
|
+
// weekend peak window and needs the field extended, not the numbers
|
|
911
|
+
// nudged.
|
|
912
|
+
// 2. THE CN-VS-US DEFAULT. `regions: ['cn', 'us']` defaults to CN on an
|
|
913
|
+
// owner decision (2026-08-01) taken when CN ran ~5.7× cheaper on real
|
|
914
|
+
// traffic. Post-change DeepInfra's US flash rates (0.08/0.18/0.016) are
|
|
915
|
+
// BELOW CN's new off-peak on both input and output — CN wins only on
|
|
916
|
+
// cache reads. Re-derive against measured cache-hit ratios before
|
|
917
|
+
// leaving the default where it is.
|
|
839
918
|
// 2026-07-31: DeepSeek-V4-Flash OFFICIAL API launched in public beta — the
|
|
840
919
|
// SAME `deepseek-v4-flash` id now serves the re-post-trained 0731 build
|
|
841
920
|
// (same architecture/size; much stronger agent benchmarks — beats
|
|
842
|
-
// V4-Pro-Preview on Terminal Bench 2.1 / DeepSWE).
|
|
843
|
-
// capability changes. V4-Pro official release "coming soon" — re-verify
|
|
844
|
-
// pricing THEN (the announced peak-hour 2× was tied to the V4 official
|
|
845
|
-
// rollout and is still not on the rate card).
|
|
921
|
+
// V4-Pro-Preview on Terminal Bench 2.1 / DeepSWE).
|
|
846
922
|
// OpenAI/Anthropic-compatible API; text/code only (no vision); 1M context,
|
|
847
923
|
// 384K max output, automatic context (prompt) caching with ABSOLUTE cache-hit
|
|
848
924
|
// prices (~1/50–1/120 of miss — not the old 0.1× rule). Launch discount made
|
|
849
|
-
// PERMANENT 2026-05-23 (Pro $1.74/$3.48 → $0.435/$0.87)
|
|
850
|
-
//
|
|
851
|
-
// then. Thinking now defaults ENABLED upstream and supports tool calling
|
|
925
|
+
// PERMANENT 2026-05-23 (Pro $1.74/$3.48 → $0.435/$0.87), and ENDED by the
|
|
926
|
+
// 2026-08-16 rise above. Thinking now defaults ENABLED upstream and supports tool calling
|
|
852
927
|
// (reasoning_effort: high|max), BUT tool loops in thinking mode must replay
|
|
853
928
|
// assistant reasoning_content on every subsequent request (400 on omission).
|
|
854
929
|
// The bond explicitly sends thinking:{type:"disabled"} — Synthase runs
|
|
@@ -874,28 +949,49 @@ export const MODELS = [
|
|
|
874
949
|
// DeepSeek automatic context cache: absolute cache-hit price ($/M).
|
|
875
950
|
cacheReadPricePerMTok: 0.003625,
|
|
876
951
|
cacheWritePricePerMTok: 0.435,
|
|
877
|
-
// Native-China DEFAULT (owner decision 2026-08-01
|
|
878
|
-
// (DeepInfra) bills ~3× list and ~28× cache reads, and
|
|
879
|
-
// ~94% cache hits, so US processing ran ~5.7× native on
|
|
880
|
-
//
|
|
952
|
+
// Native-China DEFAULT (owner decision 2026-08-01, re-derived 2026-08-14):
|
|
953
|
+
// the US re-host (DeepInfra) bills ~3× list and ~28× cache reads, and
|
|
954
|
+
// agentic input is ~94% cache hits, so US processing ran ~5.7× native on
|
|
955
|
+
// real traffic. The 2026-08-16 rise narrows that to ~2.3× — still decisive,
|
|
956
|
+
// so Pro stays CN while Flash flipped to US (see its note). Users opt into
|
|
957
|
+
// US per model via the picker's region control.
|
|
881
958
|
regions: ['cn', 'us'],
|
|
882
|
-
//
|
|
883
|
-
//
|
|
884
|
-
//
|
|
885
|
-
|
|
959
|
+
// No freeTierRegions: the free tier stopped planning with this model on
|
|
960
|
+
// 2026-08-14 (minimax-m3 took over — cheaper, and it beat this model on the
|
|
961
|
+
// selection self-test). The carve-out only ever existed to keep the free
|
|
962
|
+
// tier's OWN plan default usable, and `freeTierAllows` checks
|
|
963
|
+
// `FREE_TIER_MODELS[mode] === modelId` before it looks at regions, so
|
|
964
|
+
// leaving it here would widen nothing — it would just claim a free-tier
|
|
965
|
+
// relationship that no longer exists.
|
|
886
966
|
// US = DeepInfra, verified 2026-08-01 via api.deepinfra.com/models/
|
|
887
967
|
// deepseek-ai/DeepSeek-V4-Pro. No cache-write premium (omitted → region
|
|
888
968
|
// input rate).
|
|
889
969
|
regionPricing: {
|
|
890
970
|
us: { inputPricePerMTok: 1.3, outputPricePerMTok: 2.6, cacheReadPricePerMTok: 0.1 },
|
|
891
971
|
},
|
|
892
|
-
// The
|
|
893
|
-
//
|
|
894
|
-
//
|
|
895
|
-
//
|
|
896
|
-
//
|
|
897
|
-
//
|
|
898
|
-
//
|
|
972
|
+
// The peak-hour 2× surcharge is now ON the rate card with a dated switch
|
|
973
|
+
// (2026-08-13 announcement, effective 2026-08-16T16:00Z) — so it is staged
|
|
974
|
+
// below rather than live. The previously pre-wired windows had been REMOVED
|
|
975
|
+
// for over-billing every peak-window turn 2× for weeks against a rate card
|
|
976
|
+
// that showed a single flat rate; staging is what keeps this from repeating
|
|
977
|
+
// in the other direction. Peak = 01:00-04:00 and 06:00-10:00 UTC (Beijing
|
|
978
|
+
// business hours), which is 2× the off-peak rates exactly.
|
|
979
|
+
scheduledPricing: {
|
|
980
|
+
effectiveFrom: '2026-08-16T16:00:00Z',
|
|
981
|
+
inputPricePerMTok: 0.66,
|
|
982
|
+
outputPricePerMTok: 1.98,
|
|
983
|
+
cacheReadPricePerMTok: 0.022,
|
|
984
|
+
// DeepSeek charges no cache-write premium — write bills at input.
|
|
985
|
+
cacheWritePricePerMTok: 0.66,
|
|
986
|
+
peakPricing: {
|
|
987
|
+
windows: [
|
|
988
|
+
{ startMinuteUtc: 60, endMinuteUtc: 240 },
|
|
989
|
+
{ startMinuteUtc: 360, endMinuteUtc: 600 },
|
|
990
|
+
],
|
|
991
|
+
multiplier: 2,
|
|
992
|
+
},
|
|
993
|
+
source: 'https://api-docs.deepseek.com/quick_start/pricing/',
|
|
994
|
+
},
|
|
899
995
|
// Not published by DeepSeek — best-effort estimate.
|
|
900
996
|
knowledgeCutoff: '2025-07-01',
|
|
901
997
|
},
|
|
@@ -923,10 +1019,19 @@ export const MODELS = [
|
|
|
923
1019
|
// DeepSeek automatic context cache: absolute cache-hit price ($/M).
|
|
924
1020
|
cacheReadPricePerMTok: 0.0028,
|
|
925
1021
|
cacheWritePricePerMTok: 0.14,
|
|
926
|
-
//
|
|
927
|
-
//
|
|
928
|
-
//
|
|
929
|
-
|
|
1022
|
+
// US (DeepInfra) DEFAULT as of 2026-08-16 — flipped from CN when DeepSeek's
|
|
1023
|
+
// rise landed (owner decision 2026-08-14). CN was cheaper on real traffic
|
|
1024
|
+
// only because of its cache reads; the rise takes those from $0.0028 to
|
|
1025
|
+
// $0.007 (peak $0.014) against DeepInfra's flat $0.016, which is no longer
|
|
1026
|
+
// enough to carry the 1.6-3.1x it now loses on fresh input and output. On
|
|
1027
|
+
// the agentic mix this model actually serves (~94% cache hits) US is
|
|
1028
|
+
// cheaper at EVERY hour: 0.114c/turn flat vs 0.152c off-peak and 0.303c at
|
|
1029
|
+
// peak. It is also flat-rate, so free-tier cost stops varying by Beijing
|
|
1030
|
+
// business hours. Re-derive if the cache-hit ratio drops much below ~90%,
|
|
1031
|
+
// where CN's cheaper reads start winning again. This deliberately splits
|
|
1032
|
+
// the plan/execute pair across regions — Pro stays CN because its US
|
|
1033
|
+
// re-host is ~2.3x its own native rate even after the rise.
|
|
1034
|
+
regions: ['us', 'cn'],
|
|
930
1035
|
// US = DeepInfra, verified 2026-08-13 against the id the bond actually
|
|
931
1036
|
// sends: `deepseek-ai/DeepSeek-V4-Flash-0731`, the official release that
|
|
932
1037
|
// supersedes the preview weights still served under the un-dated id
|
|
@@ -935,7 +1040,23 @@ export const MODELS = [
|
|
|
935
1040
|
regionPricing: {
|
|
936
1041
|
us: { inputPricePerMTok: 0.08, outputPricePerMTok: 0.18, cacheReadPricePerMTok: 0.016 },
|
|
937
1042
|
},
|
|
938
|
-
// Peak-hour surcharge
|
|
1043
|
+
// Peak-hour surcharge staged, not live (see deepseek-v4-pro).
|
|
1044
|
+
scheduledPricing: {
|
|
1045
|
+
effectiveFrom: '2026-08-16T16:00:00Z',
|
|
1046
|
+
inputPricePerMTok: 0.22,
|
|
1047
|
+
outputPricePerMTok: 0.66,
|
|
1048
|
+
cacheReadPricePerMTok: 0.007,
|
|
1049
|
+
// DeepSeek charges no cache-write premium — write bills at input.
|
|
1050
|
+
cacheWritePricePerMTok: 0.22,
|
|
1051
|
+
peakPricing: {
|
|
1052
|
+
windows: [
|
|
1053
|
+
{ startMinuteUtc: 60, endMinuteUtc: 240 },
|
|
1054
|
+
{ startMinuteUtc: 360, endMinuteUtc: 600 },
|
|
1055
|
+
],
|
|
1056
|
+
multiplier: 2,
|
|
1057
|
+
},
|
|
1058
|
+
source: 'https://api-docs.deepseek.com/quick_start/pricing/',
|
|
1059
|
+
},
|
|
939
1060
|
// Not published by DeepSeek — best-effort estimate.
|
|
940
1061
|
knowledgeCutoff: '2025-07-01',
|
|
941
1062
|
},
|
|
@@ -1137,8 +1258,21 @@ export const MODELS = [
|
|
|
1137
1258
|
// US default. DeepInfra list matches native; only the cache write differs
|
|
1138
1259
|
// (no premium → region input rate). Verified 2026-08-01.
|
|
1139
1260
|
regions: ['us', 'cn'],
|
|
1261
|
+
// The free tier PLANS with this model (molecule-dev FREE_TIER_MODELS.plan,
|
|
1262
|
+
// 2026-08-14), so its default US region must be free-tier selectable. It
|
|
1263
|
+
// took over from deepseek-v4-pro@cn: measured on the real starting-point
|
|
1264
|
+
// selection it scored 8/8 against Pro's 7/8 — including the case Pro failed
|
|
1265
|
+
// — at 1.28c/plan-turn flat versus Pro's 2.62c off-peak and 5.24c inside
|
|
1266
|
+
// DeepSeek's Beijing-hours windows, and it adds vision, which Pro (text
|
|
1267
|
+
// only) could not offer discovery. CN is NOT listed: it is dearer than US
|
|
1268
|
+
// here, so free planning stays on the cheaper host.
|
|
1269
|
+
freeTierRegions: ['us'],
|
|
1140
1270
|
regionPricing: {
|
|
1141
|
-
|
|
1271
|
+
// Verified 2026-08-14 against api.deepinfra.com/models/MiniMaxAI/MiniMax-M3
|
|
1272
|
+
// (cache read = 0.2 × input). Was 0.3/1.2/0.06 — DeepInfra had repriced
|
|
1273
|
+
// and nothing noticed, because the freshness gate's re-host check only
|
|
1274
|
+
// covered deepseek and moonshot until this date.
|
|
1275
|
+
us: { inputPricePerMTok: 0.28, outputPricePerMTok: 1.1, cacheReadPricePerMTok: 0.056 },
|
|
1142
1276
|
},
|
|
1143
1277
|
// From the official HF chat template ("Knowledge cutoff: January 2026").
|
|
1144
1278
|
knowledgeCutoff: '2026-01-01',
|
|
@@ -1245,6 +1379,14 @@ export const MODELS = [
|
|
|
1245
1379
|
cacheReadPricePerMTok: 0.4,
|
|
1246
1380
|
cacheWritePricePerMTok: 2,
|
|
1247
1381
|
regions: ['us', 'cn'],
|
|
1382
|
+
// US = DeepInfra (Qwen/Qwen3.8-Max), verified 2026-08-14 against
|
|
1383
|
+
// api.deepinfra.com/models/ (cache read = 0.1248 x input). ABSENT until then:
|
|
1384
|
+
// every US turn was metered at Alibaba's native rates while running on
|
|
1385
|
+
// DeepInfra, and the model 404'd outright because the bond's modelMap had
|
|
1386
|
+
// never been updated past qwen3.7-max.
|
|
1387
|
+
regionPricing: {
|
|
1388
|
+
us: { inputPricePerMTok: 1.65, outputPricePerMTok: 4.951, cacheReadPricePerMTok: 0.206 },
|
|
1389
|
+
},
|
|
1248
1390
|
// Not published by Alibaba — best-effort estimate.
|
|
1249
1391
|
knowledgeCutoff: '2026-04-01',
|
|
1250
1392
|
},
|
|
@@ -1274,6 +1416,11 @@ export const MODELS = [
|
|
|
1274
1416
|
// US default. DeepInfra bills identical rates (no regionPricing needed).
|
|
1275
1417
|
// Verified 2026-08-01.
|
|
1276
1418
|
regions: ['us', 'cn'],
|
|
1419
|
+
// US = DeepInfra, verified 2026-08-14 (cache read = 0.2 x input). Superseded,
|
|
1420
|
+
// but still priceable for historical usage, so its region rates must be real.
|
|
1421
|
+
regionPricing: {
|
|
1422
|
+
us: { inputPricePerMTok: 2.5, outputPricePerMTok: 7.5, cacheReadPricePerMTok: 0.5 },
|
|
1423
|
+
},
|
|
1277
1424
|
// Not published by Alibaba — best-effort estimate.
|
|
1278
1425
|
knowledgeCutoff: '2026-01-01',
|
|
1279
1426
|
// Superseded by qwen3.8-max (GA 2026-08-03): same tier and mechanism, and
|
package/dist/types.d.ts
CHANGED
|
@@ -232,6 +232,60 @@ export interface ModelDefinition {
|
|
|
232
232
|
}[];
|
|
233
233
|
multiplier: number;
|
|
234
234
|
};
|
|
235
|
+
/**
|
|
236
|
+
* A price change the provider has ANNOUNCED with a dated effective instant,
|
|
237
|
+
* staged ahead of time. Until `effectiveFrom` the model bills at the base
|
|
238
|
+
* rates above; from that instant on, these rates replace them.
|
|
239
|
+
*
|
|
240
|
+
* This exists because the freshness gate can only ever see prices that are
|
|
241
|
+
* ALREADY live: it diffs the catalog against models.dev's *current* rates, so
|
|
242
|
+
* a change announced today and effective in three days is invisible to it
|
|
243
|
+
* until after it lands — and the cron runs every 8h, so the catalog would
|
|
244
|
+
* under-meter for up to a third of a day at whatever the new rate is. Landing
|
|
245
|
+
* the new numbers early is not an option either: that over-bills every turn
|
|
246
|
+
* until the switch (the exact mistake the removed DeepSeek peak windows made
|
|
247
|
+
* for weeks against the free-tier default model). Staging with a timestamp is
|
|
248
|
+
* the only form that is correct on BOTH sides of the instant, and it needs no
|
|
249
|
+
* one awake at the switch.
|
|
250
|
+
*
|
|
251
|
+
* Applies to the BASE rates only — a `regionPricing` entry is a different
|
|
252
|
+
* host's rate card (a US re-host does not reprice because the native provider
|
|
253
|
+
* did) and is never touched by a scheduled change.
|
|
254
|
+
*
|
|
255
|
+
* `peakPricing` here, when declared, replaces the model's peak windows from
|
|
256
|
+
* the same instant; when omitted, the model's existing windows carry through
|
|
257
|
+
* unchanged. To schedule the END of peak pricing, declare an explicit
|
|
258
|
+
* `{ windows: [], multiplier: 1 }`.
|
|
259
|
+
*
|
|
260
|
+
* Resolution is `effectiveBaseRates()` / `effectivePeakPricing()`, and every
|
|
261
|
+
* consumer reaches it through `modelRegionRates()` / `priceMultiplierAt()` /
|
|
262
|
+
* the `withEffectivePricing()` projection the list handler serves — so a
|
|
263
|
+
* scheduled change lands everywhere at once with no follow-up edit. Once the
|
|
264
|
+
* instant has passed, fold the rates into the base fields and delete this
|
|
265
|
+
* (the freshness gate now verifies them against models.dev normally).
|
|
266
|
+
*/
|
|
267
|
+
scheduledPricing?: {
|
|
268
|
+
/** ISO-8601 UTC instant the new rates take effect. */
|
|
269
|
+
effectiveFrom: string;
|
|
270
|
+
/** Input price per million *uncached* tokens in USD, from `effectiveFrom`. */
|
|
271
|
+
inputPricePerMTok: number;
|
|
272
|
+
/** Output price per million tokens in USD, from `effectiveFrom`. */
|
|
273
|
+
outputPricePerMTok: number;
|
|
274
|
+
/** Prompt-cache *read* price per million tokens in USD, from `effectiveFrom`. */
|
|
275
|
+
cacheReadPricePerMTok: number;
|
|
276
|
+
/** Prompt-cache *write* price per million tokens in USD, from `effectiveFrom`. */
|
|
277
|
+
cacheWritePricePerMTok: number;
|
|
278
|
+
/** Peak-hour pricing from `effectiveFrom` (omitted → existing windows carry through). */
|
|
279
|
+
peakPricing?: {
|
|
280
|
+
windows: {
|
|
281
|
+
startMinuteUtc: number;
|
|
282
|
+
endMinuteUtc: number;
|
|
283
|
+
}[];
|
|
284
|
+
multiplier: number;
|
|
285
|
+
};
|
|
286
|
+
/** Where the change was announced, for the re-verify pass after it lands. */
|
|
287
|
+
source?: string;
|
|
288
|
+
};
|
|
235
289
|
/**
|
|
236
290
|
* Fast-mode ("priority speed") pricing — the per-MTok rates billed when a
|
|
237
291
|
* request runs with the provider's fast/priority tier (e.g. Anthropic's
|
package/dist/types.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"types.d.ts","sourceRoot":"","sources":["../src/types.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AAEH;;;;;GAKG;AACH,MAAM,MAAM,YAAY,GACpB,WAAW,GACX,QAAQ,GACR,QAAQ,GACR,KAAK,GACL,UAAU,GACV,MAAM,GACN,UAAU,GACV,SAAS,GACT,SAAS,GACT,OAAO;AACT;;;;;GAKG;GACD,QAAQ,CAAA;AAEZ;;;;;;;;;;;;;GAaG;AACH,MAAM,MAAM,WAAW,GAAG,MAAM,CAAA;AAEhC;;;GAGG;AACH,MAAM,WAAW,eAAe;IAC9B,8DAA8D;IAC9D,EAAE,EAAE,MAAM,CAAA;IACV,2CAA2C;IAC3C,QAAQ,EAAE,YAAY,CAAA;IACtB,yDAAyD;IACzD,KAAK,EAAE,MAAM,CAAA;IACb,wCAAwC;IACxC,WAAW,EAAE,MAAM,CAAA;IACnB,8CAA8C;IAC9C,aAAa,EAAE,MAAM,CAAA;IACrB,0CAA0C;IAC1C,eAAe,EAAE,MAAM,CAAA;IACvB,uEAAuE;IACvE,gBAAgB,EAAE,OAAO,CAAA;IACzB,yFAAyF;IACzF,oBAAoB,EAAE,MAAM,CAAA;IAC5B;;;OAGG;IACH,oBAAoB,EAAE,OAAO,CAAA;IAC7B;;;;;;;;;;;;;;;;;;;OAmBG;IACH,qBAAqB,CAAC,EAAE,WAAW,EAAE,CAAA;IACrC;;;;OAIG;IACH,kBAAkB,CAAC,EAAE,WAAW,CAAA;IAChC;;;;;;;;;;;OAWG;IACH,kBAAkB,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAA;IAC3C,mEAAmE;IACnE,cAAc,EAAE,OAAO,CAAA;IACvB,iDAAiD;IACjD,qBAAqB,EAAE,OAAO,CAAA;IAC9B,8DAA8D;IAC9D,aAAa,EAAE,OAAO,CAAA;IACtB;;;;;;;;;;;;;;;;;;;OAmBG;IACH,wBAAwB,CAAC,EAAE,OAAO,CAAA;IAClC;;;;OAIG;IACH,iBAAiB,CAAC,EAAE,MAAM,CAAA;IAC1B;;;OAGG;IACH,qBAAqB,CAAC,EAAE,MAAM,CAAA;IAC9B;;;OAGG;IACH,gBAAgB,CAAC,EAAE,MAAM,CAAA;IACzB,wFAAwF;IACxF,QAAQ,CAAC,EAAE,OAAO,CAAA;IAClB;;;;;;OAMG;IACH,eAAe,CAAC,EAAE,MAAM,EAAE,CAAA;IAC1B;;;;;;;OAOG;IACH,OAAO,CAAC,EAAE,MAAM,EAAE,CAAA;IAClB;;;;;;;;OAQG;IACH,aAAa,CAAC,EAAE,MAAM,CACpB,MAAM,EACN;QACE,6DAA6D;QAC7D,iBAAiB,EAAE,MAAM,CAAA;QACzB,qDAAqD;QACrD,kBAAkB,EAAE,MAAM,CAAA;QAC1B,gEAAgE;QAChE,qBAAqB,CAAC,EAAE,MAAM,CAAA;QAC9B,iEAAiE;QACjE,sBAAsB,CAAC,EAAE,MAAM,CAAA;KAChC,CACF,CAAA;IACD,sEAAsE;IACtE,iBAAiB,EAAE,MAAM,CAAA;IACzB,8CAA8C;IAC9C,kBAAkB,EAAE,MAAM,CAAA;IAC1B;;;;;;;;;;;OAWG;IACH,qBAAqB,EAAE,MAAM,CAAA;IAC7B;;;;;;;;;;OAUG;IACH,sBAAsB,EAAE,MAAM,CAAA;IAC9B;;;;;;;;;;OAUG;IACH,WAAW,CAAC,EAAE;QACZ,OAAO,EAAE;YAAE,cAAc,EAAE,MAAM,CAAC;YAAC,YAAY,EAAE,MAAM,CAAA;SAAE,EAAE,CAAA;QAC3D,UAAU,EAAE,MAAM,CAAA;KACnB,CAAA;IACD;;;;;;;;;;OAUG;IACH,WAAW,CAAC,EAAE;QACZ,gEAAgE;QAChE,iBAAiB,EAAE,MAAM,CAAA;QACzB,wDAAwD;QACxD,kBAAkB,EAAE,MAAM,CAAA;QAC1B,mEAAmE;QACnE,qBAAqB,EAAE,MAAM,CAAA;QAC7B,oEAAoE;QACpE,sBAAsB,EAAE,MAAM,CAAA;KAC/B,CAAA;IACD,mDAAmD;IACnD,eAAe,EAAE,MAAM,CAAA;IACvB;;;;;;;;;OASG;IACH,YAAY,CAAC,EAAE,MAAM,CAAA;IACrB;;;;;;;;;;;;;;OAcG;IACH,QAAQ,CAAC,EAAE,OAAO,CAAA;IAClB;;;;;;;;;;;;;;;;;;;;;;;;;;;OA2BG;IACH,YAAY,CAAC,EAAE,MAAM,CAAA;CACtB;AAED;;;;;;GAMG;AACH,MAAM,WAAW,iBAAiB;IAChC,6DAA6D;IAC7D,IAAI,EAAE,MAAM,CAAA;IACZ,gEAAgE;IAChE,OAAO,EAAE,MAAM,CAAA;IACf,8EAA8E;IAC9E,MAAM,EAAE,MAAM,CAAA;IACd,4EAA4E;IAC5E,OAAO,EAAE,MAAM,CAAA;CAChB;AAED;;GAEG;AACH,MAAM,WAAW,kBAAkB;IACjC,MAAM,EAAE,eAAe,EAAE,CAAA;IACzB;;;;OAIG;IACH,QAAQ,CAAC,EAAE,iBAAiB,CAAA;CAC7B"}
|
|
1
|
+
{"version":3,"file":"types.d.ts","sourceRoot":"","sources":["../src/types.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AAEH;;;;;GAKG;AACH,MAAM,MAAM,YAAY,GACpB,WAAW,GACX,QAAQ,GACR,QAAQ,GACR,KAAK,GACL,UAAU,GACV,MAAM,GACN,UAAU,GACV,SAAS,GACT,SAAS,GACT,OAAO;AACT;;;;;GAKG;GACD,QAAQ,CAAA;AAEZ;;;;;;;;;;;;;GAaG;AACH,MAAM,MAAM,WAAW,GAAG,MAAM,CAAA;AAEhC;;;GAGG;AACH,MAAM,WAAW,eAAe;IAC9B,8DAA8D;IAC9D,EAAE,EAAE,MAAM,CAAA;IACV,2CAA2C;IAC3C,QAAQ,EAAE,YAAY,CAAA;IACtB,yDAAyD;IACzD,KAAK,EAAE,MAAM,CAAA;IACb,wCAAwC;IACxC,WAAW,EAAE,MAAM,CAAA;IACnB,8CAA8C;IAC9C,aAAa,EAAE,MAAM,CAAA;IACrB,0CAA0C;IAC1C,eAAe,EAAE,MAAM,CAAA;IACvB,uEAAuE;IACvE,gBAAgB,EAAE,OAAO,CAAA;IACzB,yFAAyF;IACzF,oBAAoB,EAAE,MAAM,CAAA;IAC5B;;;OAGG;IACH,oBAAoB,EAAE,OAAO,CAAA;IAC7B;;;;;;;;;;;;;;;;;;;OAmBG;IACH,qBAAqB,CAAC,EAAE,WAAW,EAAE,CAAA;IACrC;;;;OAIG;IACH,kBAAkB,CAAC,EAAE,WAAW,CAAA;IAChC;;;;;;;;;;;OAWG;IACH,kBAAkB,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAA;IAC3C,mEAAmE;IACnE,cAAc,EAAE,OAAO,CAAA;IACvB,iDAAiD;IACjD,qBAAqB,EAAE,OAAO,CAAA;IAC9B,8DAA8D;IAC9D,aAAa,EAAE,OAAO,CAAA;IACtB;;;;;;;;;;;;;;;;;;;OAmBG;IACH,wBAAwB,CAAC,EAAE,OAAO,CAAA;IAClC;;;;OAIG;IACH,iBAAiB,CAAC,EAAE,MAAM,CAAA;IAC1B;;;OAGG;IACH,qBAAqB,CAAC,EAAE,MAAM,CAAA;IAC9B;;;OAGG;IACH,gBAAgB,CAAC,EAAE,MAAM,CAAA;IACzB,wFAAwF;IACxF,QAAQ,CAAC,EAAE,OAAO,CAAA;IAClB;;;;;;OAMG;IACH,eAAe,CAAC,EAAE,MAAM,EAAE,CAAA;IAC1B;;;;;;;OAOG;IACH,OAAO,CAAC,EAAE,MAAM,EAAE,CAAA;IAClB;;;;;;;;OAQG;IACH,aAAa,CAAC,EAAE,MAAM,CACpB,MAAM,EACN;QACE,6DAA6D;QAC7D,iBAAiB,EAAE,MAAM,CAAA;QACzB,qDAAqD;QACrD,kBAAkB,EAAE,MAAM,CAAA;QAC1B,gEAAgE;QAChE,qBAAqB,CAAC,EAAE,MAAM,CAAA;QAC9B,iEAAiE;QACjE,sBAAsB,CAAC,EAAE,MAAM,CAAA;KAChC,CACF,CAAA;IACD,sEAAsE;IACtE,iBAAiB,EAAE,MAAM,CAAA;IACzB,8CAA8C;IAC9C,kBAAkB,EAAE,MAAM,CAAA;IAC1B;;;;;;;;;;;OAWG;IACH,qBAAqB,EAAE,MAAM,CAAA;IAC7B;;;;;;;;;;OAUG;IACH,sBAAsB,EAAE,MAAM,CAAA;IAC9B;;;;;;;;;;OAUG;IACH,WAAW,CAAC,EAAE;QACZ,OAAO,EAAE;YAAE,cAAc,EAAE,MAAM,CAAC;YAAC,YAAY,EAAE,MAAM,CAAA;SAAE,EAAE,CAAA;QAC3D,UAAU,EAAE,MAAM,CAAA;KACnB,CAAA;IACD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;OA+BG;IACH,gBAAgB,CAAC,EAAE;QACjB,sDAAsD;QACtD,aAAa,EAAE,MAAM,CAAA;QACrB,8EAA8E;QAC9E,iBAAiB,EAAE,MAAM,CAAA;QACzB,oEAAoE;QACpE,kBAAkB,EAAE,MAAM,CAAA;QAC1B,iFAAiF;QACjF,qBAAqB,EAAE,MAAM,CAAA;QAC7B,kFAAkF;QAClF,sBAAsB,EAAE,MAAM,CAAA;QAC9B,yFAAyF;QACzF,WAAW,CAAC,EAAE;YACZ,OAAO,EAAE;gBAAE,cAAc,EAAE,MAAM,CAAC;gBAAC,YAAY,EAAE,MAAM,CAAA;aAAE,EAAE,CAAA;YAC3D,UAAU,EAAE,MAAM,CAAA;SACnB,CAAA;QACD,6EAA6E;QAC7E,MAAM,CAAC,EAAE,MAAM,CAAA;KAChB,CAAA;IACD;;;;;;;;;;OAUG;IACH,WAAW,CAAC,EAAE;QACZ,gEAAgE;QAChE,iBAAiB,EAAE,MAAM,CAAA;QACzB,wDAAwD;QACxD,kBAAkB,EAAE,MAAM,CAAA;QAC1B,mEAAmE;QACnE,qBAAqB,EAAE,MAAM,CAAA;QAC7B,oEAAoE;QACpE,sBAAsB,EAAE,MAAM,CAAA;KAC/B,CAAA;IACD,mDAAmD;IACnD,eAAe,EAAE,MAAM,CAAA;IACvB;;;;;;;;;OASG;IACH,YAAY,CAAC,EAAE,MAAM,CAAA;IACrB;;;;;;;;;;;;;;OAcG;IACH,QAAQ,CAAC,EAAE,OAAO,CAAA;IAClB;;;;;;;;;;;;;;;;;;;;;;;;;;;OA2BG;IACH,YAAY,CAAC,EAAE,MAAM,CAAA;CACtB;AAED;;;;;;GAMG;AACH,MAAM,WAAW,iBAAiB;IAChC,6DAA6D;IAC7D,IAAI,EAAE,MAAM,CAAA;IACZ,gEAAgE;IAChE,OAAO,EAAE,MAAM,CAAA;IACf,8EAA8E;IAC9E,MAAM,EAAE,MAAM,CAAA;IACd,4EAA4E;IAC5E,OAAO,EAAE,MAAM,CAAA;CAChB;AAED;;GAEG;AACH,MAAM,WAAW,kBAAkB;IACjC,MAAM,EAAE,eAAe,EAAE,CAAA;IACzB;;;;OAIG;IACH,QAAQ,CAAC,EAAE,iBAAiB,CAAA;CAC7B"}
|
package/package.json
CHANGED