@outputai/llm 0.12.1-next.39bbc03.0 → 0.12.1-next.4abbd65.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +6 -4
- package/src/index.d.ts +3 -0
- package/src/utils/cost.js +7 -5
- package/src/utils/models_pricing.js +38 -13
- package/src/utils/models_pricing_snapshot.json +3324 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@outputai/llm",
|
|
3
|
-
"version": "0.12.1-next.
|
|
3
|
+
"version": "0.12.1-next.4abbd65.0",
|
|
4
4
|
"description": "Framework abstraction to interact with LLM models",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "src/index.js",
|
|
@@ -13,7 +13,8 @@
|
|
|
13
13
|
"./src/index.d.ts",
|
|
14
14
|
"./src/prompt/!(*.spec).js",
|
|
15
15
|
"./src/prompt/markup/!(*.spec).js",
|
|
16
|
-
"./src/utils/!(*.spec).js"
|
|
16
|
+
"./src/utils/!(*.spec).js",
|
|
17
|
+
"./src/utils/*.json"
|
|
17
18
|
],
|
|
18
19
|
"dependencies": {
|
|
19
20
|
"decimal.js": "10.6.0",
|
|
@@ -21,7 +22,7 @@
|
|
|
21
22
|
"gray-matter": "4.0.3",
|
|
22
23
|
"liquidjs": "10.27.2",
|
|
23
24
|
"undici": "8.9.0",
|
|
24
|
-
"@outputai/core": "0.12.1-next.
|
|
25
|
+
"@outputai/core": "0.12.1-next.4abbd65.0"
|
|
25
26
|
},
|
|
26
27
|
"devDependencies": {
|
|
27
28
|
"@ai-sdk/amazon-bedrock": "5.0.57",
|
|
@@ -46,6 +47,7 @@
|
|
|
46
47
|
"access": "public"
|
|
47
48
|
},
|
|
48
49
|
"scripts": {
|
|
49
|
-
"build": "pnpm exec tsc -p ./tsconfig.typecheck.json"
|
|
50
|
+
"build": "pnpm exec tsc -p ./tsconfig.typecheck.json",
|
|
51
|
+
"take-models-pricing-snapshot": "node ./scripts/take_models_pricing_snapshot.js"
|
|
50
52
|
}
|
|
51
53
|
}
|
package/src/index.d.ts
CHANGED
|
@@ -383,6 +383,7 @@ export interface LLMUsageEvent {
|
|
|
383
383
|
|
|
384
384
|
export type LLMGenerationCostItemStatus = 'ok' | 'fallback' | 'missing';
|
|
385
385
|
export type LLMGenerationCostStatus = 'precise' | 'imprecise' | 'incomplete';
|
|
386
|
+
export type LLMGenerationCostPricingFreshness = 'live' | 'cached' | 'stale' | 'snapshot';
|
|
386
387
|
|
|
387
388
|
/** Cost calculated for one normalized LLM usage item. */
|
|
388
389
|
export interface LLMGenerationCostItem {
|
|
@@ -404,6 +405,8 @@ export interface LLMGenerationCost extends BaseAttribute {
|
|
|
404
405
|
request: number | null;
|
|
405
406
|
total: number | null;
|
|
406
407
|
status: LLMGenerationCostStatus;
|
|
408
|
+
/** How current the rate table was. Absent on costs recorded before this field existed. */
|
|
409
|
+
pricingFreshness?: LLMGenerationCostPricingFreshness | null;
|
|
407
410
|
items: LLMGenerationCostItem[];
|
|
408
411
|
}
|
|
409
412
|
|
package/src/utils/cost.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { fetchModelsPricing } from './models_pricing.js';
|
|
1
|
+
import { fetchModelsPricing, Freshness } from './models_pricing.js';
|
|
2
2
|
import { Tracing } from '@outputai/core/sdk/runtime';
|
|
3
3
|
import { LLMGenerationUsage, LLMGenerationUsageItem } from './usage.js';
|
|
4
4
|
import { GroundingPpmMap } from './grounding.js';
|
|
@@ -45,19 +45,21 @@ export class LLMGenerationCost extends Tracing.Attribute.BaseAttribute {
|
|
|
45
45
|
request = null;
|
|
46
46
|
total = null;
|
|
47
47
|
status = LLMGenerationCost.Status.INCOMPLETE;
|
|
48
|
+
pricingFreshness = null;
|
|
48
49
|
items = [];
|
|
49
50
|
|
|
50
|
-
constructor( modelId, providerId, items, usageStatus ) {
|
|
51
|
+
constructor( modelId, providerId, items, usageStatus, pricingFreshness ) {
|
|
51
52
|
super( LLMGenerationCost.TYPE );
|
|
52
53
|
this.modelId = modelId;
|
|
53
54
|
this.providerId = providerId;
|
|
54
55
|
this.items = items;
|
|
56
|
+
this.pricingFreshness = pricingFreshness;
|
|
55
57
|
|
|
56
58
|
const meaningfulItems = items.filter( v => v.amount > 0 );
|
|
57
59
|
|
|
58
60
|
if ( meaningfulItems.some( v => v.status === LLMGenerationCostItem.Status.MISSING ) || usageStatus === LLMGenerationUsage.Status.INCOMPLETE ) {
|
|
59
61
|
this.status = LLMGenerationCost.Status.INCOMPLETE;
|
|
60
|
-
} else if ( meaningfulItems.some( v => v.status === LLMGenerationCostItem.Status.FALLBACK ) ) {
|
|
62
|
+
} else if ( meaningfulItems.some( v => v.status === LLMGenerationCostItem.Status.FALLBACK ) || pricingFreshness === Freshness.SNAPSHOT ) {
|
|
61
63
|
this.status = LLMGenerationCost.Status.IMPRECISE;
|
|
62
64
|
} else {
|
|
63
65
|
this.status = LLMGenerationCost.Status.PRECISE;
|
|
@@ -118,7 +120,7 @@ const resolvePrice = ( { group, label, pricing } ) => {
|
|
|
118
120
|
* @returns {Promise<LLMGenerationCost | null>} LLM generation cost with input, output, total and breakdown
|
|
119
121
|
*/
|
|
120
122
|
export const calculateCosts = async usage => {
|
|
121
|
-
const models = await fetchModelsPricing();
|
|
123
|
+
const { models, freshness: pricingFreshness } = await fetchModelsPricing();
|
|
122
124
|
|
|
123
125
|
if ( !models ) {
|
|
124
126
|
Logger.warn( 'Failed to fetch models pricing', { namespace: 'LLM' } );
|
|
@@ -144,5 +146,5 @@ export const calculateCosts = async usage => {
|
|
|
144
146
|
Logger.warn( 'Grounded call with no grounding rate for model', { namespace: 'LLM', modelId, providerId } );
|
|
145
147
|
}
|
|
146
148
|
|
|
147
|
-
return new LLMGenerationCost( modelId, providerId, items, usage.status );
|
|
149
|
+
return new LLMGenerationCost( modelId, providerId, items, usage.status, pricingFreshness );
|
|
148
150
|
};
|
|
@@ -1,9 +1,11 @@
|
|
|
1
1
|
import { Logger } from '@outputai/core';
|
|
2
2
|
import { EnvHttpProxyAgent, fetch } from 'undici';
|
|
3
|
+
import modelsPricingSnapshot from './models_pricing_snapshot.json' with { type: 'json' };
|
|
3
4
|
|
|
4
5
|
const logger = Logger.createLogger( 'LLM' );
|
|
5
6
|
const costTableUrl = 'https://models.dev/api.json';
|
|
6
7
|
const cacheTTL = 1000 * 60 * 60 * 24; // 1 day
|
|
8
|
+
const cooldownTTL = 1000 * 60 * 10; // 10 minutes
|
|
7
9
|
|
|
8
10
|
/* Ignore HTTP/2. Check: https://github.com/growthxai/output/issues/299 */
|
|
9
11
|
const dispatcher = new EnvHttpProxyAgent( { allowH2: false } );
|
|
@@ -13,23 +15,42 @@ export const cache = {
|
|
|
13
15
|
expiresAt: 0
|
|
14
16
|
};
|
|
15
17
|
|
|
18
|
+
export const Freshness = {
|
|
19
|
+
LIVE: 'live',
|
|
20
|
+
CACHED: 'cached',
|
|
21
|
+
STALE: 'stale',
|
|
22
|
+
SNAPSHOT: 'snapshot'
|
|
23
|
+
};
|
|
24
|
+
|
|
25
|
+
export const state = {
|
|
26
|
+
ignoreLiveRequestsUntil: 0
|
|
27
|
+
};
|
|
28
|
+
|
|
16
29
|
const parseData = data => {
|
|
17
30
|
const map = new Map();
|
|
18
31
|
try {
|
|
19
|
-
for ( const provider of Object.
|
|
32
|
+
for ( const [ providerId, provider ] of Object.entries( data ) ) {
|
|
33
|
+
if ( providerId === '_meta' ) {
|
|
34
|
+
continue;
|
|
35
|
+
}
|
|
20
36
|
for ( const [ modelName, { cost } ] of Object.entries( provider.models ?? {} ) ) {
|
|
21
37
|
if ( cost ) { // some models don't have cost
|
|
22
|
-
map.set( `${
|
|
38
|
+
map.set( `${providerId}/${modelName}`, cost );
|
|
23
39
|
}
|
|
24
40
|
}
|
|
25
41
|
}
|
|
42
|
+
if ( map.size === 0 ) {
|
|
43
|
+
throw new Error( 'Empty response' );
|
|
44
|
+
}
|
|
26
45
|
return map;
|
|
27
46
|
} catch ( error ) {
|
|
28
|
-
logger.error( `Models pricing: Data parsing failure "${error.
|
|
47
|
+
logger.error( `Models pricing: Data parsing failure "${error.message}".` );
|
|
29
48
|
return null;
|
|
30
49
|
}
|
|
31
50
|
};
|
|
32
51
|
|
|
52
|
+
const fallbackTable = parseData( modelsPricingSnapshot );
|
|
53
|
+
|
|
33
54
|
const fetchData = async () => {
|
|
34
55
|
try {
|
|
35
56
|
const res = await fetch( costTableUrl, { dispatcher } );
|
|
@@ -47,22 +68,26 @@ const fetchData = async () => {
|
|
|
47
68
|
|
|
48
69
|
export const fetchModelsPricing = async () => {
|
|
49
70
|
if ( cache.content && cache.expiresAt > Date.now() ) {
|
|
50
|
-
return cache.content;
|
|
71
|
+
return { models: cache.content, freshness: Freshness.CACHED };
|
|
51
72
|
}
|
|
52
73
|
|
|
53
|
-
|
|
54
|
-
|
|
74
|
+
if ( state.ignoreLiveRequestsUntil < Date.now() ) {
|
|
75
|
+
const table = await fetchData();
|
|
76
|
+
const models = table ? parseData( table ) : null;
|
|
55
77
|
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
78
|
+
if ( models ) {
|
|
79
|
+
cache.content = models;
|
|
80
|
+
cache.expiresAt = Date.now() + cacheTTL;
|
|
81
|
+
return { models, freshness: Freshness.LIVE };
|
|
82
|
+
} else {
|
|
83
|
+
state.ignoreLiveRequestsUntil = Date.now() + cooldownTTL;
|
|
84
|
+
}
|
|
60
85
|
}
|
|
61
86
|
|
|
62
87
|
if ( cache.content ) {
|
|
63
88
|
logger.warn( 'Models pricing: using stale cache.' );
|
|
64
|
-
return cache.content;
|
|
89
|
+
return { models: cache.content, freshness: Freshness.STALE };
|
|
65
90
|
}
|
|
66
|
-
|
|
67
|
-
return
|
|
91
|
+
logger.warn( 'Models pricing: using built-in fallback.' );
|
|
92
|
+
return { models: fallbackTable, freshness: Freshness.SNAPSHOT };
|
|
68
93
|
};
|