@plurnk/plurnk-providers 1.5.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +36 -22
- package/SPEC.md +133 -59
- package/dist/AiSdkProvider.d.ts +19 -26
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +318 -106
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +4 -9
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +36 -9
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +2 -21
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +19 -14
- package/dist/Pool.js.map +1 -1
- package/dist/accounting.d.ts +5 -2
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js +100 -16
- package/dist/accounting.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +9 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +160 -62
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +7 -3
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +30 -24
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +18 -7
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +10 -10
- package/dist/cost.d.ts.map +1 -1
- package/dist/cost.js +90 -42
- package/dist/cost.js.map +1 -1
- package/dist/env.d.ts +5 -1
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +30 -10
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +14 -2
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +58 -2
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +4 -4
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -2
- package/dist/index.js.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/sdkModels.d.ts +6 -2
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +38 -5
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +33 -31
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +21 -5
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +164 -83
- package/dist/usage.js.map +1 -1
- package/package.json +7 -6
- package/src/AiSdkProvider.test.ts +788 -191
- package/src/AiSdkProvider.ts +381 -124
- package/src/Mock.test.ts +37 -12
- package/src/Mock.ts +45 -14
- package/src/Pool.test.ts +19 -6
- package/src/Pool.ts +20 -16
- package/src/ProviderRegistry.test.ts +16 -11
- package/src/accounting.test.ts +58 -22
- package/src/accounting.ts +120 -18
- package/src/aiSdkTransport.test.ts +42 -49
- package/src/aiSdkTransport.ts +174 -62
- package/src/boundaries.test.ts +1 -0
- package/src/catalogProvider.test.ts +258 -22
- package/src/catalogProvider.ts +42 -27
- package/src/compatibleProvider.test.ts +6 -3
- package/src/compatibleProvider.ts +20 -7
- package/src/cost.test.ts +55 -36
- package/src/cost.ts +111 -50
- package/src/defaults.test.ts +13 -3
- package/src/env.test.ts +54 -5
- package/src/env.ts +43 -18
- package/src/errors.test.ts +47 -2
- package/src/errors.ts +67 -3
- package/src/index.ts +21 -5
- package/src/ollama.test.ts +4 -1
- package/src/ollama.ts +3 -3
- package/src/sdkModels.test.ts +76 -4
- package/src/sdkModels.ts +45 -7
- package/src/types.ts +77 -38
- package/src/usage.test.ts +112 -116
- package/src/usage.ts +209 -93
package/src/usage.ts
CHANGED
|
@@ -1,94 +1,207 @@
|
|
|
1
|
-
//
|
|
2
|
-
//
|
|
3
|
-
//
|
|
4
|
-
// - OpenAI-style: reasoning is a SUBSET of completion_tokens, surfaced via
|
|
5
|
-
// completion_tokens_details.reasoning_tokens; total = prompt + completion.
|
|
6
|
-
// - Gemini-style: reasoning is OMITTED from completion_tokens and only
|
|
7
|
-
// recoverable as total - prompt - completion (no details field at all).
|
|
8
|
-
// - Fireworks-style: reasoning ships as TEXT (reasoning_content) but is folded
|
|
9
|
-
// into completion_tokens with NO reasoning_tokens itemization -- unrecoverable
|
|
10
|
-
// from the numbers alone, so it is re-split from the emitted text lengths.
|
|
11
|
-
// normalizeUsage collapses all three into one invariant (see ProviderUsage):
|
|
12
|
-
// total = prompt + completion + reasoning; cached ⊆ prompt;
|
|
13
|
-
// completion EXCLUDES reasoning; billable output = completion + reasoning.
|
|
1
|
+
// Provider usage normalization. The public shape follows {§provider-usage}:
|
|
2
|
+
// input/output totals own their optional cache/reasoning details, absence is
|
|
3
|
+
// unknown, and no token category is inferred from text length.
|
|
14
4
|
|
|
15
5
|
import type { ProviderUsage } from "./types.ts";
|
|
16
6
|
|
|
17
|
-
// Raw OpenAI-compatible usage block — the superset of fields providers emit.
|
|
18
7
|
export type RawUsage = {
|
|
8
|
+
input_tokens?: number;
|
|
9
|
+
output_tokens?: number;
|
|
19
10
|
prompt_tokens?: number;
|
|
20
11
|
completion_tokens?: number;
|
|
21
12
|
total_tokens?: number;
|
|
22
13
|
cached_tokens?: number;
|
|
14
|
+
cache_creation_input_tokens?: number;
|
|
15
|
+
cache_read_input_tokens?: number;
|
|
23
16
|
prompt_cache_hit_tokens?: number;
|
|
24
17
|
prompt_cache_miss_tokens?: number;
|
|
25
|
-
prompt_tokens_details?: {
|
|
18
|
+
prompt_tokens_details?: {
|
|
19
|
+
cached_tokens?: number;
|
|
20
|
+
cache_read_tokens?: number;
|
|
21
|
+
cache_write_tokens?: number;
|
|
22
|
+
};
|
|
23
|
+
input_tokens_details?: {
|
|
24
|
+
cached_tokens?: number;
|
|
25
|
+
cache_read_tokens?: number;
|
|
26
|
+
cache_write_tokens?: number;
|
|
27
|
+
};
|
|
26
28
|
completion_tokens_details?: { reasoning_tokens?: number };
|
|
29
|
+
output_tokens_details?: { reasoning_tokens?: number };
|
|
30
|
+
reasoning_tokens?: number;
|
|
27
31
|
};
|
|
28
32
|
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
return
|
|
41
|
-
...usage,
|
|
42
|
-
completion: usage.completion - reasoning,
|
|
43
|
-
reasoning,
|
|
44
|
-
};
|
|
33
|
+
const knownTokens = (value: unknown, name: string): number | undefined => {
|
|
34
|
+
if (value === undefined) return undefined;
|
|
35
|
+
if (!Number.isSafeInteger(value) || (value as number) < 0) {
|
|
36
|
+
throw new TypeError(`${name} must be a non-negative safe integer`);
|
|
37
|
+
}
|
|
38
|
+
return value as number;
|
|
39
|
+
};
|
|
40
|
+
|
|
41
|
+
const nonEmptyDetails = <T extends Record<string, number | undefined>>(details: T): T | undefined => {
|
|
42
|
+
const entries = Object.entries(details).filter((entry): entry is [string, number] =>
|
|
43
|
+
entry[1] !== undefined);
|
|
44
|
+
return entries.length === 0 ? undefined : Object.fromEntries(entries) as T;
|
|
45
45
|
};
|
|
46
46
|
|
|
47
|
-
export const
|
|
48
|
-
const
|
|
49
|
-
const
|
|
50
|
-
const
|
|
51
|
-
|
|
52
|
-
const
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
47
|
+
export const validateProviderUsage = (usage: ProviderUsage): ProviderUsage => {
|
|
48
|
+
const input = knownTokens(usage.inputTokens, "provider usage.inputTokens");
|
|
49
|
+
const output = knownTokens(usage.outputTokens, "provider usage.outputTokens");
|
|
50
|
+
const total = knownTokens(usage.totalTokens, "provider usage.totalTokens");
|
|
51
|
+
const inputDetails = usage.inputTokenDetails;
|
|
52
|
+
const outputDetails = usage.outputTokenDetails;
|
|
53
|
+
|
|
54
|
+
if (input === undefined && output === undefined && total === undefined
|
|
55
|
+
&& inputDetails === undefined && outputDetails === undefined) {
|
|
56
|
+
throw new TypeError("provider usage must contain at least one known quantity");
|
|
57
|
+
}
|
|
58
|
+
if (input !== undefined && output !== undefined && total !== undefined
|
|
59
|
+
&& total !== input + output) {
|
|
60
|
+
throw new TypeError("provider usage.totalTokens must equal inputTokens + outputTokens");
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
if (inputDetails !== undefined) {
|
|
64
|
+
const values = [
|
|
65
|
+
knownTokens(inputDetails.noCacheTokens, "provider usage.inputTokenDetails.noCacheTokens"),
|
|
66
|
+
knownTokens(inputDetails.cacheReadTokens, "provider usage.inputTokenDetails.cacheReadTokens"),
|
|
67
|
+
knownTokens(inputDetails.cacheWriteTokens, "provider usage.inputTokenDetails.cacheWriteTokens"),
|
|
68
|
+
];
|
|
69
|
+
if (values.every((value) => value === undefined)) {
|
|
70
|
+
throw new TypeError("provider usage.inputTokenDetails must contain a known quantity");
|
|
71
|
+
}
|
|
72
|
+
const known = values.filter((value): value is number => value !== undefined);
|
|
73
|
+
if (input !== undefined && known.some((value) => value > input)) {
|
|
74
|
+
throw new TypeError("provider input-token detail must not exceed inputTokens");
|
|
75
|
+
}
|
|
76
|
+
if (input !== undefined && values.every((value) => value !== undefined)
|
|
77
|
+
&& known.reduce((sum, value) => sum + value, 0) !== input) {
|
|
78
|
+
throw new TypeError("complete provider input-token details must sum to inputTokens");
|
|
79
|
+
}
|
|
79
80
|
}
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
81
|
+
|
|
82
|
+
if (outputDetails !== undefined) {
|
|
83
|
+
const values = [
|
|
84
|
+
knownTokens(outputDetails.textTokens, "provider usage.outputTokenDetails.textTokens"),
|
|
85
|
+
knownTokens(outputDetails.reasoningTokens, "provider usage.outputTokenDetails.reasoningTokens"),
|
|
86
|
+
];
|
|
87
|
+
if (values.every((value) => value === undefined)) {
|
|
88
|
+
throw new TypeError("provider usage.outputTokenDetails must contain a known quantity");
|
|
89
|
+
}
|
|
90
|
+
const known = values.filter((value): value is number => value !== undefined);
|
|
91
|
+
if (output !== undefined && known.some((value) => value > output)) {
|
|
92
|
+
throw new TypeError("provider output-token detail must not exceed outputTokens");
|
|
93
|
+
}
|
|
94
|
+
if (output !== undefined && values.every((value) => value !== undefined)
|
|
95
|
+
&& known.reduce((sum, value) => sum + value, 0) !== output) {
|
|
96
|
+
throw new TypeError("complete provider output-token details must sum to outputTokens");
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
return usage;
|
|
88
100
|
};
|
|
89
101
|
|
|
90
|
-
|
|
91
|
-
|
|
102
|
+
export const normalizeUsage = (raw: RawUsage | null | undefined): ProviderUsage | undefined => {
|
|
103
|
+
if (raw === null || raw === undefined) return undefined;
|
|
104
|
+
|
|
105
|
+
const inputTokens = knownTokens(
|
|
106
|
+
raw.input_tokens ?? raw.prompt_tokens,
|
|
107
|
+
"provider usage input tokens",
|
|
108
|
+
);
|
|
109
|
+
const rawOutputTokens = knownTokens(
|
|
110
|
+
raw.output_tokens ?? raw.completion_tokens,
|
|
111
|
+
"provider usage output tokens",
|
|
112
|
+
);
|
|
113
|
+
const reportedTotal = knownTokens(raw.total_tokens, "provider usage total tokens");
|
|
114
|
+
const reasoningTokens = knownTokens(
|
|
115
|
+
raw.output_tokens_details?.reasoning_tokens
|
|
116
|
+
?? raw.completion_tokens_details?.reasoning_tokens
|
|
117
|
+
?? raw.reasoning_tokens,
|
|
118
|
+
"provider usage reasoning tokens",
|
|
119
|
+
);
|
|
120
|
+
|
|
121
|
+
let outputTokens = rawOutputTokens;
|
|
122
|
+
let textTokens: number | undefined;
|
|
123
|
+
let normalizedReasoning = reasoningTokens;
|
|
124
|
+
if (inputTokens !== undefined && rawOutputTokens !== undefined && reportedTotal !== undefined) {
|
|
125
|
+
if (reportedTotal === inputTokens + rawOutputTokens) {
|
|
126
|
+
outputTokens = rawOutputTokens;
|
|
127
|
+
if (reasoningTokens !== undefined) textTokens = rawOutputTokens - reasoningTokens;
|
|
128
|
+
} else if (reasoningTokens !== undefined
|
|
129
|
+
&& reportedTotal === inputTokens + rawOutputTokens + reasoningTokens) {
|
|
130
|
+
outputTokens = rawOutputTokens + reasoningTokens;
|
|
131
|
+
textTokens = rawOutputTokens;
|
|
132
|
+
} else if (reportedTotal >= inputTokens + rawOutputTokens) {
|
|
133
|
+
outputTokens = reportedTotal - inputTokens;
|
|
134
|
+
normalizedReasoning = outputTokens - rawOutputTokens;
|
|
135
|
+
textTokens = rawOutputTokens;
|
|
136
|
+
} else {
|
|
137
|
+
throw new TypeError("provider usage total is inconsistent with input and output tokens");
|
|
138
|
+
}
|
|
139
|
+
} else if (rawOutputTokens !== undefined && reasoningTokens !== undefined) {
|
|
140
|
+
if (reasoningTokens > rawOutputTokens) {
|
|
141
|
+
outputTokens = rawOutputTokens + reasoningTokens;
|
|
142
|
+
textTokens = rawOutputTokens;
|
|
143
|
+
} else {
|
|
144
|
+
textTokens = rawOutputTokens - reasoningTokens;
|
|
145
|
+
}
|
|
146
|
+
} else if (outputTokens === undefined && inputTokens !== undefined && reportedTotal !== undefined) {
|
|
147
|
+
if (reportedTotal < inputTokens) {
|
|
148
|
+
throw new TypeError("provider usage total must not be less than input tokens");
|
|
149
|
+
}
|
|
150
|
+
outputTokens = reportedTotal - inputTokens;
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
const totalTokens = reportedTotal
|
|
154
|
+
?? (inputTokens !== undefined && outputTokens !== undefined
|
|
155
|
+
? inputTokens + outputTokens
|
|
156
|
+
: undefined);
|
|
157
|
+
const cacheReadTokens = knownTokens(
|
|
158
|
+
raw.input_tokens_details?.cache_read_tokens
|
|
159
|
+
?? raw.input_tokens_details?.cached_tokens
|
|
160
|
+
?? raw.prompt_tokens_details?.cache_read_tokens
|
|
161
|
+
?? raw.prompt_tokens_details?.cached_tokens
|
|
162
|
+
?? raw.cache_read_input_tokens
|
|
163
|
+
?? raw.prompt_cache_hit_tokens
|
|
164
|
+
?? raw.cached_tokens,
|
|
165
|
+
"provider usage cache-read tokens",
|
|
166
|
+
);
|
|
167
|
+
const cacheWriteTokens = knownTokens(
|
|
168
|
+
raw.input_tokens_details?.cache_write_tokens
|
|
169
|
+
?? raw.prompt_tokens_details?.cache_write_tokens
|
|
170
|
+
?? raw.cache_creation_input_tokens,
|
|
171
|
+
"provider usage cache-write tokens",
|
|
172
|
+
);
|
|
173
|
+
const explicitNoCacheTokens = knownTokens(
|
|
174
|
+
raw.prompt_cache_miss_tokens,
|
|
175
|
+
"provider usage non-cache tokens",
|
|
176
|
+
);
|
|
177
|
+
const noCacheTokens = explicitNoCacheTokens
|
|
178
|
+
?? (inputTokens !== undefined && cacheReadTokens !== undefined && cacheWriteTokens !== undefined
|
|
179
|
+
? inputTokens - cacheReadTokens - cacheWriteTokens
|
|
180
|
+
: undefined);
|
|
181
|
+
|
|
182
|
+
const usage: ProviderUsage = {
|
|
183
|
+
...(inputTokens === undefined ? {} : { inputTokens }),
|
|
184
|
+
...(outputTokens === undefined ? {} : { outputTokens }),
|
|
185
|
+
...(totalTokens === undefined ? {} : { totalTokens }),
|
|
186
|
+
...(nonEmptyDetails({ noCacheTokens, cacheReadTokens, cacheWriteTokens }) === undefined
|
|
187
|
+
? {}
|
|
188
|
+
: { inputTokenDetails: nonEmptyDetails({ noCacheTokens, cacheReadTokens, cacheWriteTokens })! }),
|
|
189
|
+
...(nonEmptyDetails({ textTokens, reasoningTokens: normalizedReasoning }) === undefined
|
|
190
|
+
? {}
|
|
191
|
+
: { outputTokenDetails: nonEmptyDetails({ textTokens, reasoningTokens: normalizedReasoning })! }),
|
|
192
|
+
};
|
|
193
|
+
if (Object.keys(usage).length === 0) return undefined;
|
|
194
|
+
return validateProviderUsage(usage);
|
|
195
|
+
};
|
|
196
|
+
|
|
197
|
+
// Models.dev rates are USD per million tokens. Optional cache rates inherit the
|
|
198
|
+
// input rate, so a missing usage detail matters only when its rate differs.
|
|
199
|
+
export type TokenRates = {
|
|
200
|
+
input: number;
|
|
201
|
+
output: number;
|
|
202
|
+
cacheRead?: number;
|
|
203
|
+
cacheWrite?: number;
|
|
204
|
+
};
|
|
92
205
|
|
|
93
206
|
const decimalParts = (value: number): { coefficient: bigint; scale: number } => {
|
|
94
207
|
if (!Number.isFinite(value) || value < 0) {
|
|
@@ -105,7 +218,7 @@ const decimalParts = (value: number): { coefficient: bigint; scale: number } =>
|
|
|
105
218
|
: { coefficient, scale };
|
|
106
219
|
};
|
|
107
220
|
|
|
108
|
-
const
|
|
221
|
+
export const canonicalDecimal = (coefficient: bigint, scale: number): string => {
|
|
109
222
|
const digits = String(coefficient).padStart(scale + 1, "0");
|
|
110
223
|
if (scale === 0) return digits;
|
|
111
224
|
const integer = digits.slice(0, -scale);
|
|
@@ -113,25 +226,28 @@ const decimalString = (coefficient: bigint, scale: number): string => {
|
|
|
113
226
|
return fraction.length === 0 ? integer : `${integer}.${fraction}`;
|
|
114
227
|
};
|
|
115
228
|
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
229
|
+
export const calculateCostUsdDecimal = (
|
|
230
|
+
usage: ProviderUsage,
|
|
231
|
+
rates: TokenRates,
|
|
232
|
+
): string | null => {
|
|
233
|
+
validateProviderUsage(usage);
|
|
234
|
+
if (usage.inputTokens === undefined || usage.outputTokens === undefined) return null;
|
|
235
|
+
|
|
236
|
+
const cacheReadRate = rates.cacheRead ?? rates.input;
|
|
237
|
+
const cacheWriteRate = rates.cacheWrite ?? rates.input;
|
|
238
|
+
const cacheReadTokens = usage.inputTokenDetails?.cacheReadTokens;
|
|
239
|
+
const cacheWriteTokens = usage.inputTokenDetails?.cacheWriteTokens;
|
|
240
|
+
if (cacheReadRate !== rates.input && cacheReadTokens === undefined) return null;
|
|
241
|
+
if (cacheWriteRate !== rates.input && cacheWriteTokens === undefined) return null;
|
|
242
|
+
|
|
243
|
+
const parts = [rates.input, rates.output, cacheReadRate, cacheWriteRate].map(decimalParts);
|
|
121
244
|
const rateScale = Math.max(...parts.map(({ scale }) => scale));
|
|
122
|
-
const [input,
|
|
245
|
+
const [input, output, cacheRead, cacheWrite] = parts.map(({ coefficient, scale }) =>
|
|
123
246
|
coefficient * 10n ** BigInt(rateScale - scale));
|
|
124
|
-
const
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
+ BigInt(
|
|
128
|
-
|
|
129
|
-
return
|
|
130
|
-
};
|
|
131
|
-
|
|
132
|
-
// The one cost formula every provider uses: non-cached prompt at the input
|
|
133
|
-
// rate, cached prompt at the cache rate, and billable output (completion +
|
|
134
|
-
// reasoning) at the output rate.
|
|
135
|
-
export const calculateCostUsd = (usage: ProviderUsage, rates: TokenRates): number => {
|
|
136
|
-
return Number(calculateCostUsdDecimal(usage, rates));
|
|
247
|
+
const coefficient = BigInt(usage.inputTokens) * input!
|
|
248
|
+
+ BigInt(usage.outputTokens) * output!
|
|
249
|
+
+ BigInt(cacheReadTokens ?? 0) * (cacheRead! - input!)
|
|
250
|
+
+ BigInt(cacheWriteTokens ?? 0) * (cacheWrite! - input!);
|
|
251
|
+
if (coefficient < 0n) throw new TypeError("provider token-rate details produced a negative estimate");
|
|
252
|
+
return canonicalDecimal(coefficient, rateScale + 6);
|
|
137
253
|
};
|