@plurnk/plurnk-providers 1.4.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. package/.env.defaults +40 -34
  2. package/README.md +3 -0
  3. package/SPEC.md +153 -62
  4. package/dist/AiSdkProvider.d.ts +19 -25
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +353 -120
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +7 -13
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +36 -8
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +2 -21
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +19 -14
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/accounting.d.ts +6 -0
  17. package/dist/accounting.d.ts.map +1 -0
  18. package/dist/accounting.js +168 -0
  19. package/dist/accounting.js.map +1 -0
  20. package/dist/aiSdkTransport.d.ts +11 -3
  21. package/dist/aiSdkTransport.d.ts.map +1 -1
  22. package/dist/aiSdkTransport.js +198 -29
  23. package/dist/aiSdkTransport.js.map +1 -1
  24. package/dist/catalogProvider.d.ts +7 -2
  25. package/dist/catalogProvider.d.ts.map +1 -1
  26. package/dist/catalogProvider.js +32 -26
  27. package/dist/catalogProvider.js.map +1 -1
  28. package/dist/compatibleProvider.d.ts.map +1 -1
  29. package/dist/compatibleProvider.js +18 -7
  30. package/dist/compatibleProvider.js.map +1 -1
  31. package/dist/cost.d.ts +10 -10
  32. package/dist/cost.d.ts.map +1 -1
  33. package/dist/cost.js +88 -43
  34. package/dist/cost.js.map +1 -1
  35. package/dist/env.d.ts +5 -7
  36. package/dist/env.d.ts.map +1 -1
  37. package/dist/env.js +30 -32
  38. package/dist/env.js.map +1 -1
  39. package/dist/errors.d.ts +14 -2
  40. package/dist/errors.d.ts.map +1 -1
  41. package/dist/errors.js +60 -2
  42. package/dist/errors.js.map +1 -1
  43. package/dist/index.d.ts +4 -4
  44. package/dist/index.d.ts.map +1 -1
  45. package/dist/index.js +3 -2
  46. package/dist/index.js.map +1 -1
  47. package/dist/ollama.js +3 -3
  48. package/dist/ollama.js.map +1 -1
  49. package/dist/sdkModels.d.ts +6 -0
  50. package/dist/sdkModels.d.ts.map +1 -1
  51. package/dist/sdkModels.js +46 -3
  52. package/dist/sdkModels.js.map +1 -1
  53. package/dist/types.d.ts +40 -29
  54. package/dist/types.d.ts.map +1 -1
  55. package/dist/usage.d.ts +21 -4
  56. package/dist/usage.d.ts.map +1 -1
  57. package/dist/usage.js +188 -74
  58. package/dist/usage.js.map +1 -1
  59. package/package.json +9 -7
  60. package/src/AiSdkProvider.test.ts +1039 -182
  61. package/src/AiSdkProvider.ts +428 -141
  62. package/src/Mock.test.ts +37 -12
  63. package/src/Mock.ts +46 -12
  64. package/src/Pool.test.ts +19 -6
  65. package/src/Pool.ts +20 -16
  66. package/src/ProviderRegistry.test.ts +16 -11
  67. package/src/accounting.test.ts +94 -0
  68. package/src/accounting.ts +190 -0
  69. package/src/aiSdkTransport.test.ts +42 -49
  70. package/src/aiSdkTransport.ts +218 -32
  71. package/src/boundaries.test.ts +2 -0
  72. package/src/catalogProvider.test.ts +271 -24
  73. package/src/catalogProvider.ts +44 -28
  74. package/src/compatibleProvider.test.ts +6 -3
  75. package/src/compatibleProvider.ts +20 -7
  76. package/src/cost.test.ts +55 -35
  77. package/src/cost.ts +110 -54
  78. package/src/defaults.test.ts +13 -3
  79. package/src/env.test.ts +50 -26
  80. package/src/env.ts +43 -42
  81. package/src/errors.test.ts +47 -2
  82. package/src/errors.ts +68 -3
  83. package/src/index.ts +21 -5
  84. package/src/ollama.test.ts +4 -1
  85. package/src/ollama.ts +3 -3
  86. package/src/sdkModels.test.ts +94 -3
  87. package/src/sdkModels.ts +53 -3
  88. package/src/types.ts +91 -33
  89. package/src/usage.test.ts +112 -108
  90. package/src/usage.ts +233 -84
package/src/usage.ts CHANGED
@@ -1,104 +1,253 @@
1
- // Usage normalization + cost — the shared token-accounting model.
2
- //
3
- // Providers report token usage in two incompatible ways:
4
- // - OpenAI-style: reasoning is a SUBSET of completion_tokens, surfaced via
5
- // completion_tokens_details.reasoning_tokens; total = prompt + completion.
6
- // - Gemini-style: reasoning is OMITTED from completion_tokens and only
7
- // recoverable as total - prompt - completion (no details field at all).
8
- // - Fireworks-style: reasoning ships as TEXT (reasoning_content) but is folded
9
- // into completion_tokens with NO reasoning_tokens itemization -- unrecoverable
10
- // from the numbers alone, so it is re-split from the emitted text lengths.
11
- // normalizeUsage collapses all three into one invariant (see ProviderUsage):
12
- // total = prompt + completion + reasoning; cached ⊆ prompt;
13
- // completion EXCLUDES reasoning; billable output = completion + reasoning.
1
+ // Provider usage normalization. The public shape follows {§provider-usage}:
2
+ // input/output totals own their optional cache/reasoning details, absence is
3
+ // unknown, and no token category is inferred from text length.
14
4
 
15
5
  import type { ProviderUsage } from "./types.ts";
16
6
 
17
- // Raw OpenAI-compatible usage block — the superset of fields providers emit.
18
7
  export type RawUsage = {
8
+ input_tokens?: number;
9
+ output_tokens?: number;
19
10
  prompt_tokens?: number;
20
11
  completion_tokens?: number;
21
12
  total_tokens?: number;
22
13
  cached_tokens?: number;
14
+ cache_creation_input_tokens?: number;
15
+ cache_read_input_tokens?: number;
23
16
  prompt_cache_hit_tokens?: number;
24
17
  prompt_cache_miss_tokens?: number;
25
- prompt_tokens_details?: { cached_tokens?: number };
18
+ prompt_tokens_details?: {
19
+ cached_tokens?: number;
20
+ cache_read_tokens?: number;
21
+ cache_write_tokens?: number;
22
+ };
23
+ input_tokens_details?: {
24
+ cached_tokens?: number;
25
+ cache_read_tokens?: number;
26
+ cache_write_tokens?: number;
27
+ };
26
28
  completion_tokens_details?: { reasoning_tokens?: number };
29
+ output_tokens_details?: { reasoning_tokens?: number };
30
+ reasoning_tokens?: number;
27
31
  };
28
32
 
29
- // Some providers return distinct reasoning text while reporting one combined
30
- // output count. Attribute that count without disturbing an upstream split.
31
- export const attributeUnitemizedReasoning = (
32
- usage: ProviderUsage,
33
- reasoningText: string,
34
- contentText: string,
35
- ): ProviderUsage => {
36
- if (usage.reasoning !== 0 || usage.completion === 0 || reasoningText.length === 0) return usage;
37
- const reasoning = contentText.length === 0
38
- ? usage.completion
39
- : Math.round(usage.completion * reasoningText.length / (reasoningText.length + contentText.length));
40
- return {
41
- ...usage,
42
- completion: usage.completion - reasoning,
43
- reasoning,
33
+ const knownTokens = (value: unknown, name: string): number | undefined => {
34
+ if (value === undefined) return undefined;
35
+ if (!Number.isSafeInteger(value) || (value as number) < 0) {
36
+ throw new TypeError(`${name} must be a non-negative safe integer`);
37
+ }
38
+ return value as number;
39
+ };
40
+
41
+ const nonEmptyDetails = <T extends Record<string, number | undefined>>(details: T): T | undefined => {
42
+ const entries = Object.entries(details).filter((entry): entry is [string, number] =>
43
+ entry[1] !== undefined);
44
+ return entries.length === 0 ? undefined : Object.fromEntries(entries) as T;
45
+ };
46
+
47
+ export const validateProviderUsage = (usage: ProviderUsage): ProviderUsage => {
48
+ const input = knownTokens(usage.inputTokens, "provider usage.inputTokens");
49
+ const output = knownTokens(usage.outputTokens, "provider usage.outputTokens");
50
+ const total = knownTokens(usage.totalTokens, "provider usage.totalTokens");
51
+ const inputDetails = usage.inputTokenDetails;
52
+ const outputDetails = usage.outputTokenDetails;
53
+
54
+ if (input === undefined && output === undefined && total === undefined
55
+ && inputDetails === undefined && outputDetails === undefined) {
56
+ throw new TypeError("provider usage must contain at least one known quantity");
57
+ }
58
+ if (input !== undefined && output !== undefined && total !== undefined
59
+ && total !== input + output) {
60
+ throw new TypeError("provider usage.totalTokens must equal inputTokens + outputTokens");
61
+ }
62
+
63
+ if (inputDetails !== undefined) {
64
+ const values = [
65
+ knownTokens(inputDetails.noCacheTokens, "provider usage.inputTokenDetails.noCacheTokens"),
66
+ knownTokens(inputDetails.cacheReadTokens, "provider usage.inputTokenDetails.cacheReadTokens"),
67
+ knownTokens(inputDetails.cacheWriteTokens, "provider usage.inputTokenDetails.cacheWriteTokens"),
68
+ ];
69
+ if (values.every((value) => value === undefined)) {
70
+ throw new TypeError("provider usage.inputTokenDetails must contain a known quantity");
71
+ }
72
+ const known = values.filter((value): value is number => value !== undefined);
73
+ if (input !== undefined && known.some((value) => value > input)) {
74
+ throw new TypeError("provider input-token detail must not exceed inputTokens");
75
+ }
76
+ if (input !== undefined && values.every((value) => value !== undefined)
77
+ && known.reduce((sum, value) => sum + value, 0) !== input) {
78
+ throw new TypeError("complete provider input-token details must sum to inputTokens");
79
+ }
80
+ }
81
+
82
+ if (outputDetails !== undefined) {
83
+ const values = [
84
+ knownTokens(outputDetails.textTokens, "provider usage.outputTokenDetails.textTokens"),
85
+ knownTokens(outputDetails.reasoningTokens, "provider usage.outputTokenDetails.reasoningTokens"),
86
+ ];
87
+ if (values.every((value) => value === undefined)) {
88
+ throw new TypeError("provider usage.outputTokenDetails must contain a known quantity");
89
+ }
90
+ const known = values.filter((value): value is number => value !== undefined);
91
+ if (output !== undefined && known.some((value) => value > output)) {
92
+ throw new TypeError("provider output-token detail must not exceed outputTokens");
93
+ }
94
+ if (output !== undefined && values.every((value) => value !== undefined)
95
+ && known.reduce((sum, value) => sum + value, 0) !== output) {
96
+ throw new TypeError("complete provider output-token details must sum to outputTokens");
97
+ }
98
+ }
99
+ return usage;
100
+ };
101
+
102
+ export const normalizeUsage = (raw: RawUsage | null | undefined): ProviderUsage | undefined => {
103
+ if (raw === null || raw === undefined) return undefined;
104
+
105
+ const inputTokens = knownTokens(
106
+ raw.input_tokens ?? raw.prompt_tokens,
107
+ "provider usage input tokens",
108
+ );
109
+ const rawOutputTokens = knownTokens(
110
+ raw.output_tokens ?? raw.completion_tokens,
111
+ "provider usage output tokens",
112
+ );
113
+ const reportedTotal = knownTokens(raw.total_tokens, "provider usage total tokens");
114
+ const reasoningTokens = knownTokens(
115
+ raw.output_tokens_details?.reasoning_tokens
116
+ ?? raw.completion_tokens_details?.reasoning_tokens
117
+ ?? raw.reasoning_tokens,
118
+ "provider usage reasoning tokens",
119
+ );
120
+
121
+ let outputTokens = rawOutputTokens;
122
+ let textTokens: number | undefined;
123
+ let normalizedReasoning = reasoningTokens;
124
+ if (inputTokens !== undefined && rawOutputTokens !== undefined && reportedTotal !== undefined) {
125
+ if (reportedTotal === inputTokens + rawOutputTokens) {
126
+ outputTokens = rawOutputTokens;
127
+ if (reasoningTokens !== undefined) textTokens = rawOutputTokens - reasoningTokens;
128
+ } else if (reasoningTokens !== undefined
129
+ && reportedTotal === inputTokens + rawOutputTokens + reasoningTokens) {
130
+ outputTokens = rawOutputTokens + reasoningTokens;
131
+ textTokens = rawOutputTokens;
132
+ } else if (reportedTotal >= inputTokens + rawOutputTokens) {
133
+ outputTokens = reportedTotal - inputTokens;
134
+ normalizedReasoning = outputTokens - rawOutputTokens;
135
+ textTokens = rawOutputTokens;
136
+ } else {
137
+ throw new TypeError("provider usage total is inconsistent with input and output tokens");
138
+ }
139
+ } else if (rawOutputTokens !== undefined && reasoningTokens !== undefined) {
140
+ if (reasoningTokens > rawOutputTokens) {
141
+ outputTokens = rawOutputTokens + reasoningTokens;
142
+ textTokens = rawOutputTokens;
143
+ } else {
144
+ textTokens = rawOutputTokens - reasoningTokens;
145
+ }
146
+ } else if (outputTokens === undefined && inputTokens !== undefined && reportedTotal !== undefined) {
147
+ if (reportedTotal < inputTokens) {
148
+ throw new TypeError("provider usage total must not be less than input tokens");
149
+ }
150
+ outputTokens = reportedTotal - inputTokens;
151
+ }
152
+
153
+ const totalTokens = reportedTotal
154
+ ?? (inputTokens !== undefined && outputTokens !== undefined
155
+ ? inputTokens + outputTokens
156
+ : undefined);
157
+ const cacheReadTokens = knownTokens(
158
+ raw.input_tokens_details?.cache_read_tokens
159
+ ?? raw.input_tokens_details?.cached_tokens
160
+ ?? raw.prompt_tokens_details?.cache_read_tokens
161
+ ?? raw.prompt_tokens_details?.cached_tokens
162
+ ?? raw.cache_read_input_tokens
163
+ ?? raw.prompt_cache_hit_tokens
164
+ ?? raw.cached_tokens,
165
+ "provider usage cache-read tokens",
166
+ );
167
+ const cacheWriteTokens = knownTokens(
168
+ raw.input_tokens_details?.cache_write_tokens
169
+ ?? raw.prompt_tokens_details?.cache_write_tokens
170
+ ?? raw.cache_creation_input_tokens,
171
+ "provider usage cache-write tokens",
172
+ );
173
+ const explicitNoCacheTokens = knownTokens(
174
+ raw.prompt_cache_miss_tokens,
175
+ "provider usage non-cache tokens",
176
+ );
177
+ const noCacheTokens = explicitNoCacheTokens
178
+ ?? (inputTokens !== undefined && cacheReadTokens !== undefined && cacheWriteTokens !== undefined
179
+ ? inputTokens - cacheReadTokens - cacheWriteTokens
180
+ : undefined);
181
+
182
+ const usage: ProviderUsage = {
183
+ ...(inputTokens === undefined ? {} : { inputTokens }),
184
+ ...(outputTokens === undefined ? {} : { outputTokens }),
185
+ ...(totalTokens === undefined ? {} : { totalTokens }),
186
+ ...(nonEmptyDetails({ noCacheTokens, cacheReadTokens, cacheWriteTokens }) === undefined
187
+ ? {}
188
+ : { inputTokenDetails: nonEmptyDetails({ noCacheTokens, cacheReadTokens, cacheWriteTokens })! }),
189
+ ...(nonEmptyDetails({ textTokens, reasoningTokens: normalizedReasoning }) === undefined
190
+ ? {}
191
+ : { outputTokenDetails: nonEmptyDetails({ textTokens, reasoningTokens: normalizedReasoning })! }),
44
192
  };
193
+ if (Object.keys(usage).length === 0) return undefined;
194
+ return validateProviderUsage(usage);
45
195
  };
46
196
 
47
- export const normalizeUsage = (raw: RawUsage | null | undefined, reasoningText = "", contentText = ""): ProviderUsage => {
48
- const prompt = raw?.prompt_tokens ?? 0;
49
- const completionRaw = raw?.completion_tokens ?? 0;
50
- const reportedTotal = raw?.total_tokens ?? 0;
51
- // OpenAI nests cached under prompt_tokens_details; others put it top-level.
52
- const cached = raw?.prompt_tokens_details?.cached_tokens
53
- ?? raw?.prompt_cache_hit_tokens
54
- ?? raw?.cached_tokens
55
- ?? 0;
56
- const reasoningDetail = raw?.completion_tokens_details?.reasoning_tokens;
57
-
58
- let completion: number;
59
- let reasoning: number;
60
- if (reasoningDetail !== undefined) {
61
- reasoning = reasoningDetail;
62
- // reasoning_tokens is reported two incompatible ways when detailed. OpenAI
63
- // o-series folds it INTO completion_tokens (subset: total = prompt + completion,
64
- // so completion must have reasoning subtracted out). xAI/Grok reports it
65
- // ADDITIVE to a visible-only completion_tokens (total = prompt + completion +
66
- // reasoning), where subtracting wrongly zeroes the visible output. Tell
67
- // them apart by the total identity; with no total reported, fall back on the
68
- // impossible-subset signal — reasoning can't exceed the completion it's a
69
- // subset of.
70
- const additive = reportedTotal > 0
71
- ? reportedTotal === prompt + completionRaw + reasoningDetail
72
- : completionRaw < reasoningDetail;
73
- completion = additive ? completionRaw : Math.max(0, completionRaw - reasoningDetail);
74
- } else {
75
- // Gemini-style (or no reasoning): tokens beyond prompt+completion are
76
- // reasoning. Only trust the gap when a total was actually reported.
77
- reasoning = reportedTotal > 0 ? Math.max(0, reportedTotal - prompt - completionRaw) : 0;
78
- completion = completionRaw;
197
+ // Models.dev rates are USD per million tokens. Optional cache rates inherit the
198
+ // input rate, so a missing usage detail matters only when its rate differs.
199
+ export type TokenRates = {
200
+ input: number;
201
+ output: number;
202
+ cacheRead?: number;
203
+ cacheWrite?: number;
204
+ };
205
+
206
+ const decimalParts = (value: number): { coefficient: bigint; scale: number } => {
207
+ if (!Number.isFinite(value) || value < 0) {
208
+ throw new TypeError("token rates must be finite non-negative numbers");
79
209
  }
80
- const total = reportedTotal > 0 ? reportedTotal : prompt + completion + reasoning;
81
- const usage = { prompt, completion, reasoning, cached, total };
82
- // Fireworks folds reasoning INTO completion_tokens and itemizes no
83
- // reasoning_tokens. A reported total establishes that completion is an
84
- // upstream quantity rather than a locally synthesized fallback.
85
- return reasoningDetail === undefined && reportedTotal > 0
86
- ? attributeUnitemizedReasoning(usage, reasoningText, contentText)
87
- : usage;
210
+ const match = /^(\d+)(?:\.(\d+))?(?:e([+-]?\d+))?$/i.exec(String(value));
211
+ if (match === null) throw new TypeError(`cannot represent token rate ${value} as a decimal`);
212
+ const fraction = match[2] ?? "";
213
+ const exponent = Number(match[3] ?? "0");
214
+ const scale = fraction.length - exponent;
215
+ const coefficient = BigInt(`${match[1]}${fraction}`);
216
+ return scale < 0
217
+ ? { coefficient: coefficient * 10n ** BigInt(-scale), scale: 0 }
218
+ : { coefficient, scale };
219
+ };
220
+
221
+ export const canonicalDecimal = (coefficient: bigint, scale: number): string => {
222
+ const digits = String(coefficient).padStart(scale + 1, "0");
223
+ if (scale === 0) return digits;
224
+ const integer = digits.slice(0, -scale);
225
+ const fraction = digits.slice(-scale).replace(/0+$/, "");
226
+ return fraction.length === 0 ? integer : `${integer}.${fraction}`;
88
227
  };
89
228
 
90
- // Conventional provider pricing: USD per million tokens, matching Models.dev.
91
- export type TokenRates = { input: number; output: number; cached: number };
92
-
93
- // The one cost formula every provider uses: non-cached prompt at the input
94
- // rate, cached prompt at the cache rate, and billable output (completion +
95
- // reasoning) at the output rate.
96
- export const calculateCostUsd = (usage: ProviderUsage, rates: TokenRates): number => {
97
- const nonCachedPrompt = Math.max(0, usage.prompt - usage.cached);
98
- const output = usage.completion + usage.reasoning;
99
- return (
100
- nonCachedPrompt * rates.input
101
- + usage.cached * rates.cached
102
- + output * rates.output
103
- ) / 1_000_000;
229
+ export const calculateCostUsdDecimal = (
230
+ usage: ProviderUsage,
231
+ rates: TokenRates,
232
+ ): string | null => {
233
+ validateProviderUsage(usage);
234
+ if (usage.inputTokens === undefined || usage.outputTokens === undefined) return null;
235
+
236
+ const cacheReadRate = rates.cacheRead ?? rates.input;
237
+ const cacheWriteRate = rates.cacheWrite ?? rates.input;
238
+ const cacheReadTokens = usage.inputTokenDetails?.cacheReadTokens;
239
+ const cacheWriteTokens = usage.inputTokenDetails?.cacheWriteTokens;
240
+ if (cacheReadRate !== rates.input && cacheReadTokens === undefined) return null;
241
+ if (cacheWriteRate !== rates.input && cacheWriteTokens === undefined) return null;
242
+
243
+ const parts = [rates.input, rates.output, cacheReadRate, cacheWriteRate].map(decimalParts);
244
+ const rateScale = Math.max(...parts.map(({ scale }) => scale));
245
+ const [input, output, cacheRead, cacheWrite] = parts.map(({ coefficient, scale }) =>
246
+ coefficient * 10n ** BigInt(rateScale - scale));
247
+ const coefficient = BigInt(usage.inputTokens) * input!
248
+ + BigInt(usage.outputTokens) * output!
249
+ + BigInt(cacheReadTokens ?? 0) * (cacheRead! - input!)
250
+ + BigInt(cacheWriteTokens ?? 0) * (cacheWrite! - input!);
251
+ if (coefficient < 0n) throw new TypeError("provider token-rate details produced a negative estimate");
252
+ return canonicalDecimal(coefficient, rateScale + 6);
104
253
  };