@plurnk/plurnk-providers 1.5.0 → 1.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/.env.defaults +41 -34
  2. package/README.md +15 -0
  3. package/SPEC.md +242 -89
  4. package/dist/AiSdkProvider.d.ts +33 -33
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +442 -133
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +10 -11
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +87 -25
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +9 -24
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +86 -25
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/accounting.d.ts +5 -2
  17. package/dist/accounting.d.ts.map +1 -1
  18. package/dist/accounting.js +100 -16
  19. package/dist/accounting.js.map +1 -1
  20. package/dist/accountingPublic.d.ts +5 -0
  21. package/dist/accountingPublic.d.ts.map +1 -0
  22. package/dist/accountingPublic.js +3 -0
  23. package/dist/accountingPublic.js.map +1 -0
  24. package/dist/aiSdkTransport.d.ts +9 -2
  25. package/dist/aiSdkTransport.d.ts.map +1 -1
  26. package/dist/aiSdkTransport.js +160 -62
  27. package/dist/aiSdkTransport.js.map +1 -1
  28. package/dist/capacity.d.ts +26 -0
  29. package/dist/capacity.d.ts.map +1 -0
  30. package/dist/capacity.js +90 -0
  31. package/dist/capacity.js.map +1 -0
  32. package/dist/catalogProvider.d.ts +8 -3
  33. package/dist/catalogProvider.d.ts.map +1 -1
  34. package/dist/catalogProvider.js +45 -41
  35. package/dist/catalogProvider.js.map +1 -1
  36. package/dist/compatibleProvider.d.ts.map +1 -1
  37. package/dist/compatibleProvider.js +26 -12
  38. package/dist/compatibleProvider.js.map +1 -1
  39. package/dist/cost.d.ts +10 -10
  40. package/dist/cost.d.ts.map +1 -1
  41. package/dist/cost.js +90 -42
  42. package/dist/cost.js.map +1 -1
  43. package/dist/env.d.ts +13 -11
  44. package/dist/env.d.ts.map +1 -1
  45. package/dist/env.js +83 -46
  46. package/dist/env.js.map +1 -1
  47. package/dist/errors.d.ts +17 -3
  48. package/dist/errors.d.ts.map +1 -1
  49. package/dist/errors.js +91 -8
  50. package/dist/errors.js.map +1 -1
  51. package/dist/index.d.ts +7 -6
  52. package/dist/index.d.ts.map +1 -1
  53. package/dist/index.js +5 -3
  54. package/dist/index.js.map +1 -1
  55. package/dist/ollama.js +3 -3
  56. package/dist/ollama.js.map +1 -1
  57. package/dist/promptTokens.d.ts.map +1 -1
  58. package/dist/promptTokens.js +7 -4
  59. package/dist/promptTokens.js.map +1 -1
  60. package/dist/sdkModels.d.ts +7 -2
  61. package/dist/sdkModels.d.ts.map +1 -1
  62. package/dist/sdkModels.js +43 -13
  63. package/dist/sdkModels.js.map +1 -1
  64. package/dist/types.d.ts +55 -33
  65. package/dist/types.d.ts.map +1 -1
  66. package/dist/usage.d.ts +22 -5
  67. package/dist/usage.d.ts.map +1 -1
  68. package/dist/usage.js +169 -83
  69. package/dist/usage.js.map +1 -1
  70. package/package.json +18 -7
  71. package/src/AiSdkProvider.test.ts +964 -206
  72. package/src/AiSdkProvider.ts +545 -155
  73. package/src/Mock.test.ts +69 -30
  74. package/src/Mock.ts +99 -29
  75. package/src/Pool.test.ts +90 -19
  76. package/src/Pool.ts +96 -27
  77. package/src/ProviderRegistry.test.ts +16 -11
  78. package/src/accounting.test.ts +58 -22
  79. package/src/accounting.ts +119 -18
  80. package/src/accountingPublic.ts +9 -0
  81. package/src/aiSdkTransport.test.ts +42 -49
  82. package/src/aiSdkTransport.ts +174 -62
  83. package/src/boundaries.test.ts +2 -0
  84. package/src/capacity.test.ts +92 -0
  85. package/src/capacity.ts +140 -0
  86. package/src/catalogProvider.test.ts +339 -30
  87. package/src/catalogProvider.ts +65 -47
  88. package/src/compatibleProvider.test.ts +7 -5
  89. package/src/compatibleProvider.ts +29 -13
  90. package/src/cost.test.ts +86 -36
  91. package/src/cost.ts +111 -50
  92. package/src/defaults.test.ts +13 -3
  93. package/src/env.test.ts +103 -25
  94. package/src/env.ts +153 -65
  95. package/src/errors.test.ts +80 -2
  96. package/src/errors.ts +107 -8
  97. package/src/index.ts +26 -7
  98. package/src/ollama.test.ts +5 -3
  99. package/src/ollama.ts +3 -3
  100. package/src/promptTokens.ts +8 -5
  101. package/src/sdkModels.test.ts +77 -8
  102. package/src/sdkModels.ts +51 -15
  103. package/src/types.ts +112 -51
  104. package/src/usage.test.ts +112 -116
  105. package/src/usage.ts +214 -93
package/src/usage.ts CHANGED
@@ -1,94 +1,208 @@
1
- // Usage normalization + cost — the shared token-accounting model.
2
- //
3
- // Providers report token usage in two incompatible ways:
4
- // - OpenAI-style: reasoning is a SUBSET of completion_tokens, surfaced via
5
- // completion_tokens_details.reasoning_tokens; total = prompt + completion.
6
- // - Gemini-style: reasoning is OMITTED from completion_tokens and only
7
- // recoverable as total - prompt - completion (no details field at all).
8
- // - Fireworks-style: reasoning ships as TEXT (reasoning_content) but is folded
9
- // into completion_tokens with NO reasoning_tokens itemization -- unrecoverable
10
- // from the numbers alone, so it is re-split from the emitted text lengths.
11
- // normalizeUsage collapses all three into one invariant (see ProviderUsage):
12
- // total = prompt + completion + reasoning; cached ⊆ prompt;
13
- // completion EXCLUDES reasoning; billable output = completion + reasoning.
1
+ // Provider usage normalization. The public shape follows {§provider-usage}:
2
+ // input/output totals own their optional cache/reasoning details, absence is
3
+ // unknown, and no token category is inferred from text length.
14
4
 
15
5
  import type { ProviderUsage } from "./types.ts";
16
6
 
17
- // Raw OpenAI-compatible usage block — the superset of fields providers emit.
18
7
  export type RawUsage = {
8
+ input_tokens?: number;
9
+ output_tokens?: number;
19
10
  prompt_tokens?: number;
20
11
  completion_tokens?: number;
21
12
  total_tokens?: number;
22
13
  cached_tokens?: number;
14
+ cache_creation_input_tokens?: number;
15
+ cache_read_input_tokens?: number;
23
16
  prompt_cache_hit_tokens?: number;
24
17
  prompt_cache_miss_tokens?: number;
25
- prompt_tokens_details?: { cached_tokens?: number };
18
+ prompt_tokens_details?: {
19
+ cached_tokens?: number;
20
+ cache_read_tokens?: number;
21
+ cache_write_tokens?: number;
22
+ };
23
+ input_tokens_details?: {
24
+ cached_tokens?: number;
25
+ cache_read_tokens?: number;
26
+ cache_write_tokens?: number;
27
+ };
26
28
  completion_tokens_details?: { reasoning_tokens?: number };
29
+ output_tokens_details?: { reasoning_tokens?: number };
30
+ reasoning_tokens?: number;
27
31
  };
28
32
 
29
- // Some providers return distinct reasoning text while reporting one combined
30
- // output count. Attribute that count without disturbing an upstream split.
31
- export const attributeUnitemizedReasoning = (
32
- usage: ProviderUsage,
33
- reasoningText: string,
34
- contentText: string,
35
- ): ProviderUsage => {
36
- if (usage.reasoning !== 0 || usage.completion === 0 || reasoningText.length === 0) return usage;
37
- const reasoning = contentText.length === 0
38
- ? usage.completion
39
- : Math.round(usage.completion * reasoningText.length / (reasoningText.length + contentText.length));
40
- return {
41
- ...usage,
42
- completion: usage.completion - reasoning,
43
- reasoning,
44
- };
33
+ const knownTokens = (value: unknown, name: string): number | undefined => {
34
+ if (value === undefined) return undefined;
35
+ if (!Number.isSafeInteger(value) || (value as number) < 0) {
36
+ throw new TypeError(`${name} must be a non-negative safe integer`);
37
+ }
38
+ return value as number;
39
+ };
40
+
41
+ const nonEmptyDetails = <T extends Record<string, number | undefined>>(details: T): T | undefined => {
42
+ const entries = Object.entries(details).filter((entry): entry is [string, number] =>
43
+ entry[1] !== undefined);
44
+ return entries.length === 0 ? undefined : Object.fromEntries(entries) as T;
45
45
  };
46
46
 
47
- export const normalizeUsage = (raw: RawUsage | null | undefined, reasoningText = "", contentText = ""): ProviderUsage => {
48
- const prompt = raw?.prompt_tokens ?? 0;
49
- const completionRaw = raw?.completion_tokens ?? 0;
50
- const reportedTotal = raw?.total_tokens ?? 0;
51
- // OpenAI nests cached under prompt_tokens_details; others put it top-level.
52
- const cached = raw?.prompt_tokens_details?.cached_tokens
53
- ?? raw?.prompt_cache_hit_tokens
54
- ?? raw?.cached_tokens
55
- ?? 0;
56
- const reasoningDetail = raw?.completion_tokens_details?.reasoning_tokens;
57
-
58
- let completion: number;
59
- let reasoning: number;
60
- if (reasoningDetail !== undefined) {
61
- reasoning = reasoningDetail;
62
- // reasoning_tokens is reported two incompatible ways when detailed. OpenAI
63
- // o-series folds it INTO completion_tokens (subset: total = prompt + completion,
64
- // so completion must have reasoning subtracted out). xAI/Grok reports it
65
- // ADDITIVE to a visible-only completion_tokens (total = prompt + completion +
66
- // reasoning), where subtracting wrongly zeroes the visible output. Tell
67
- // them apart by the total identity; with no total reported, fall back on the
68
- // impossible-subset signal — reasoning can't exceed the completion it's a
69
- // subset of.
70
- const additive = reportedTotal > 0
71
- ? reportedTotal === prompt + completionRaw + reasoningDetail
72
- : completionRaw < reasoningDetail;
73
- completion = additive ? completionRaw : Math.max(0, completionRaw - reasoningDetail);
74
- } else {
75
- // Gemini-style (or no reasoning): tokens beyond prompt+completion are
76
- // reasoning. Only trust the gap when a total was actually reported.
77
- reasoning = reportedTotal > 0 ? Math.max(0, reportedTotal - prompt - completionRaw) : 0;
78
- completion = completionRaw;
47
+ export const validateProviderUsage = (usage: ProviderUsage): ProviderUsage => {
48
+ const input = knownTokens(usage.inputTokens, "provider usage.inputTokens");
49
+ const output = knownTokens(usage.outputTokens, "provider usage.outputTokens");
50
+ const total = knownTokens(usage.totalTokens, "provider usage.totalTokens");
51
+ const inputDetails = usage.inputTokenDetails;
52
+ const outputDetails = usage.outputTokenDetails;
53
+
54
+ if (input === undefined && output === undefined && total === undefined
55
+ && inputDetails === undefined && outputDetails === undefined) {
56
+ throw new TypeError("provider usage must contain at least one known quantity");
57
+ }
58
+ if (input !== undefined && output !== undefined && total !== undefined
59
+ && total !== input + output) {
60
+ throw new TypeError("provider usage.totalTokens must equal inputTokens + outputTokens");
61
+ }
62
+
63
+ if (inputDetails !== undefined) {
64
+ const values = [
65
+ knownTokens(inputDetails.noCacheTokens, "provider usage.inputTokenDetails.noCacheTokens"),
66
+ knownTokens(inputDetails.cacheReadTokens, "provider usage.inputTokenDetails.cacheReadTokens"),
67
+ knownTokens(inputDetails.cacheWriteTokens, "provider usage.inputTokenDetails.cacheWriteTokens"),
68
+ ];
69
+ if (values.every((value) => value === undefined)) {
70
+ throw new TypeError("provider usage.inputTokenDetails must contain a known quantity");
71
+ }
72
+ const known = values.filter((value): value is number => value !== undefined);
73
+ if (input !== undefined && known.some((value) => value > input)) {
74
+ throw new TypeError("provider input-token detail must not exceed inputTokens");
75
+ }
76
+ if (input !== undefined && values.every((value) => value !== undefined)
77
+ && known.reduce((sum, value) => sum + value, 0) !== input) {
78
+ throw new TypeError("complete provider input-token details must sum to inputTokens");
79
+ }
79
80
  }
80
- const total = reportedTotal > 0 ? reportedTotal : prompt + completion + reasoning;
81
- const usage = { prompt, completion, reasoning, cached, total };
82
- // Fireworks folds reasoning INTO completion_tokens and itemizes no
83
- // reasoning_tokens. A reported total establishes that completion is an
84
- // upstream quantity rather than a locally synthesized fallback.
85
- return reasoningDetail === undefined && reportedTotal > 0
86
- ? attributeUnitemizedReasoning(usage, reasoningText, contentText)
87
- : usage;
81
+
82
+ if (outputDetails !== undefined) {
83
+ const values = [
84
+ knownTokens(outputDetails.textTokens, "provider usage.outputTokenDetails.textTokens"),
85
+ knownTokens(outputDetails.reasoningTokens, "provider usage.outputTokenDetails.reasoningTokens"),
86
+ ];
87
+ if (values.every((value) => value === undefined)) {
88
+ throw new TypeError("provider usage.outputTokenDetails must contain a known quantity");
89
+ }
90
+ const known = values.filter((value): value is number => value !== undefined);
91
+ if (output !== undefined && known.some((value) => value > output)) {
92
+ throw new TypeError("provider output-token detail must not exceed outputTokens");
93
+ }
94
+ if (output !== undefined && values.every((value) => value !== undefined)
95
+ && known.reduce((sum, value) => sum + value, 0) !== output) {
96
+ throw new TypeError("complete provider output-token details must sum to outputTokens");
97
+ }
98
+ }
99
+ return usage;
88
100
  };
89
101
 
90
- // Conventional provider pricing: USD per million tokens, matching Models.dev.
91
- export type TokenRates = { input: number; output: number; cached: number };
102
+ export const normalizeUsage = (raw: RawUsage | null | undefined): ProviderUsage | undefined => {
103
+ if (raw === null || raw === undefined) return undefined;
104
+
105
+ const inputTokens = knownTokens(
106
+ raw.input_tokens ?? raw.prompt_tokens,
107
+ "provider usage input tokens",
108
+ );
109
+ const rawOutputTokens = knownTokens(
110
+ raw.output_tokens ?? raw.completion_tokens,
111
+ "provider usage output tokens",
112
+ );
113
+ const reportedTotal = knownTokens(raw.total_tokens, "provider usage total tokens");
114
+ const reasoningTokens = knownTokens(
115
+ raw.output_tokens_details?.reasoning_tokens
116
+ ?? raw.completion_tokens_details?.reasoning_tokens
117
+ ?? raw.reasoning_tokens,
118
+ "provider usage reasoning tokens",
119
+ );
120
+
121
+ let outputTokens = rawOutputTokens;
122
+ let textTokens: number | undefined;
123
+ let normalizedReasoning = reasoningTokens;
124
+ if (inputTokens !== undefined && rawOutputTokens !== undefined && reportedTotal !== undefined) {
125
+ if (reportedTotal === inputTokens + rawOutputTokens) {
126
+ outputTokens = rawOutputTokens;
127
+ if (reasoningTokens !== undefined) textTokens = rawOutputTokens - reasoningTokens;
128
+ } else if (reasoningTokens !== undefined
129
+ && reportedTotal === inputTokens + rawOutputTokens + reasoningTokens) {
130
+ outputTokens = rawOutputTokens + reasoningTokens;
131
+ textTokens = rawOutputTokens;
132
+ } else if (reportedTotal >= inputTokens + rawOutputTokens) {
133
+ outputTokens = reportedTotal - inputTokens;
134
+ normalizedReasoning = outputTokens - rawOutputTokens;
135
+ textTokens = rawOutputTokens;
136
+ } else {
137
+ throw new TypeError("provider usage total is inconsistent with input and output tokens");
138
+ }
139
+ } else if (rawOutputTokens !== undefined && reasoningTokens !== undefined) {
140
+ if (reasoningTokens > rawOutputTokens) {
141
+ outputTokens = rawOutputTokens + reasoningTokens;
142
+ textTokens = rawOutputTokens;
143
+ } else {
144
+ textTokens = rawOutputTokens - reasoningTokens;
145
+ }
146
+ } else if (outputTokens === undefined && inputTokens !== undefined && reportedTotal !== undefined) {
147
+ if (reportedTotal < inputTokens) {
148
+ throw new TypeError("provider usage total must not be less than input tokens");
149
+ }
150
+ outputTokens = reportedTotal - inputTokens;
151
+ }
152
+
153
+ const totalTokens = reportedTotal
154
+ ?? (inputTokens !== undefined && outputTokens !== undefined
155
+ ? inputTokens + outputTokens
156
+ : undefined);
157
+ const cacheReadTokens = knownTokens(
158
+ raw.input_tokens_details?.cache_read_tokens
159
+ ?? raw.input_tokens_details?.cached_tokens
160
+ ?? raw.prompt_tokens_details?.cache_read_tokens
161
+ ?? raw.prompt_tokens_details?.cached_tokens
162
+ ?? raw.cache_read_input_tokens
163
+ ?? raw.prompt_cache_hit_tokens
164
+ ?? raw.cached_tokens,
165
+ "provider usage cache-read tokens",
166
+ );
167
+ const cacheWriteTokens = knownTokens(
168
+ raw.input_tokens_details?.cache_write_tokens
169
+ ?? raw.prompt_tokens_details?.cache_write_tokens
170
+ ?? raw.cache_creation_input_tokens,
171
+ "provider usage cache-write tokens",
172
+ );
173
+ const explicitNoCacheTokens = knownTokens(
174
+ raw.prompt_cache_miss_tokens,
175
+ "provider usage non-cache tokens",
176
+ );
177
+ const noCacheTokens = explicitNoCacheTokens
178
+ ?? (inputTokens !== undefined && cacheReadTokens !== undefined && cacheWriteTokens !== undefined
179
+ ? inputTokens - cacheReadTokens - cacheWriteTokens
180
+ : undefined);
181
+
182
+ const usage: ProviderUsage = {
183
+ ...(inputTokens === undefined ? {} : { inputTokens }),
184
+ ...(outputTokens === undefined ? {} : { outputTokens }),
185
+ ...(totalTokens === undefined ? {} : { totalTokens }),
186
+ ...(nonEmptyDetails({ noCacheTokens, cacheReadTokens, cacheWriteTokens }) === undefined
187
+ ? {}
188
+ : { inputTokenDetails: nonEmptyDetails({ noCacheTokens, cacheReadTokens, cacheWriteTokens })! }),
189
+ ...(nonEmptyDetails({ textTokens, reasoningTokens: normalizedReasoning }) === undefined
190
+ ? {}
191
+ : { outputTokenDetails: nonEmptyDetails({ textTokens, reasoningTokens: normalizedReasoning })! }),
192
+ };
193
+ if (Object.keys(usage).length === 0) return undefined;
194
+ return validateProviderUsage(usage);
195
+ };
196
+
197
+ // Models.dev rates are USD per million tokens. Optional cache rates inherit the
198
+ // input rate, so a missing usage detail matters only when its rate differs.
199
+ export type TokenRates = {
200
+ input: number;
201
+ output: number;
202
+ reasoning?: number;
203
+ cacheRead?: number;
204
+ cacheWrite?: number;
205
+ };
92
206
 
93
207
  const decimalParts = (value: number): { coefficient: bigint; scale: number } => {
94
208
  if (!Number.isFinite(value) || value < 0) {
@@ -105,7 +219,7 @@ const decimalParts = (value: number): { coefficient: bigint; scale: number } =>
105
219
  : { coefficient, scale };
106
220
  };
107
221
 
108
- const decimalString = (coefficient: bigint, scale: number): string => {
222
+ export const canonicalDecimal = (coefficient: bigint, scale: number): string => {
109
223
  const digits = String(coefficient).padStart(scale + 1, "0");
110
224
  if (scale === 0) return digits;
111
225
  const integer = digits.slice(0, -scale);
@@ -113,25 +227,32 @@ const decimalString = (coefficient: bigint, scale: number): string => {
113
227
  return fraction.length === 0 ? integer : `${integer}.${fraction}`;
114
228
  };
115
229
 
116
- // Models.dev rates are decimal USD-per-million values. Calculate their
117
- // projection as decimal arithmetic so ProviderCost preserves the table and
118
- // exact response counts without binary floating-point artifacts.
119
- export const calculateCostUsdDecimal = (usage: ProviderUsage, rates: TokenRates): string => {
120
- const parts = [rates.input, rates.cached, rates.output].map(decimalParts);
230
+ export const calculateCostUsdDecimal = (
231
+ usage: ProviderUsage,
232
+ rates: TokenRates,
233
+ ): string | null => {
234
+ validateProviderUsage(usage);
235
+ if (usage.inputTokens === undefined || usage.outputTokens === undefined) return null;
236
+
237
+ const cacheReadRate = rates.cacheRead ?? rates.input;
238
+ const cacheWriteRate = rates.cacheWrite ?? rates.input;
239
+ const reasoningRate = rates.reasoning ?? rates.output;
240
+ const cacheReadTokens = usage.inputTokenDetails?.cacheReadTokens;
241
+ const cacheWriteTokens = usage.inputTokenDetails?.cacheWriteTokens;
242
+ const reasoningTokens = usage.outputTokenDetails?.reasoningTokens;
243
+ if (cacheReadRate !== rates.input && cacheReadTokens === undefined) return null;
244
+ if (cacheWriteRate !== rates.input && cacheWriteTokens === undefined) return null;
245
+ if (reasoningRate !== rates.output && reasoningTokens === undefined) return null;
246
+
247
+ const parts = [rates.input, rates.output, reasoningRate, cacheReadRate, cacheWriteRate].map(decimalParts);
121
248
  const rateScale = Math.max(...parts.map(({ scale }) => scale));
122
- const [input, cached, output] = parts.map(({ coefficient, scale }) =>
249
+ const [input, output, reasoning, cacheRead, cacheWrite] = parts.map(({ coefficient, scale }) =>
123
250
  coefficient * 10n ** BigInt(rateScale - scale));
124
- const nonCachedPrompt = Math.max(0, usage.prompt - usage.cached);
125
- const outputTokens = usage.completion + usage.reasoning;
126
- const coefficient = BigInt(nonCachedPrompt) * input!
127
- + BigInt(usage.cached) * cached!
128
- + BigInt(outputTokens) * output!;
129
- return decimalString(coefficient, rateScale + 6);
130
- };
131
-
132
- // The one cost formula every provider uses: non-cached prompt at the input
133
- // rate, cached prompt at the cache rate, and billable output (completion +
134
- // reasoning) at the output rate.
135
- export const calculateCostUsd = (usage: ProviderUsage, rates: TokenRates): number => {
136
- return Number(calculateCostUsdDecimal(usage, rates));
251
+ const coefficient = BigInt(usage.inputTokens) * input!
252
+ + BigInt(usage.outputTokens) * output!
253
+ + BigInt(reasoningTokens ?? 0) * (reasoning! - output!)
254
+ + BigInt(cacheReadTokens ?? 0) * (cacheRead! - input!)
255
+ + BigInt(cacheWriteTokens ?? 0) * (cacheWrite! - input!);
256
+ if (coefficient < 0n) throw new TypeError("provider token-rate details produced a negative estimate");
257
+ return canonicalDecimal(coefficient, rateScale + 6);
137
258
  };