@plurnk/plurnk-providers 1.5.0 → 1.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/.env.defaults +41 -34
  2. package/README.md +15 -0
  3. package/SPEC.md +242 -89
  4. package/dist/AiSdkProvider.d.ts +33 -33
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +442 -133
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +10 -11
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +87 -25
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +9 -24
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +86 -25
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/accounting.d.ts +5 -2
  17. package/dist/accounting.d.ts.map +1 -1
  18. package/dist/accounting.js +100 -16
  19. package/dist/accounting.js.map +1 -1
  20. package/dist/accountingPublic.d.ts +5 -0
  21. package/dist/accountingPublic.d.ts.map +1 -0
  22. package/dist/accountingPublic.js +3 -0
  23. package/dist/accountingPublic.js.map +1 -0
  24. package/dist/aiSdkTransport.d.ts +9 -2
  25. package/dist/aiSdkTransport.d.ts.map +1 -1
  26. package/dist/aiSdkTransport.js +160 -62
  27. package/dist/aiSdkTransport.js.map +1 -1
  28. package/dist/capacity.d.ts +26 -0
  29. package/dist/capacity.d.ts.map +1 -0
  30. package/dist/capacity.js +90 -0
  31. package/dist/capacity.js.map +1 -0
  32. package/dist/catalogProvider.d.ts +8 -3
  33. package/dist/catalogProvider.d.ts.map +1 -1
  34. package/dist/catalogProvider.js +45 -41
  35. package/dist/catalogProvider.js.map +1 -1
  36. package/dist/compatibleProvider.d.ts.map +1 -1
  37. package/dist/compatibleProvider.js +26 -12
  38. package/dist/compatibleProvider.js.map +1 -1
  39. package/dist/cost.d.ts +10 -10
  40. package/dist/cost.d.ts.map +1 -1
  41. package/dist/cost.js +90 -42
  42. package/dist/cost.js.map +1 -1
  43. package/dist/env.d.ts +13 -11
  44. package/dist/env.d.ts.map +1 -1
  45. package/dist/env.js +83 -46
  46. package/dist/env.js.map +1 -1
  47. package/dist/errors.d.ts +17 -3
  48. package/dist/errors.d.ts.map +1 -1
  49. package/dist/errors.js +91 -8
  50. package/dist/errors.js.map +1 -1
  51. package/dist/index.d.ts +7 -6
  52. package/dist/index.d.ts.map +1 -1
  53. package/dist/index.js +5 -3
  54. package/dist/index.js.map +1 -1
  55. package/dist/ollama.js +3 -3
  56. package/dist/ollama.js.map +1 -1
  57. package/dist/promptTokens.d.ts.map +1 -1
  58. package/dist/promptTokens.js +7 -4
  59. package/dist/promptTokens.js.map +1 -1
  60. package/dist/sdkModels.d.ts +7 -2
  61. package/dist/sdkModels.d.ts.map +1 -1
  62. package/dist/sdkModels.js +43 -13
  63. package/dist/sdkModels.js.map +1 -1
  64. package/dist/types.d.ts +55 -33
  65. package/dist/types.d.ts.map +1 -1
  66. package/dist/usage.d.ts +22 -5
  67. package/dist/usage.d.ts.map +1 -1
  68. package/dist/usage.js +169 -83
  69. package/dist/usage.js.map +1 -1
  70. package/package.json +18 -7
  71. package/src/AiSdkProvider.test.ts +964 -206
  72. package/src/AiSdkProvider.ts +545 -155
  73. package/src/Mock.test.ts +69 -30
  74. package/src/Mock.ts +99 -29
  75. package/src/Pool.test.ts +90 -19
  76. package/src/Pool.ts +96 -27
  77. package/src/ProviderRegistry.test.ts +16 -11
  78. package/src/accounting.test.ts +58 -22
  79. package/src/accounting.ts +119 -18
  80. package/src/accountingPublic.ts +9 -0
  81. package/src/aiSdkTransport.test.ts +42 -49
  82. package/src/aiSdkTransport.ts +174 -62
  83. package/src/boundaries.test.ts +2 -0
  84. package/src/capacity.test.ts +92 -0
  85. package/src/capacity.ts +140 -0
  86. package/src/catalogProvider.test.ts +339 -30
  87. package/src/catalogProvider.ts +65 -47
  88. package/src/compatibleProvider.test.ts +7 -5
  89. package/src/compatibleProvider.ts +29 -13
  90. package/src/cost.test.ts +86 -36
  91. package/src/cost.ts +111 -50
  92. package/src/defaults.test.ts +13 -3
  93. package/src/env.test.ts +103 -25
  94. package/src/env.ts +153 -65
  95. package/src/errors.test.ts +80 -2
  96. package/src/errors.ts +107 -8
  97. package/src/index.ts +26 -7
  98. package/src/ollama.test.ts +5 -3
  99. package/src/ollama.ts +3 -3
  100. package/src/promptTokens.ts +8 -5
  101. package/src/sdkModels.test.ts +77 -8
  102. package/src/sdkModels.ts +51 -15
  103. package/src/types.ts +112 -51
  104. package/src/usage.test.ts +112 -116
  105. package/src/usage.ts +214 -93
package/src/types.ts CHANGED
@@ -9,15 +9,30 @@ import type {
9
9
  PluginAttributionContext,
10
10
  PluginAttributionSource,
11
11
  } from "@plurnk/plurnk-meta";
12
- import type { ProviderCost } from "@plurnk/plurnk-contracts";
12
+ import type {
13
+ ProviderCost,
14
+ ProviderRequestAccounting,
15
+ } from "@plurnk/plurnk-contracts";
16
+
17
+ export type {
18
+ ProviderAccounting,
19
+ ProviderCost,
20
+ ProviderRequestAccounting,
21
+ ProviderUsage,
22
+ } from "@plurnk/plurnk-contracts";
13
23
 
14
24
  export interface ChatMessage {
15
25
  role: "system" | "user" | "assistant";
16
26
  content: string;
17
27
  }
18
28
 
29
+ // {§provider-call-kind} The caller-owned semantic contract for one logical model call. Providers do
30
+ // not infer this from messages or grammar presence: an emission expects PLURNK
31
+ // output, while a bare call expects unconstrained response text.
32
+ export type ProviderCallKind = "emission" | "bare";
33
+
19
34
  // Preflight evidence for the complete provider request. An empirical estimate
20
- // is useful telemetry but cannot authorize a hard physical-capacity decision.
35
+ // is useful telemetry but cannot authorize hard context-envelope admission.
21
36
  export type PromptTokenMeasurement =
22
37
  | {
23
38
  readonly kind: "exact" | "upper_bound";
@@ -29,41 +44,67 @@ export type PromptTokenMeasurement =
29
44
  readonly tokens: number;
30
45
  readonly source: string;
31
46
  readonly detail: string;
47
+ }
48
+ | {
49
+ readonly kind: "unavailable";
50
+ readonly source: string;
51
+ readonly detail: string;
32
52
  };
33
53
 
34
- // Normalized token accounting. Invariant (enforced by normalizeUsage at the
35
- // provider boundary): total = prompt + completion + reasoning; cached is a
36
- // subset of prompt. `completion` is visible output EXCLUDING reasoning; the
37
- // billable output is `completion + reasoning` (frontier providers bill reasoning
38
- // tokens at the output rate).
39
- export interface ProviderUsage {
40
- readonly prompt: number; // input tokens (cached ones included)
41
- readonly completion: number; // visible output tokens, excluding reasoning
42
- readonly reasoning: number; // reasoning tokens, billed as output
43
- readonly cached: number; // subset of prompt served from cache
44
- readonly total: number; // prompt + completion + reasoning
54
+ export type ProviderRequestCapacityDecision = "admit" | "defer" | "reject";
55
+
56
+ // Complete pre-I/O evidence for one logical request. `defer` is intentional:
57
+ // an estimate or incomplete limit set cannot safely veto a request, so the
58
+ // upstream provider remains the capacity oracle.
59
+ export interface ProviderRequestCapacity {
60
+ readonly decision: ProviderRequestCapacityDecision;
61
+ readonly contextWindow: number | null;
62
+ readonly maxInputTokens: number | null;
63
+ readonly maxOutputTokens: number | null;
64
+ readonly outputBudget: number | null;
65
+ readonly reasoningBudget: number | null;
66
+ readonly inputCapacity: number | null;
67
+ readonly prompt: PromptTokenMeasurement;
45
68
  }
46
69
 
47
- export type AuthoritativeCharge = Extract<ProviderCost, { kind: "authoritative" }>;
70
+ export type ChargedCost = Extract<ProviderCost, { kind: "charged" }>;
48
71
 
49
72
  // Evidence exposed by the transport to the provider adapter that owns its
50
73
  // vendor protocol. Core and downstream consumers receive only the normalized
51
74
  // charge, never a requirement to understand provider metadata fields.
52
75
  export interface ProviderChargeEvidence {
53
76
  readonly providerMetadata?: unknown;
77
+ // A protocol-owned direct monetary field. It remains unknown until the
78
+ // selected adapter explicitly validates and normalizes it.
79
+ readonly charge?: unknown;
54
80
  // Provider-owned raw usage projection retained independently of optional
55
81
  // full-body capture. Accounting fields cannot disappear merely because
56
82
  // forensic raw-body capture is disabled.
57
83
  readonly usage?: unknown;
58
84
  readonly response: {
59
- readonly id: string;
85
+ readonly id?: string;
60
86
  readonly headers?: Readonly<Record<string, string>>;
61
87
  };
62
88
  }
63
89
 
64
- export type AuthoritativeChargeNormalizer = (
90
+ export type ProviderCostNormalizer = (
65
91
  evidence: ProviderChargeEvidence,
66
- ) => AuthoritativeCharge | undefined;
92
+ ) => ProviderCost | undefined;
93
+
94
+ export interface ProviderRequestIdentity {
95
+ readonly provider: string;
96
+ readonly model: string;
97
+ }
98
+
99
+ export type ProviderRequestSettlement = (
100
+ accounting: ProviderRequestAccounting,
101
+ ) => Promise<void>;
102
+
103
+ // Core opens durable physical-request identity through this observer before
104
+ // provider I/O. The returned settlement closes that exact identity.
105
+ export type ProviderRequestObserver = (
106
+ identity: ProviderRequestIdentity,
107
+ ) => Promise<ProviderRequestSettlement>;
67
108
 
68
109
  // A successful exchange's closed finish set. ProviderAttemptFinishReason adds
69
110
  // the failed disposition that may occur only on ProviderError attempt evidence.
@@ -102,7 +143,6 @@ export interface ProviderAssistant<TFinish extends ProviderAttemptFinishReason =
102
143
  readonly reasoning: string | null;
103
144
  // Encrypted reasoning remains distinct from readable `reasoning`.
104
145
  readonly reasoningEncrypted?: ReadonlyArray<ProviderEncryptedReasoningItem>;
105
- readonly usage: ProviderUsage;
106
146
  readonly finishReason: TFinish;
107
147
  readonly model: string;
108
148
  // Per-token logprobs, present only when PLURNK_PROVIDERS_TOP_LOGPROBS is set
@@ -114,8 +154,9 @@ export interface ProviderAssistant<TFinish extends ProviderAttemptFinishReason =
114
154
  }
115
155
 
116
156
  export interface GrammarEvidence {
117
- // Exact sentence observed at the grammar boundary before any reasoning/content
118
- // projection. Offsets are Unicode code points, matching @plurnk/gbnf verdicts.
157
+ // Exact pre-projection response represented by the provider. A generated
158
+ // rail's response root composes any template prefix before @plurnk/gbnf
159
+ // grades it. Offsets are Unicode code points, matching validator verdicts.
119
160
  readonly input: string;
120
161
  readonly contentStart: number;
121
162
  readonly transported: boolean;
@@ -124,10 +165,10 @@ export interface GrammarEvidence {
124
165
  export interface ProviderResponse<TFinish extends ProviderAttemptFinishReason = FinishReason> {
125
166
  readonly assistant: ProviderAssistant<TFinish>;
126
167
  readonly assistantRaw: unknown;
127
- // A settled upstream charge is a validated public fact, not opaque metadata.
128
- // Non-USD settlement carries an explicit provider-owned USD equivalent for
129
- // the platform's existing USD aggregate. Core never supplies an FX rate.
130
- readonly charge?: AuthoritativeCharge;
168
+ // Ordered physical request evidence, including automatic retries and pool
169
+ // failover that preceded this response. {§provider-request-accounting}
170
+ readonly accounting: readonly ProviderRequestAccounting[];
171
+ readonly capacity: ProviderRequestCapacity;
131
172
  // {§gbnf-response-observation} — evidence only; the consumer owns the verdict.
132
173
  readonly grammarEvidence?: GrammarEvidence;
133
174
  // Per-turn provider→client metadata bag: the backend's non-standard top-level
@@ -152,22 +193,38 @@ export interface ProviderResponse<TFinish extends ProviderAttemptFinishReason =
152
193
 
153
194
  export type ProviderAttempt = ProviderResponse<ProviderAttemptFinishReason>;
154
195
 
196
+ export interface ProviderGenerateArgs {
197
+ readonly messages: ChatMessage[];
198
+ readonly workerId: string;
199
+ readonly primaryWorkerId?: string;
200
+ readonly signal?: AbortSignal;
201
+ readonly grammar?: string;
202
+ readonly maxOutputTokens?: number;
203
+ readonly attributions?: string[];
204
+ readonly client?: string;
205
+ readonly strikes?: number;
206
+ readonly workspaceId?: string;
207
+ readonly loop?: number;
208
+ readonly turn?: number;
209
+ readonly sampling?: Record<string, unknown>;
210
+ readonly observeRequest?: ProviderRequestObserver;
211
+ readonly callKind?: ProviderCallKind;
212
+ }
213
+
155
214
  export interface Provider {
156
215
  // Optional package-authored folksonomy evaluated by the consumer immediately
157
216
  // before a provider emission attempt ({§plugin-attribution}).
158
217
  attributions?(context: PluginAttributionContext): PluginAttribution;
159
218
  // `grammar` is an optional GBNF string (canonically @plurnk/plurnk-contracts'
160
- // plurnk.gbnf, possibly root-substituted by the consumer). Backends that
219
+ // plurnk.gemma.gbnf or plurnk.qwen.gbnf, possibly root-substituted by the consumer). Backends that
161
220
  // support grammar-constrained sampling attach it verbatim; all others
162
221
  // ignore it. The provider never chooses or modifies the grammar — whether
163
222
  // to constrain and which root variant to send is consumer policy
164
223
  // ({§gbnf-response-observation}).
165
224
  //
166
- // `maxTokens` is the consumer's per-call output ceiling (wire `max_tokens`).
167
- // Without it, most servers generate UNBOUNDED (llama-server n_predict -1) —
168
- // under a multi-op grammar that degenerates to the context wall,
169
- // so a constrained consumer is expected to pass it. Policy stays the
170
- // consumer's; the provider only transports.
225
+ // `maxOutputTokens` may tighten the provider's configured total output
226
+ // budget for this call. It includes visible output and hidden reasoning;
227
+ // the adapter owns projection into each backend's native wire semantics.
171
228
  //
172
229
  // `workerId` is the REQUIRED, opaque, stable identity of the consumer's work
173
230
  // stream (loop/run). Providers MAY key backend affinity on it — e.g.
@@ -206,10 +263,21 @@ export interface Provider {
206
263
  // `Plurnk-Turn` ONLY under the same firstPartyMetadata gate; dropped
207
264
  // everywhere else. Coordinates are 1-based: absent/0 emits no header (no
208
265
  // strikes-style zero exception). Headers only, never the packet.
209
- generate(args: { messages: ChatMessage[]; workerId: string; primaryWorkerId?: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse>;
210
- // {§model-fact-resolution} — effective physical context in tokens. `null`
211
- // means unknown; under llama-server parallelism the probed value is per slot.
266
+ //
267
+ // `callKind` is the caller's explicit output contract. It is transported as
268
+ // `Plurnk-Call-Kind` only under the first-party metadata gate and never
269
+ // inferred from the request shape. Generic callers may omit it; Core always
270
+ // supplies `emission` or `bare`.
271
+ generate(args: ProviderGenerateArgs): Promise<ProviderResponse>;
272
+ // {§model-fact-resolution} — effective total context envelope in tokens,
273
+ // including any stricter operator cap. `null` means unknown; under
274
+ // llama-server parallelism the probed natural value is per slot.
212
275
  readonly contextWindow: number | null;
276
+ readonly maxInputTokens: number | null;
277
+ readonly maxOutputTokens: number | null;
278
+ readonly outputBudget: number | null;
279
+ readonly reasoningBudget: number | null;
280
+ readonly inputCapacity: number | null;
213
281
  readonly model: string;
214
282
  // Optional: the backend's self-reported served model id, from a
215
283
  // /v1/models-shaped probe (llama-server today; any such backend). For a local
@@ -229,20 +297,19 @@ export interface Provider {
229
297
  // clamp an over-ask (fireworks/xai, verified live) never set this; undefined
230
298
  // = no claim. Introspectable so a consumer can refuse AT BOOT a local alias
231
299
  // with no declared envelope, instead of dying mid-turn in partition math.
232
- readonly requiresMaxTokens?: boolean;
233
- // Optional generation-envelope reserves ({§provider-generation-envelope}) — the amounts of
234
- // the DETECTED window reserved for reasoning and completion: floor
235
- // percentages of `contextWindow`, or absolute per-alias pins that win
236
- // outright. The consumer's prompt budget is `contextWindow - reasoningReserve
237
- // - completionReserve - <its own packing-safety margin>`; the generation cap
238
- // is the two pooled. `null` = underivable (window unknown, no absolute pin) →
239
- // the consumer's no-cap path. Absent = a bare sibling makes NO claim (treated
240
- // as null). All first-party providers claim, so null means genuinely-unknown.
241
- readonly reasoningReserve?: number | null;
242
- readonly completionReserve?: number | null;
300
+ readonly requiresOutputBudget?: boolean;
301
+ // The adapter owns request-specific physical admission. A proven fit may
302
+ // admit and an exact overflow may reject before I/O. Incomplete limits,
303
+ // estimates, bounds that do not prove fit, and unavailable measurements
304
+ // defer to the upstream provider as the capacity oracle.
305
+ assessRequestCapacity(
306
+ messages: readonly ChatMessage[],
307
+ maxOutputTokens?: number,
308
+ signal?: AbortSignal,
309
+ ): Promise<ProviderRequestCapacity>;
243
310
  // Provider-owned preflight measurement of the complete chat request,
244
311
  // including provider/template framing when the adapter can know it.
245
- // Estimates are explicit and MUST NOT authorize hard physical admission.
312
+ // Estimates are explicit and MUST NOT authorize hard context-envelope admission.
246
313
  countPromptTokens(messages: readonly ChatMessage[], signal?: AbortSignal): Promise<PromptTokenMeasurement>;
247
314
  // OPTIONAL capability: exact tokenization served by the backend's own vocab
248
315
  // (llama-server /tokenize) — token ids in the model's real vocabulary.
@@ -250,12 +317,6 @@ export interface Provider {
250
317
  // `tokenize === undefined` means the backend can't. Exact-counting
251
318
  // consumers (the tokenizer seam) prefer this over any client-side data.
252
319
  tokenize?(text: string): Promise<number[]>;
253
- // {§model-fact-resolution} — frozen 1.x local USD estimate compatibility.
254
- // This is not a monetary-reporting authority.
255
- calculateCost(usage: ProviderUsage): number;
256
- // Models.dev-derived monetary fallback. Direct response charges travel on
257
- // ProviderResponse and take precedence at the consuming attempt boundary.
258
- calculateCharge?(usage: ProviderUsage): Exclude<ProviderCost, AuthoritativeCharge>;
259
320
  }
260
321
 
261
322
  // ProviderAlias lives in @plurnk/plurnk-aliases (the zero-dependency parser);
package/src/usage.test.ts CHANGED
@@ -1,153 +1,149 @@
1
1
  import test from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
- import { normalizeUsage, calculateCostUsd, calculateCostUsdDecimal } from "./usage.ts";
3
+ import { calculateCostUsdDecimal, normalizeUsage } from "./usage.ts";
4
4
 
5
- // — normalizeUsage —
6
-
7
- test("normalizeUsage: Gemini-style — reasoning recovered from total gap", () => {
8
- // Real Gemini OAI-compat shape: no details, reasoning hidden in total.
9
- const u = normalizeUsage({ prompt_tokens: 19, completion_tokens: 285, total_tokens: 1165 });
10
- assert.deepEqual(u, { prompt: 19, completion: 285, reasoning: 861, cached: 0, total: 1165 });
11
- assert.equal(u.prompt + u.completion + u.reasoning, u.total); // invariant
5
+ test("normalizeUsage recovers additive hidden reasoning from an exact total", () => {
6
+ const usage = normalizeUsage({
7
+ prompt_tokens: 19,
8
+ completion_tokens: 285,
9
+ total_tokens: 1165,
10
+ });
11
+ assert.deepEqual(usage, {
12
+ inputTokens: 19,
13
+ outputTokens: 1146,
14
+ totalTokens: 1165,
15
+ outputTokenDetails: { textTokens: 285, reasoningTokens: 861 },
16
+ });
12
17
  });
13
18
 
14
- test("normalizeUsage: OpenAI-style — reasoning split out of completion_tokens", () => {
15
- // completion_tokens includes reasoning; total = prompt + completion.
16
- const u = normalizeUsage({
19
+ test("normalizeUsage preserves OpenAI output totals while itemizing reasoning", () => {
20
+ const usage = normalizeUsage({
17
21
  prompt_tokens: 10,
18
22
  completion_tokens: 100,
19
23
  total_tokens: 110,
20
24
  completion_tokens_details: { reasoning_tokens: 40 },
21
25
  });
22
- assert.deepEqual(u, { prompt: 10, completion: 60, reasoning: 40, cached: 0, total: 110 });
23
- assert.equal(u.completion + u.reasoning, 100); // billable output unchanged
26
+ assert.deepEqual(usage, {
27
+ inputTokens: 10,
28
+ outputTokens: 100,
29
+ totalTokens: 110,
30
+ outputTokenDetails: { textTokens: 60, reasoningTokens: 40 },
31
+ });
24
32
  });
25
33
 
26
- test("normalizeUsage: xAI/Grok-style — reasoning is ADDITIVE, not subtracted from completion", () => {
27
- // Real grok-4.3 shape: completion_tokens is visible-only; reasoning_tokens is
28
- // detailed but ADDITIVE — total = prompt + completion + reasoning.
29
- const u = normalizeUsage({
34
+ test("normalizeUsage recognizes xAI reasoning as additive from the total identity", () => {
35
+ const usage = normalizeUsage({
30
36
  prompt_tokens: 143,
31
37
  completion_tokens: 1,
32
38
  total_tokens: 441,
33
39
  prompt_tokens_details: { cached_tokens: 128 },
34
40
  completion_tokens_details: { reasoning_tokens: 297 },
35
41
  });
36
- assert.deepEqual(u, { prompt: 143, completion: 1, reasoning: 297, cached: 128, total: 441 });
37
- assert.equal(u.completion, 1); // visible output preserved, NOT zeroed by the subtraction
38
- assert.equal(u.completion + u.reasoning, 298); // billable output = visible + reasoning
39
- });
40
-
41
- test("normalizeUsage: cached read from prompt_tokens_details (OpenAI nesting)", () => {
42
- const u = normalizeUsage({ prompt_tokens: 50, completion_tokens: 10, total_tokens: 60, prompt_tokens_details: { cached_tokens: 30 } });
43
- assert.equal(u.cached, 30);
42
+ assert.deepEqual(usage, {
43
+ inputTokens: 143,
44
+ outputTokens: 298,
45
+ totalTokens: 441,
46
+ inputTokenDetails: { cacheReadTokens: 128 },
47
+ outputTokenDetails: { textTokens: 1, reasoningTokens: 297 },
48
+ });
44
49
  });
45
50
 
46
- test("normalizeUsage: top-level cached_tokens still honored", () => {
47
- const u = normalizeUsage({ prompt_tokens: 50, completion_tokens: 10, total_tokens: 60, cached_tokens: 12 });
48
- assert.equal(u.cached, 12);
51
+ test("normalizeUsage maps cache-read spellings without inventing uncached tokens", () => {
52
+ assert.deepEqual(normalizeUsage({
53
+ prompt_tokens: 50,
54
+ completion_tokens: 10,
55
+ total_tokens: 60,
56
+ prompt_tokens_details: { cached_tokens: 30 },
57
+ })?.inputTokenDetails, { cacheReadTokens: 30 });
58
+ assert.deepEqual(normalizeUsage({
59
+ prompt_tokens: 50,
60
+ completion_tokens: 10,
61
+ total_tokens: 60,
62
+ cached_tokens: 12,
63
+ })?.inputTokenDetails, { cacheReadTokens: 12 });
49
64
  });
50
65
 
51
- test("#157: normalizeUsage maps DeepSeek's prompt cache hit count", () => {
52
- const u = normalizeUsage({
66
+ test("#157: normalizeUsage maps DeepSeek cache hit and miss counts", () => {
67
+ assert.deepEqual(normalizeUsage({
53
68
  prompt_tokens: 50,
54
69
  prompt_cache_hit_tokens: 30,
55
70
  prompt_cache_miss_tokens: 20,
56
71
  completion_tokens: 10,
57
72
  total_tokens: 60,
73
+ }), {
74
+ inputTokens: 50,
75
+ outputTokens: 10,
76
+ totalTokens: 60,
77
+ inputTokenDetails: { noCacheTokens: 20, cacheReadTokens: 30 },
58
78
  });
59
- assert.deepEqual(u, { prompt: 50, completion: 10, reasoning: 0, cached: 30, total: 60 });
60
- });
61
-
62
- test("normalizeUsage: no reasoning — plain prompt+completion", () => {
63
- const u = normalizeUsage({ prompt_tokens: 10, completion_tokens: 20, total_tokens: 30 });
64
- assert.deepEqual(u, { prompt: 10, completion: 20, reasoning: 0, cached: 0, total: 30 });
65
- });
66
-
67
- test("normalizeUsage: missing total is reconstructed, never negative reasoning", () => {
68
- const u = normalizeUsage({ prompt_tokens: 10, completion_tokens: 20 });
69
- assert.deepEqual(u, { prompt: 10, completion: 20, reasoning: 0, cached: 0, total: 30 });
70
79
  });
71
80
 
72
- test("normalizeUsage: absent usage → all zeros", () => {
73
- assert.deepEqual(normalizeUsage(null), { prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 });
74
- assert.deepEqual(normalizeUsage(undefined), { prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 });
75
- });
76
-
77
- // -- Fireworks-style: reasoning shipped as TEXT, folded into completion, not itemized --
78
-
79
- test("normalizeUsage: fireworks folds reasoning into completion -- re-split by text proportion, sum preserved", () => {
80
- // total = prompt + completion (no gap), reasoning_tokens absent, but 750 vs 250
81
- // chars of reasoning vs content came back. Split completion 75/25; cost base held.
82
- const u = normalizeUsage(
83
- { prompt_tokens: 100, completion_tokens: 1000, total_tokens: 1100 },
84
- "r".repeat(750),
85
- "c".repeat(250),
86
- );
87
- assert.deepEqual(u, { prompt: 100, completion: 250, reasoning: 750, cached: 0, total: 1100 });
88
- assert.equal(u.completion + u.reasoning, 1000); // billable output byte-identical
89
- assert.equal(u.prompt + u.completion + u.reasoning, u.total); // invariant
90
- });
91
-
92
- test("normalizeUsage: pure-reasoning turn (empty content) attributes all completion to reasoning", () => {
93
- // The run52 runaway shape: 0 visible content, the whole budget spent reasoning.
94
- const u = normalizeUsage(
95
- { prompt_tokens: 100, completion_tokens: 500, total_tokens: 600 },
96
- "t".repeat(9000),
97
- "",
98
- );
99
- assert.deepEqual(u, { prompt: 100, completion: 0, reasoning: 500, cached: 0, total: 600 });
100
- });
101
-
102
- test("normalizeUsage: text args never perturb the itemized (reasoning_tokens) path", () => {
103
- // OpenAI o-series reports reasoning_tokens -> that split wins, text is ignored.
104
- const u = normalizeUsage(
105
- { prompt_tokens: 10, completion_tokens: 100, total_tokens: 110, completion_tokens_details: { reasoning_tokens: 40 } },
106
- "r".repeat(999), "c".repeat(1),
107
- );
108
- assert.deepEqual(u, { prompt: 10, completion: 60, reasoning: 40, cached: 0, total: 110 });
109
- });
110
-
111
- test("normalizeUsage: text args never perturb the Gemini gap path (gap already yields reasoning)", () => {
112
- // A real total gap means reasoning is itemized-by-subtraction; do not re-split.
113
- const u = normalizeUsage(
114
- { prompt_tokens: 19, completion_tokens: 285, total_tokens: 1165 },
115
- "r".repeat(500), "c".repeat(500),
116
- );
117
- assert.equal(u.reasoning, 861); // from the gap, NOT a text re-split
118
- assert.equal(u.completion, 285);
119
- });
120
-
121
- test("normalizeUsage: no total reported -> re-split skipped, reasoning stays 0 (cannot split an unknown base)", () => {
122
- const u = normalizeUsage(
123
- { prompt_tokens: 10, completion_tokens: 20 },
124
- "r".repeat(500), "c".repeat(500),
125
- );
126
- assert.deepEqual(u, { prompt: 10, completion: 20, reasoning: 0, cached: 0, total: 30 });
127
- });
128
-
129
- // — calculateCostUsd —
130
-
131
- test("calculateCostUsd: bills reasoning at the USD-per-million output rate", () => {
132
- // 100 input, 0 cached, 50 completion + 200 reasoning = 250 output.
133
- const usage = { prompt: 100, completion: 50, reasoning: 200, cached: 0, total: 350 };
134
- assert.equal(calculateCostUsd(usage, { input: 1, output: 10, cached: 0 }), 0.0026);
81
+ test("normalizeUsage derives only exact totals", () => {
82
+ assert.deepEqual(normalizeUsage({ prompt_tokens: 10, completion_tokens: 20 }), {
83
+ inputTokens: 10,
84
+ outputTokens: 20,
85
+ totalTokens: 30,
86
+ });
87
+ assert.deepEqual(normalizeUsage({ prompt_tokens: 10, total_tokens: 30 }), {
88
+ inputTokens: 10,
89
+ outputTokens: 20,
90
+ totalTokens: 30,
91
+ });
135
92
  });
136
93
 
137
- test("calculateCostUsd: cached prompt billed at the cache rate, remainder at input", () => {
138
- const usage = { prompt: 1000, completion: 0, reasoning: 0, cached: 400, total: 1000 };
139
- assert.equal(calculateCostUsd(usage, { input: 5, output: 99, cached: 1 }), 0.0034);
94
+ test("normalizeUsage preserves unknown usage as absence", () => {
95
+ assert.equal(normalizeUsage(null), undefined);
96
+ assert.equal(normalizeUsage(undefined), undefined);
140
97
  });
141
98
 
142
- test("calculateCostUsd: zero rates → 0", () => {
143
- const usage = { prompt: 9, completion: 9, reasoning: 9, cached: 9, total: 27 };
144
- assert.equal(calculateCostUsd(usage, { input: 0, output: 0, cached: 0 }), 0);
99
+ test("normalizeUsage never apportions tokens from reasoning or content length", () => {
100
+ assert.deepEqual(normalizeUsage({
101
+ prompt_tokens: 100,
102
+ completion_tokens: 1000,
103
+ total_tokens: 1100,
104
+ }), {
105
+ inputTokens: 100,
106
+ outputTokens: 1000,
107
+ totalTokens: 1100,
108
+ });
145
109
  });
146
110
 
147
- test("calculateCostUsdDecimal preserves Models.dev rates without floating-point artifacts", () => {
148
- const usage = { prompt: 1_000, completion: 100, reasoning: 50, cached: 400, total: 1_150 };
149
- assert.equal(
150
- calculateCostUsdDecimal(usage, { input: 0.14, output: 0.28, cached: 0.0028 }),
151
- "0.00012712",
152
- );
111
+ test("calculateCostUsdDecimal bills all output, including reasoning, at the output rate", () => {
112
+ assert.equal(calculateCostUsdDecimal({
113
+ inputTokens: 100,
114
+ outputTokens: 250,
115
+ totalTokens: 350,
116
+ outputTokenDetails: { textTokens: 50, reasoningTokens: 200 },
117
+ }, { input: 1, output: 10 }), "0.0026");
118
+ });
119
+
120
+ test("calculateCostUsdDecimal applies distinct cache-read and cache-write rates", () => {
121
+ assert.equal(calculateCostUsdDecimal({
122
+ inputTokens: 1000,
123
+ outputTokens: 0,
124
+ totalTokens: 1000,
125
+ inputTokenDetails: {
126
+ noCacheTokens: 500,
127
+ cacheReadTokens: 400,
128
+ cacheWriteTokens: 100,
129
+ },
130
+ }, { input: 5, output: 99, cacheRead: 1, cacheWrite: 8 }), "0.0037");
131
+ });
132
+
133
+ test("calculateCostUsdDecimal returns unknown when a differently-priced category is absent", () => {
134
+ assert.equal(calculateCostUsdDecimal({
135
+ inputTokens: 1000,
136
+ outputTokens: 0,
137
+ totalTokens: 1000,
138
+ }, { input: 5, output: 99, cacheRead: 1 }), null);
139
+ });
140
+
141
+ test("calculateCostUsdDecimal preserves Models.dev decimals without floating-point artifacts", () => {
142
+ assert.equal(calculateCostUsdDecimal({
143
+ inputTokens: 1000,
144
+ outputTokens: 150,
145
+ totalTokens: 1150,
146
+ inputTokenDetails: { cacheReadTokens: 400 },
147
+ outputTokenDetails: { textTokens: 100, reasoningTokens: 50 },
148
+ }, { input: 0.14, output: 0.28, cacheRead: 0.0028 }), "0.00012712");
153
149
  });