@plurnk/plurnk-providers 1.5.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. package/.env.defaults +36 -22
  2. package/SPEC.md +133 -59
  3. package/dist/AiSdkProvider.d.ts +19 -26
  4. package/dist/AiSdkProvider.d.ts.map +1 -1
  5. package/dist/AiSdkProvider.js +318 -106
  6. package/dist/AiSdkProvider.js.map +1 -1
  7. package/dist/Mock.d.ts +4 -9
  8. package/dist/Mock.d.ts.map +1 -1
  9. package/dist/Mock.js +36 -9
  10. package/dist/Mock.js.map +1 -1
  11. package/dist/Pool.d.ts +2 -21
  12. package/dist/Pool.d.ts.map +1 -1
  13. package/dist/Pool.js +19 -14
  14. package/dist/Pool.js.map +1 -1
  15. package/dist/accounting.d.ts +5 -2
  16. package/dist/accounting.d.ts.map +1 -1
  17. package/dist/accounting.js +100 -16
  18. package/dist/accounting.js.map +1 -1
  19. package/dist/aiSdkTransport.d.ts +9 -2
  20. package/dist/aiSdkTransport.d.ts.map +1 -1
  21. package/dist/aiSdkTransport.js +160 -62
  22. package/dist/aiSdkTransport.js.map +1 -1
  23. package/dist/catalogProvider.d.ts +7 -3
  24. package/dist/catalogProvider.d.ts.map +1 -1
  25. package/dist/catalogProvider.js +30 -24
  26. package/dist/catalogProvider.js.map +1 -1
  27. package/dist/compatibleProvider.d.ts.map +1 -1
  28. package/dist/compatibleProvider.js +18 -7
  29. package/dist/compatibleProvider.js.map +1 -1
  30. package/dist/cost.d.ts +10 -10
  31. package/dist/cost.d.ts.map +1 -1
  32. package/dist/cost.js +90 -42
  33. package/dist/cost.js.map +1 -1
  34. package/dist/env.d.ts +5 -1
  35. package/dist/env.d.ts.map +1 -1
  36. package/dist/env.js +30 -10
  37. package/dist/env.js.map +1 -1
  38. package/dist/errors.d.ts +14 -2
  39. package/dist/errors.d.ts.map +1 -1
  40. package/dist/errors.js +58 -2
  41. package/dist/errors.js.map +1 -1
  42. package/dist/index.d.ts +4 -4
  43. package/dist/index.d.ts.map +1 -1
  44. package/dist/index.js +3 -2
  45. package/dist/index.js.map +1 -1
  46. package/dist/ollama.js +3 -3
  47. package/dist/ollama.js.map +1 -1
  48. package/dist/sdkModels.d.ts +6 -2
  49. package/dist/sdkModels.d.ts.map +1 -1
  50. package/dist/sdkModels.js +38 -5
  51. package/dist/sdkModels.js.map +1 -1
  52. package/dist/types.d.ts +33 -31
  53. package/dist/types.d.ts.map +1 -1
  54. package/dist/usage.d.ts +21 -5
  55. package/dist/usage.d.ts.map +1 -1
  56. package/dist/usage.js +164 -83
  57. package/dist/usage.js.map +1 -1
  58. package/package.json +7 -6
  59. package/src/AiSdkProvider.test.ts +788 -191
  60. package/src/AiSdkProvider.ts +381 -124
  61. package/src/Mock.test.ts +37 -12
  62. package/src/Mock.ts +45 -14
  63. package/src/Pool.test.ts +19 -6
  64. package/src/Pool.ts +20 -16
  65. package/src/ProviderRegistry.test.ts +16 -11
  66. package/src/accounting.test.ts +58 -22
  67. package/src/accounting.ts +120 -18
  68. package/src/aiSdkTransport.test.ts +42 -49
  69. package/src/aiSdkTransport.ts +174 -62
  70. package/src/boundaries.test.ts +1 -0
  71. package/src/catalogProvider.test.ts +258 -22
  72. package/src/catalogProvider.ts +42 -27
  73. package/src/compatibleProvider.test.ts +6 -3
  74. package/src/compatibleProvider.ts +20 -7
  75. package/src/cost.test.ts +55 -36
  76. package/src/cost.ts +111 -50
  77. package/src/defaults.test.ts +13 -3
  78. package/src/env.test.ts +54 -5
  79. package/src/env.ts +43 -18
  80. package/src/errors.test.ts +47 -2
  81. package/src/errors.ts +67 -3
  82. package/src/index.ts +21 -5
  83. package/src/ollama.test.ts +4 -1
  84. package/src/ollama.ts +3 -3
  85. package/src/sdkModels.test.ts +76 -4
  86. package/src/sdkModels.ts +45 -7
  87. package/src/types.ts +77 -38
  88. package/src/usage.test.ts +112 -116
  89. package/src/usage.ts +209 -93
package/src/types.ts CHANGED
@@ -9,15 +9,32 @@ import type {
9
9
  PluginAttributionContext,
10
10
  PluginAttributionSource,
11
11
  } from "@plurnk/plurnk-meta";
12
- import type { ProviderCost } from "@plurnk/plurnk-contracts";
12
+ import type {
13
+ ProviderAccounting,
14
+ ProviderCost,
15
+ ProviderRequestAccounting,
16
+ ProviderUsage,
17
+ } from "@plurnk/plurnk-contracts";
18
+
19
+ export type {
20
+ ProviderAccounting,
21
+ ProviderCost,
22
+ ProviderRequestAccounting,
23
+ ProviderUsage,
24
+ } from "@plurnk/plurnk-contracts";
13
25
 
14
26
  export interface ChatMessage {
15
27
  role: "system" | "user" | "assistant";
16
28
  content: string;
17
29
  }
18
30
 
31
+ // {§provider-call-kind} The caller-owned semantic contract for one logical model call. Providers do
32
+ // not infer this from messages or grammar presence: an emission expects PLURNK
33
+ // output, while a bare call expects unconstrained response text.
34
+ export type ProviderCallKind = "emission" | "bare";
35
+
19
36
  // Preflight evidence for the complete provider request. An empirical estimate
20
- // is useful telemetry but cannot authorize a hard physical-capacity decision.
37
+ // is useful telemetry but cannot authorize hard context-envelope admission.
21
38
  export type PromptTokenMeasurement =
22
39
  | {
23
40
  readonly kind: "exact" | "upper_bound";
@@ -31,39 +48,44 @@ export type PromptTokenMeasurement =
31
48
  readonly detail: string;
32
49
  };
33
50
 
34
- // Normalized token accounting. Invariant (enforced by normalizeUsage at the
35
- // provider boundary): total = prompt + completion + reasoning; cached is a
36
- // subset of prompt. `completion` is visible output EXCLUDING reasoning; the
37
- // billable output is `completion + reasoning` (frontier providers bill reasoning
38
- // tokens at the output rate).
39
- export interface ProviderUsage {
40
- readonly prompt: number; // input tokens (cached ones included)
41
- readonly completion: number; // visible output tokens, excluding reasoning
42
- readonly reasoning: number; // reasoning tokens, billed as output
43
- readonly cached: number; // subset of prompt served from cache
44
- readonly total: number; // prompt + completion + reasoning
45
- }
46
-
47
- export type AuthoritativeCharge = Extract<ProviderCost, { kind: "authoritative" }>;
51
+ export type ChargedCost = Extract<ProviderCost, { kind: "charged" }>;
48
52
 
49
53
  // Evidence exposed by the transport to the provider adapter that owns its
50
54
  // vendor protocol. Core and downstream consumers receive only the normalized
51
55
  // charge, never a requirement to understand provider metadata fields.
52
56
  export interface ProviderChargeEvidence {
53
57
  readonly providerMetadata?: unknown;
58
+ // A protocol-owned direct monetary field. It remains unknown until the
59
+ // selected adapter explicitly validates and normalizes it.
60
+ readonly charge?: unknown;
54
61
  // Provider-owned raw usage projection retained independently of optional
55
62
  // full-body capture. Accounting fields cannot disappear merely because
56
63
  // forensic raw-body capture is disabled.
57
64
  readonly usage?: unknown;
58
65
  readonly response: {
59
- readonly id: string;
66
+ readonly id?: string;
60
67
  readonly headers?: Readonly<Record<string, string>>;
61
68
  };
62
69
  }
63
70
 
64
- export type AuthoritativeChargeNormalizer = (
71
+ export type ProviderCostNormalizer = (
65
72
  evidence: ProviderChargeEvidence,
66
- ) => AuthoritativeCharge | undefined;
73
+ ) => ProviderCost | undefined;
74
+
75
+ export interface ProviderRequestIdentity {
76
+ readonly provider: string;
77
+ readonly model: string;
78
+ }
79
+
80
+ export type ProviderRequestSettlement = (
81
+ accounting: ProviderRequestAccounting,
82
+ ) => Promise<void>;
83
+
84
+ // Core opens durable physical-request identity through this observer before
85
+ // provider I/O. The returned settlement closes that exact identity.
86
+ export type ProviderRequestObserver = (
87
+ identity: ProviderRequestIdentity,
88
+ ) => Promise<ProviderRequestSettlement>;
67
89
 
68
90
  // A successful exchange's closed finish set. ProviderAttemptFinishReason adds
69
91
  // the failed disposition that may occur only on ProviderError attempt evidence.
@@ -102,7 +124,6 @@ export interface ProviderAssistant<TFinish extends ProviderAttemptFinishReason =
102
124
  readonly reasoning: string | null;
103
125
  // Encrypted reasoning remains distinct from readable `reasoning`.
104
126
  readonly reasoningEncrypted?: ReadonlyArray<ProviderEncryptedReasoningItem>;
105
- readonly usage: ProviderUsage;
106
127
  readonly finishReason: TFinish;
107
128
  readonly model: string;
108
129
  // Per-token logprobs, present only when PLURNK_PROVIDERS_TOP_LOGPROBS is set
@@ -114,8 +135,9 @@ export interface ProviderAssistant<TFinish extends ProviderAttemptFinishReason =
114
135
  }
115
136
 
116
137
  export interface GrammarEvidence {
117
- // Exact sentence observed at the grammar boundary before any reasoning/content
118
- // projection. Offsets are Unicode code points, matching @plurnk/gbnf verdicts.
138
+ // Exact pre-projection response represented by the provider. A generated
139
+ // rail's response root composes any template prefix before @plurnk/gbnf
140
+ // grades it. Offsets are Unicode code points, matching validator verdicts.
119
141
  readonly input: string;
120
142
  readonly contentStart: number;
121
143
  readonly transported: boolean;
@@ -124,10 +146,9 @@ export interface GrammarEvidence {
124
146
  export interface ProviderResponse<TFinish extends ProviderAttemptFinishReason = FinishReason> {
125
147
  readonly assistant: ProviderAssistant<TFinish>;
126
148
  readonly assistantRaw: unknown;
127
- // A settled upstream charge is a validated public fact, not opaque metadata.
128
- // Non-USD settlement carries an explicit provider-owned USD equivalent for
129
- // the platform's existing USD aggregate. Core never supplies an FX rate.
130
- readonly charge?: AuthoritativeCharge;
149
+ // Ordered physical request evidence, including automatic retries and pool
150
+ // failover that preceded this response. {§provider-request-accounting}
151
+ readonly accounting: readonly ProviderRequestAccounting[];
131
152
  // {§gbnf-response-observation} — evidence only; the consumer owns the verdict.
132
153
  readonly grammarEvidence?: GrammarEvidence;
133
154
  // Per-turn provider→client metadata bag: the backend's non-standard top-level
@@ -152,12 +173,30 @@ export interface ProviderResponse<TFinish extends ProviderAttemptFinishReason =
152
173
 
153
174
  export type ProviderAttempt = ProviderResponse<ProviderAttemptFinishReason>;
154
175
 
176
+ export interface ProviderGenerateArgs {
177
+ readonly messages: ChatMessage[];
178
+ readonly workerId: string;
179
+ readonly primaryWorkerId?: string;
180
+ readonly signal?: AbortSignal;
181
+ readonly grammar?: string;
182
+ readonly maxTokens?: number;
183
+ readonly attributions?: string[];
184
+ readonly client?: string;
185
+ readonly strikes?: number;
186
+ readonly workspaceId?: string;
187
+ readonly loop?: number;
188
+ readonly turn?: number;
189
+ readonly sampling?: Record<string, unknown>;
190
+ readonly observeRequest?: ProviderRequestObserver;
191
+ readonly callKind?: ProviderCallKind;
192
+ }
193
+
155
194
  export interface Provider {
156
195
  // Optional package-authored folksonomy evaluated by the consumer immediately
157
196
  // before a provider emission attempt ({§plugin-attribution}).
158
197
  attributions?(context: PluginAttributionContext): PluginAttribution;
159
198
  // `grammar` is an optional GBNF string (canonically @plurnk/plurnk-contracts'
160
- // plurnk.gbnf, possibly root-substituted by the consumer). Backends that
199
+ // plurnk.gemma.gbnf or plurnk.qwen.gbnf, possibly root-substituted by the consumer). Backends that
161
200
  // support grammar-constrained sampling attach it verbatim; all others
162
201
  // ignore it. The provider never chooses or modifies the grammar — whether
163
202
  // to constrain and which root variant to send is consumer policy
@@ -206,9 +245,15 @@ export interface Provider {
206
245
  // `Plurnk-Turn` ONLY under the same firstPartyMetadata gate; dropped
207
246
  // everywhere else. Coordinates are 1-based: absent/0 emits no header (no
208
247
  // strikes-style zero exception). Headers only, never the packet.
209
- generate(args: { messages: ChatMessage[]; workerId: string; primaryWorkerId?: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse>;
210
- // {§model-fact-resolution} — effective physical context in tokens. `null`
211
- // means unknown; under llama-server parallelism the probed value is per slot.
248
+ //
249
+ // `callKind` is the caller's explicit output contract. It is transported as
250
+ // `Plurnk-Call-Kind` only under the first-party metadata gate and never
251
+ // inferred from the request shape. Generic callers may omit it; Core always
252
+ // supplies `emission` or `bare`.
253
+ generate(args: ProviderGenerateArgs): Promise<ProviderResponse>;
254
+ // {§model-fact-resolution} — effective total context envelope in tokens,
255
+ // including any stricter operator cap. `null` means unknown; under
256
+ // llama-server parallelism the probed natural value is per slot.
212
257
  readonly contextWindow: number | null;
213
258
  readonly model: string;
214
259
  // Optional: the backend's self-reported served model id, from a
@@ -231,7 +276,7 @@ export interface Provider {
231
276
  // with no declared envelope, instead of dying mid-turn in partition math.
232
277
  readonly requiresMaxTokens?: boolean;
233
278
  // Optional generation-envelope reserves ({§provider-generation-envelope}) — the amounts of
234
- // the DETECTED window reserved for reasoning and completion: floor
279
+ // the effective window reserved for reasoning and completion: floor
235
280
  // percentages of `contextWindow`, or absolute per-alias pins that win
236
281
  // outright. The consumer's prompt budget is `contextWindow - reasoningReserve
237
282
  // - completionReserve - <its own packing-safety margin>`; the generation cap
@@ -242,7 +287,7 @@ export interface Provider {
242
287
  readonly completionReserve?: number | null;
243
288
  // Provider-owned preflight measurement of the complete chat request,
244
289
  // including provider/template framing when the adapter can know it.
245
- // Estimates are explicit and MUST NOT authorize hard physical admission.
290
+ // Estimates are explicit and MUST NOT authorize hard context-envelope admission.
246
291
  countPromptTokens(messages: readonly ChatMessage[], signal?: AbortSignal): Promise<PromptTokenMeasurement>;
247
292
  // OPTIONAL capability: exact tokenization served by the backend's own vocab
248
293
  // (llama-server /tokenize) — token ids in the model's real vocabulary.
@@ -250,12 +295,6 @@ export interface Provider {
250
295
  // `tokenize === undefined` means the backend can't. Exact-counting
251
296
  // consumers (the tokenizer seam) prefer this over any client-side data.
252
297
  tokenize?(text: string): Promise<number[]>;
253
- // {§model-fact-resolution} — frozen 1.x local USD estimate compatibility.
254
- // This is not a monetary-reporting authority.
255
- calculateCost(usage: ProviderUsage): number;
256
- // Models.dev-derived monetary fallback. Direct response charges travel on
257
- // ProviderResponse and take precedence at the consuming attempt boundary.
258
- calculateCharge?(usage: ProviderUsage): Exclude<ProviderCost, AuthoritativeCharge>;
259
298
  }
260
299
 
261
300
  // ProviderAlias lives in @plurnk/plurnk-aliases (the zero-dependency parser);
package/src/usage.test.ts CHANGED
@@ -1,153 +1,149 @@
1
1
  import test from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
- import { normalizeUsage, calculateCostUsd, calculateCostUsdDecimal } from "./usage.ts";
3
+ import { calculateCostUsdDecimal, normalizeUsage } from "./usage.ts";
4
4
 
5
- // — normalizeUsage —
6
-
7
- test("normalizeUsage: Gemini-style — reasoning recovered from total gap", () => {
8
- // Real Gemini OAI-compat shape: no details, reasoning hidden in total.
9
- const u = normalizeUsage({ prompt_tokens: 19, completion_tokens: 285, total_tokens: 1165 });
10
- assert.deepEqual(u, { prompt: 19, completion: 285, reasoning: 861, cached: 0, total: 1165 });
11
- assert.equal(u.prompt + u.completion + u.reasoning, u.total); // invariant
5
+ test("normalizeUsage recovers additive hidden reasoning from an exact total", () => {
6
+ const usage = normalizeUsage({
7
+ prompt_tokens: 19,
8
+ completion_tokens: 285,
9
+ total_tokens: 1165,
10
+ });
11
+ assert.deepEqual(usage, {
12
+ inputTokens: 19,
13
+ outputTokens: 1146,
14
+ totalTokens: 1165,
15
+ outputTokenDetails: { textTokens: 285, reasoningTokens: 861 },
16
+ });
12
17
  });
13
18
 
14
- test("normalizeUsage: OpenAI-style — reasoning split out of completion_tokens", () => {
15
- // completion_tokens includes reasoning; total = prompt + completion.
16
- const u = normalizeUsage({
19
+ test("normalizeUsage preserves OpenAI output totals while itemizing reasoning", () => {
20
+ const usage = normalizeUsage({
17
21
  prompt_tokens: 10,
18
22
  completion_tokens: 100,
19
23
  total_tokens: 110,
20
24
  completion_tokens_details: { reasoning_tokens: 40 },
21
25
  });
22
- assert.deepEqual(u, { prompt: 10, completion: 60, reasoning: 40, cached: 0, total: 110 });
23
- assert.equal(u.completion + u.reasoning, 100); // billable output unchanged
26
+ assert.deepEqual(usage, {
27
+ inputTokens: 10,
28
+ outputTokens: 100,
29
+ totalTokens: 110,
30
+ outputTokenDetails: { textTokens: 60, reasoningTokens: 40 },
31
+ });
24
32
  });
25
33
 
26
- test("normalizeUsage: xAI/Grok-style — reasoning is ADDITIVE, not subtracted from completion", () => {
27
- // Real grok-4.3 shape: completion_tokens is visible-only; reasoning_tokens is
28
- // detailed but ADDITIVE — total = prompt + completion + reasoning.
29
- const u = normalizeUsage({
34
+ test("normalizeUsage recognizes xAI reasoning as additive from the total identity", () => {
35
+ const usage = normalizeUsage({
30
36
  prompt_tokens: 143,
31
37
  completion_tokens: 1,
32
38
  total_tokens: 441,
33
39
  prompt_tokens_details: { cached_tokens: 128 },
34
40
  completion_tokens_details: { reasoning_tokens: 297 },
35
41
  });
36
- assert.deepEqual(u, { prompt: 143, completion: 1, reasoning: 297, cached: 128, total: 441 });
37
- assert.equal(u.completion, 1); // visible output preserved, NOT zeroed by the subtraction
38
- assert.equal(u.completion + u.reasoning, 298); // billable output = visible + reasoning
39
- });
40
-
41
- test("normalizeUsage: cached read from prompt_tokens_details (OpenAI nesting)", () => {
42
- const u = normalizeUsage({ prompt_tokens: 50, completion_tokens: 10, total_tokens: 60, prompt_tokens_details: { cached_tokens: 30 } });
43
- assert.equal(u.cached, 30);
42
+ assert.deepEqual(usage, {
43
+ inputTokens: 143,
44
+ outputTokens: 298,
45
+ totalTokens: 441,
46
+ inputTokenDetails: { cacheReadTokens: 128 },
47
+ outputTokenDetails: { textTokens: 1, reasoningTokens: 297 },
48
+ });
44
49
  });
45
50
 
46
- test("normalizeUsage: top-level cached_tokens still honored", () => {
47
- const u = normalizeUsage({ prompt_tokens: 50, completion_tokens: 10, total_tokens: 60, cached_tokens: 12 });
48
- assert.equal(u.cached, 12);
51
+ test("normalizeUsage maps cache-read spellings without inventing uncached tokens", () => {
52
+ assert.deepEqual(normalizeUsage({
53
+ prompt_tokens: 50,
54
+ completion_tokens: 10,
55
+ total_tokens: 60,
56
+ prompt_tokens_details: { cached_tokens: 30 },
57
+ })?.inputTokenDetails, { cacheReadTokens: 30 });
58
+ assert.deepEqual(normalizeUsage({
59
+ prompt_tokens: 50,
60
+ completion_tokens: 10,
61
+ total_tokens: 60,
62
+ cached_tokens: 12,
63
+ })?.inputTokenDetails, { cacheReadTokens: 12 });
49
64
  });
50
65
 
51
- test("#157: normalizeUsage maps DeepSeek's prompt cache hit count", () => {
52
- const u = normalizeUsage({
66
+ test("#157: normalizeUsage maps DeepSeek cache hit and miss counts", () => {
67
+ assert.deepEqual(normalizeUsage({
53
68
  prompt_tokens: 50,
54
69
  prompt_cache_hit_tokens: 30,
55
70
  prompt_cache_miss_tokens: 20,
56
71
  completion_tokens: 10,
57
72
  total_tokens: 60,
73
+ }), {
74
+ inputTokens: 50,
75
+ outputTokens: 10,
76
+ totalTokens: 60,
77
+ inputTokenDetails: { noCacheTokens: 20, cacheReadTokens: 30 },
58
78
  });
59
- assert.deepEqual(u, { prompt: 50, completion: 10, reasoning: 0, cached: 30, total: 60 });
60
- });
61
-
62
- test("normalizeUsage: no reasoning — plain prompt+completion", () => {
63
- const u = normalizeUsage({ prompt_tokens: 10, completion_tokens: 20, total_tokens: 30 });
64
- assert.deepEqual(u, { prompt: 10, completion: 20, reasoning: 0, cached: 0, total: 30 });
65
- });
66
-
67
- test("normalizeUsage: missing total is reconstructed, never negative reasoning", () => {
68
- const u = normalizeUsage({ prompt_tokens: 10, completion_tokens: 20 });
69
- assert.deepEqual(u, { prompt: 10, completion: 20, reasoning: 0, cached: 0, total: 30 });
70
79
  });
71
80
 
72
- test("normalizeUsage: absent usage → all zeros", () => {
73
- assert.deepEqual(normalizeUsage(null), { prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 });
74
- assert.deepEqual(normalizeUsage(undefined), { prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 });
75
- });
76
-
77
- // -- Fireworks-style: reasoning shipped as TEXT, folded into completion, not itemized --
78
-
79
- test("normalizeUsage: fireworks folds reasoning into completion -- re-split by text proportion, sum preserved", () => {
80
- // total = prompt + completion (no gap), reasoning_tokens absent, but 750 vs 250
81
- // chars of reasoning vs content came back. Split completion 75/25; cost base held.
82
- const u = normalizeUsage(
83
- { prompt_tokens: 100, completion_tokens: 1000, total_tokens: 1100 },
84
- "r".repeat(750),
85
- "c".repeat(250),
86
- );
87
- assert.deepEqual(u, { prompt: 100, completion: 250, reasoning: 750, cached: 0, total: 1100 });
88
- assert.equal(u.completion + u.reasoning, 1000); // billable output byte-identical
89
- assert.equal(u.prompt + u.completion + u.reasoning, u.total); // invariant
90
- });
91
-
92
- test("normalizeUsage: pure-reasoning turn (empty content) attributes all completion to reasoning", () => {
93
- // The run52 runaway shape: 0 visible content, the whole budget spent reasoning.
94
- const u = normalizeUsage(
95
- { prompt_tokens: 100, completion_tokens: 500, total_tokens: 600 },
96
- "t".repeat(9000),
97
- "",
98
- );
99
- assert.deepEqual(u, { prompt: 100, completion: 0, reasoning: 500, cached: 0, total: 600 });
100
- });
101
-
102
- test("normalizeUsage: text args never perturb the itemized (reasoning_tokens) path", () => {
103
- // OpenAI o-series reports reasoning_tokens -> that split wins, text is ignored.
104
- const u = normalizeUsage(
105
- { prompt_tokens: 10, completion_tokens: 100, total_tokens: 110, completion_tokens_details: { reasoning_tokens: 40 } },
106
- "r".repeat(999), "c".repeat(1),
107
- );
108
- assert.deepEqual(u, { prompt: 10, completion: 60, reasoning: 40, cached: 0, total: 110 });
109
- });
110
-
111
- test("normalizeUsage: text args never perturb the Gemini gap path (gap already yields reasoning)", () => {
112
- // A real total gap means reasoning is itemized-by-subtraction; do not re-split.
113
- const u = normalizeUsage(
114
- { prompt_tokens: 19, completion_tokens: 285, total_tokens: 1165 },
115
- "r".repeat(500), "c".repeat(500),
116
- );
117
- assert.equal(u.reasoning, 861); // from the gap, NOT a text re-split
118
- assert.equal(u.completion, 285);
119
- });
120
-
121
- test("normalizeUsage: no total reported -> re-split skipped, reasoning stays 0 (cannot split an unknown base)", () => {
122
- const u = normalizeUsage(
123
- { prompt_tokens: 10, completion_tokens: 20 },
124
- "r".repeat(500), "c".repeat(500),
125
- );
126
- assert.deepEqual(u, { prompt: 10, completion: 20, reasoning: 0, cached: 0, total: 30 });
127
- });
128
-
129
- // — calculateCostUsd —
130
-
131
- test("calculateCostUsd: bills reasoning at the USD-per-million output rate", () => {
132
- // 100 input, 0 cached, 50 completion + 200 reasoning = 250 output.
133
- const usage = { prompt: 100, completion: 50, reasoning: 200, cached: 0, total: 350 };
134
- assert.equal(calculateCostUsd(usage, { input: 1, output: 10, cached: 0 }), 0.0026);
81
+ test("normalizeUsage derives only exact totals", () => {
82
+ assert.deepEqual(normalizeUsage({ prompt_tokens: 10, completion_tokens: 20 }), {
83
+ inputTokens: 10,
84
+ outputTokens: 20,
85
+ totalTokens: 30,
86
+ });
87
+ assert.deepEqual(normalizeUsage({ prompt_tokens: 10, total_tokens: 30 }), {
88
+ inputTokens: 10,
89
+ outputTokens: 20,
90
+ totalTokens: 30,
91
+ });
135
92
  });
136
93
 
137
- test("calculateCostUsd: cached prompt billed at the cache rate, remainder at input", () => {
138
- const usage = { prompt: 1000, completion: 0, reasoning: 0, cached: 400, total: 1000 };
139
- assert.equal(calculateCostUsd(usage, { input: 5, output: 99, cached: 1 }), 0.0034);
94
+ test("normalizeUsage preserves unknown usage as absence", () => {
95
+ assert.equal(normalizeUsage(null), undefined);
96
+ assert.equal(normalizeUsage(undefined), undefined);
140
97
  });
141
98
 
142
- test("calculateCostUsd: zero rates → 0", () => {
143
- const usage = { prompt: 9, completion: 9, reasoning: 9, cached: 9, total: 27 };
144
- assert.equal(calculateCostUsd(usage, { input: 0, output: 0, cached: 0 }), 0);
99
+ test("normalizeUsage never apportions tokens from reasoning or content length", () => {
100
+ assert.deepEqual(normalizeUsage({
101
+ prompt_tokens: 100,
102
+ completion_tokens: 1000,
103
+ total_tokens: 1100,
104
+ }), {
105
+ inputTokens: 100,
106
+ outputTokens: 1000,
107
+ totalTokens: 1100,
108
+ });
145
109
  });
146
110
 
147
- test("calculateCostUsdDecimal preserves Models.dev rates without floating-point artifacts", () => {
148
- const usage = { prompt: 1_000, completion: 100, reasoning: 50, cached: 400, total: 1_150 };
149
- assert.equal(
150
- calculateCostUsdDecimal(usage, { input: 0.14, output: 0.28, cached: 0.0028 }),
151
- "0.00012712",
152
- );
111
+ test("calculateCostUsdDecimal bills all output, including reasoning, at the output rate", () => {
112
+ assert.equal(calculateCostUsdDecimal({
113
+ inputTokens: 100,
114
+ outputTokens: 250,
115
+ totalTokens: 350,
116
+ outputTokenDetails: { textTokens: 50, reasoningTokens: 200 },
117
+ }, { input: 1, output: 10 }), "0.0026");
118
+ });
119
+
120
+ test("calculateCostUsdDecimal applies distinct cache-read and cache-write rates", () => {
121
+ assert.equal(calculateCostUsdDecimal({
122
+ inputTokens: 1000,
123
+ outputTokens: 0,
124
+ totalTokens: 1000,
125
+ inputTokenDetails: {
126
+ noCacheTokens: 500,
127
+ cacheReadTokens: 400,
128
+ cacheWriteTokens: 100,
129
+ },
130
+ }, { input: 5, output: 99, cacheRead: 1, cacheWrite: 8 }), "0.0037");
131
+ });
132
+
133
+ test("calculateCostUsdDecimal returns unknown when a differently-priced category is absent", () => {
134
+ assert.equal(calculateCostUsdDecimal({
135
+ inputTokens: 1000,
136
+ outputTokens: 0,
137
+ totalTokens: 1000,
138
+ }, { input: 5, output: 99, cacheRead: 1 }), null);
139
+ });
140
+
141
+ test("calculateCostUsdDecimal preserves Models.dev decimals without floating-point artifacts", () => {
142
+ assert.equal(calculateCostUsdDecimal({
143
+ inputTokens: 1000,
144
+ outputTokens: 150,
145
+ totalTokens: 1150,
146
+ inputTokenDetails: { cacheReadTokens: 400 },
147
+ outputTokenDetails: { textTokens: 100, reasoningTokens: 50 },
148
+ }, { input: 0.14, output: 0.28, cacheRead: 0.0028 }), "0.00012712");
153
149
  });