@plurnk/plurnk-providers 1.4.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. package/.env.defaults +40 -34
  2. package/README.md +3 -0
  3. package/SPEC.md +153 -62
  4. package/dist/AiSdkProvider.d.ts +19 -25
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +353 -120
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +7 -13
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +36 -8
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +2 -21
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +19 -14
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/accounting.d.ts +6 -0
  17. package/dist/accounting.d.ts.map +1 -0
  18. package/dist/accounting.js +168 -0
  19. package/dist/accounting.js.map +1 -0
  20. package/dist/aiSdkTransport.d.ts +11 -3
  21. package/dist/aiSdkTransport.d.ts.map +1 -1
  22. package/dist/aiSdkTransport.js +198 -29
  23. package/dist/aiSdkTransport.js.map +1 -1
  24. package/dist/catalogProvider.d.ts +7 -2
  25. package/dist/catalogProvider.d.ts.map +1 -1
  26. package/dist/catalogProvider.js +32 -26
  27. package/dist/catalogProvider.js.map +1 -1
  28. package/dist/compatibleProvider.d.ts.map +1 -1
  29. package/dist/compatibleProvider.js +18 -7
  30. package/dist/compatibleProvider.js.map +1 -1
  31. package/dist/cost.d.ts +10 -10
  32. package/dist/cost.d.ts.map +1 -1
  33. package/dist/cost.js +88 -43
  34. package/dist/cost.js.map +1 -1
  35. package/dist/env.d.ts +5 -7
  36. package/dist/env.d.ts.map +1 -1
  37. package/dist/env.js +30 -32
  38. package/dist/env.js.map +1 -1
  39. package/dist/errors.d.ts +14 -2
  40. package/dist/errors.d.ts.map +1 -1
  41. package/dist/errors.js +60 -2
  42. package/dist/errors.js.map +1 -1
  43. package/dist/index.d.ts +4 -4
  44. package/dist/index.d.ts.map +1 -1
  45. package/dist/index.js +3 -2
  46. package/dist/index.js.map +1 -1
  47. package/dist/ollama.js +3 -3
  48. package/dist/ollama.js.map +1 -1
  49. package/dist/sdkModels.d.ts +6 -0
  50. package/dist/sdkModels.d.ts.map +1 -1
  51. package/dist/sdkModels.js +46 -3
  52. package/dist/sdkModels.js.map +1 -1
  53. package/dist/types.d.ts +40 -29
  54. package/dist/types.d.ts.map +1 -1
  55. package/dist/usage.d.ts +21 -4
  56. package/dist/usage.d.ts.map +1 -1
  57. package/dist/usage.js +188 -74
  58. package/dist/usage.js.map +1 -1
  59. package/package.json +9 -7
  60. package/src/AiSdkProvider.test.ts +1039 -182
  61. package/src/AiSdkProvider.ts +428 -141
  62. package/src/Mock.test.ts +37 -12
  63. package/src/Mock.ts +46 -12
  64. package/src/Pool.test.ts +19 -6
  65. package/src/Pool.ts +20 -16
  66. package/src/ProviderRegistry.test.ts +16 -11
  67. package/src/accounting.test.ts +94 -0
  68. package/src/accounting.ts +190 -0
  69. package/src/aiSdkTransport.test.ts +42 -49
  70. package/src/aiSdkTransport.ts +218 -32
  71. package/src/boundaries.test.ts +2 -0
  72. package/src/catalogProvider.test.ts +271 -24
  73. package/src/catalogProvider.ts +44 -28
  74. package/src/compatibleProvider.test.ts +6 -3
  75. package/src/compatibleProvider.ts +20 -7
  76. package/src/cost.test.ts +55 -35
  77. package/src/cost.ts +110 -54
  78. package/src/defaults.test.ts +13 -3
  79. package/src/env.test.ts +50 -26
  80. package/src/env.ts +43 -42
  81. package/src/errors.test.ts +47 -2
  82. package/src/errors.ts +68 -3
  83. package/src/index.ts +21 -5
  84. package/src/ollama.test.ts +4 -1
  85. package/src/ollama.ts +3 -3
  86. package/src/sdkModels.test.ts +94 -3
  87. package/src/sdkModels.ts +53 -3
  88. package/src/types.ts +91 -33
  89. package/src/usage.test.ts +112 -108
  90. package/src/usage.ts +233 -84
package/src/Mock.test.ts CHANGED
@@ -3,6 +3,7 @@ import { strict as assert } from "node:assert";
3
3
  import Mock from "./Mock.ts";
4
4
  import type { Provider } from "./types.ts";
5
5
  import type { MockResponse } from "./Mock.ts";
6
+ import { ProviderError } from "./errors.ts";
6
7
 
7
8
  const build = (responses: MockResponse[] = [{ assistant: { content: "hi", reasoning: null } }]) =>
8
9
  new Mock({ contextWindow: 100000, responses });
@@ -34,19 +35,23 @@ test("Mock: prompt counting is exact for its declared mock vocabulary", async ()
34
35
  });
35
36
  });
36
37
 
37
- test("Mock: calculateCost returns its deliberate zero estimate", () => {
38
- const m = build();
39
- assert.equal(m.calculateCost({ prompt: 100, completion: 20, reasoning: 10, cached: 5, total: 130 }), 0);
40
- });
41
-
42
38
  // — Transport ({§provider-interface}) —
43
39
 
44
40
  test("Mock: generate resolves a valid ProviderResponse shape", async () => {
45
41
  const m = build([{ assistant: { content: "hello", reasoning: "cot" } }]);
46
- const { assistant, assistantRaw } = await m.generate({ messages: [] });
42
+ const { assistant, assistantRaw, accounting } = await m.generate({ messages: [] });
47
43
  assert.equal(assistant.content, "hello");
48
44
  assert.equal(assistant.reasoning, "cot");
49
- assert.deepEqual(assistant.usage, { prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 });
45
+ assert.deepEqual(accounting[0]?.usage, {
46
+ inputTokens: 0,
47
+ outputTokens: 0,
48
+ totalTokens: 0,
49
+ });
50
+ assert.deepEqual(accounting[0]?.cost, {
51
+ kind: "estimated",
52
+ amount: { amount: "0", currency: "USD" },
53
+ source: "mock provider fixture",
54
+ });
50
55
  assert.equal(assistant.finishReason, "stop");
51
56
  assert.equal(assistant.model, "mock");
52
57
  assert.equal(assistantRaw, null); // present, defaulted
@@ -57,16 +62,20 @@ test("Mock: generate applies caller-supplied overrides", async () => {
57
62
  assistant: {
58
63
  content: "x",
59
64
  reasoning: null,
60
- usage: { prompt: 1, completion: 2, reasoning: 0, cached: 0, total: 3 },
61
65
  finishReason: "length",
62
66
  model: "mock-xl",
63
67
  },
68
+ usage: { inputTokens: 1, outputTokens: 2, totalTokens: 3 },
64
69
  assistantRaw: { wire: true },
65
70
  }]);
66
- const { assistant, assistantRaw } = await m.generate({ messages: [] });
71
+ const { assistant, assistantRaw, accounting } = await m.generate({ messages: [] });
67
72
  assert.equal(assistant.finishReason, "length");
68
73
  assert.equal(assistant.model, "mock-xl");
69
- assert.deepEqual(assistant.usage, { prompt: 1, completion: 2, reasoning: 0, cached: 0, total: 3 });
74
+ assert.deepEqual(accounting[0]?.usage, {
75
+ inputTokens: 1,
76
+ outputTokens: 2,
77
+ totalTokens: 3,
78
+ });
70
79
  assert.deepEqual(assistantRaw, { wire: true });
71
80
  });
72
81
 
@@ -116,9 +125,25 @@ test("Mock: remaining decrements as responses are consumed", async () => {
116
125
  assert.equal(m.remaining, 1);
117
126
  });
118
127
 
119
- test("Mock: exhausted queue throws a specific error", async () => {
128
+ test("Mock: exhausted queue throws a ProviderError carrying its settled accounting", async () => {
120
129
  const m = build([]);
121
- await assert.rejects(() => m.generate({ messages: [] }), /exhausted/);
130
+ await assert.rejects(
131
+ () => m.generate({ messages: [] }),
132
+ (error: unknown) => {
133
+ assert.ok(error instanceof ProviderError);
134
+ assert.match(error.message, /exhausted/);
135
+ assert.deepEqual(error.accounting, [{
136
+ provider: "provider:mock",
137
+ model: "mock",
138
+ outcome: "error",
139
+ cost: {
140
+ kind: "unknown",
141
+ reason: "mock provider exhausted before producing a response",
142
+ },
143
+ }]);
144
+ return true;
145
+ },
146
+ );
122
147
  });
123
148
 
124
149
  // -- {§provider-generation-envelope} --
package/src/Mock.ts CHANGED
@@ -5,15 +5,14 @@
5
5
  // Provider contract. Production providers don't expose the `ops` escape
6
6
  // hatch — that's an intg-only convenience.
7
7
 
8
- import type { ChatMessage, FinishReason, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderAssistant, ProviderEncryptedReasoningItem, ProviderResponse, ProviderUsage } from "./types.ts";
9
- import type { ProviderCost } from "@plurnk/plurnk-contracts";
8
+ import type { ChatMessage, FinishReason, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderAssistant, ProviderCost, ProviderEncryptedReasoningItem, ProviderRequestAccounting, ProviderResponse, ProviderUsage } from "./types.ts";
10
9
  import { resolveEnvelopeFromEnv } from "./env.ts";
10
+ import { validateProviderRequestAccounting } from "./accounting.ts";
11
+ import { ProviderError } from "./errors.ts";
11
12
 
12
13
  export type MockAssistant = {
13
14
  content: string;
14
15
  reasoning: string | null;
15
- // Partial — omitted fields fall back to DEFAULT_USAGE (e.g. reasoning: 0).
16
- usage?: Partial<ProviderUsage>;
17
16
  finishReason?: FinishReason;
18
17
  model?: string;
19
18
  // Provider-normalized encrypted reasoning fixture.
@@ -28,6 +27,9 @@ export type MockAssistant = {
28
27
  export type MockResponse = {
29
28
  assistant: MockAssistant;
30
29
  assistantRaw?: unknown;
30
+ // Partial — omitted fields fall back to the deliberate zero fixture.
31
+ usage?: Partial<ProviderUsage>;
32
+ cost?: ProviderCost;
31
33
  grammarEvidence?: GrammarEvidence;
32
34
  };
33
35
 
@@ -35,7 +37,12 @@ export type MockResponse = {
35
37
  export type MockReturnedAssistant = ProviderAssistant & { ops?: unknown[] };
36
38
  export type MockReturnedResponse = ProviderResponse & { assistant: MockReturnedAssistant };
37
39
 
38
- const DEFAULT_USAGE: ProviderUsage = { prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 };
40
+ const DEFAULT_USAGE: ProviderUsage = {
41
+ inputTokens: 0,
42
+ outputTokens: 0,
43
+ totalTokens: 0,
44
+ };
45
+ type MockGenerateArgs = Omit<Parameters<Provider["generate"]>[0], "workerId"> & { workerId?: string };
39
46
 
40
47
  export default class Mock implements Provider {
41
48
  #contextWindow: number | null;
@@ -75,22 +82,48 @@ export default class Mock implements Provider {
75
82
  };
76
83
  }
77
84
 
78
- // Mock is free.
79
- calculateCost(_usage: ProviderUsage): number { return 0; }
80
- calculateCharge(_usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }> { return { kind: "free", source: "mock provider" }; }
81
-
82
- async generate({ signal, grammar }: { messages: ChatMessage[]; workerId?: string; signal?: AbortSignal; grammar?: string }): Promise<MockReturnedResponse> {
85
+ async generate({ signal, grammar, observeRequest }: MockGenerateArgs): Promise<MockReturnedResponse> {
83
86
  // Honor abort before consuming the queue — an aborted call makes no
84
87
  // "wire call" and must not exhaust a queued response
85
88
  // ({§provider-failure-normalization}).
86
89
  signal?.throwIfAborted();
90
+ const settle = await observeRequest?.({ provider: "provider:mock", model: this.model });
87
91
  const next = this.#queue.shift();
88
- if (next === undefined) throw new Error("Mock provider exhausted: no more queued responses");
92
+ if (next === undefined) {
93
+ const accounting = validateProviderRequestAccounting({
94
+ provider: "provider:mock",
95
+ model: this.model,
96
+ outcome: "error",
97
+ cost: { kind: "unknown", reason: "mock provider exhausted before producing a response" },
98
+ });
99
+ await settle?.(accounting);
100
+ throw new ProviderError(
101
+ "mock",
102
+ "invalid_response",
103
+ "Mock provider exhausted: no more queued responses",
104
+ { accounting: [accounting] },
105
+ );
106
+ }
89
107
  const a = next.assistant;
108
+ const usage: ProviderUsage = {
109
+ ...DEFAULT_USAGE,
110
+ ...next.usage,
111
+ };
112
+ const requestAccounting: ProviderRequestAccounting = validateProviderRequestAccounting({
113
+ provider: "provider:mock",
114
+ model: a.model ?? this.model,
115
+ outcome: "response",
116
+ usage,
117
+ cost: next.cost ?? {
118
+ kind: "estimated",
119
+ amount: { amount: "0", currency: "USD" },
120
+ source: "mock provider fixture",
121
+ },
122
+ });
123
+ await settle?.(requestAccounting);
90
124
  const assistant: MockReturnedAssistant = {
91
125
  content: a.content,
92
126
  reasoning: a.reasoning,
93
- usage: { ...DEFAULT_USAGE, ...a.usage },
94
127
  ...(a.reasoningEncrypted !== undefined ? { reasoningEncrypted: a.reasoningEncrypted } : {}),
95
128
  finishReason: a.finishReason ?? "stop",
96
129
  model: a.model ?? "mock",
@@ -103,6 +136,7 @@ export default class Mock implements Provider {
103
136
  return {
104
137
  assistant,
105
138
  assistantRaw: next.assistantRaw ?? null,
139
+ accounting: [requestAccounting],
106
140
  ...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
107
141
  };
108
142
  }
package/src/Pool.test.ts CHANGED
@@ -7,13 +7,28 @@ import { resetEmittedWarnings } from "./warnings.ts";
7
7
 
8
8
  test.afterEach(() => { resetEmittedWarnings(); });
9
9
 
10
- const RESP = { assistant: { usage: { prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 2 } }, assistantRaw: null } as unknown as ProviderResponse;
10
+ const RESP: ProviderResponse = {
11
+ assistant: {
12
+ content: "ok",
13
+ reasoning: null,
14
+ finishReason: "stop",
15
+ model: "gemma",
16
+ },
17
+ assistantRaw: null,
18
+ accounting: [{
19
+ provider: "provider:test",
20
+ model: "gemma",
21
+ outcome: "response",
22
+ usage: { inputTokens: 1, outputTokens: 1, totalTokens: 2 },
23
+ cost: { kind: "unknown", reason: "test fixture has no cost" },
24
+ }],
25
+ };
11
26
 
12
27
  type FakeOpts = {
13
28
  model?: string; window?: number | null; servedModel?: string;
14
29
  constrainsOutput?: boolean; requiresMaxTokens?: boolean;
15
30
  reasoningReserve?: number | null; completionReserve?: number | null;
16
- tokenize?: boolean; cost?: number; throws?: Error;
31
+ tokenize?: boolean; throws?: Error;
17
32
  promptMeasurement?: PromptTokenMeasurement;
18
33
  };
19
34
  // A fake backend that records which workers it served, and optionally throws.
@@ -33,7 +48,6 @@ const backend = (opts: FakeOpts = {}) => {
33
48
  tokens: messages.reduce((sum, { content }) => sum + content.length, 0),
34
49
  source: "test:exact",
35
50
  }),
36
- calculateCost: () => opts.cost ?? 0,
37
51
  generate: async (args: Parameters<Provider["generate"]>[0]): Promise<ProviderResponse> => {
38
52
  served.push(args.workerId);
39
53
  if (opts.throws !== undefined) throw opts.throws;
@@ -94,12 +108,11 @@ test("Pool: tokenize is exposed iff every backend has it", () => {
94
108
  assert.equal(new Pool([backend({ tokenize: true }).b, backend({ tokenize: false }).b]).tokenize, undefined);
95
109
  });
96
110
 
97
- test("Pool: prompt counting + calculateCost delegate to a backend", async () => {
98
- const p = new Pool([backend({ cost: 42 }).b]);
111
+ test("Pool: prompt counting delegates conservatively to its backends", async () => {
112
+ const p = new Pool([backend().b]);
99
113
  assert.deepEqual(await p.countPromptTokens([{ role: "user", content: "abcd" }]), {
100
114
  kind: "exact", tokens: 4, source: "pool:test:exact",
101
115
  });
102
- assert.equal(p.calculateCost({ prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 }), 42);
103
116
  });
104
117
 
105
118
  test("Pool: prompt evidence is conservative across every routable backend", async () => {
package/src/Pool.ts CHANGED
@@ -1,6 +1,4 @@
1
- import type { Provider, ProviderResponse, ProviderUsage, ChatMessage, PromptTokenMeasurement } from "./types.ts";
2
- import type { ProviderCost } from "@plurnk/plurnk-contracts";
3
- import { resolveProviderCost } from "./cost.ts";
1
+ import type { Provider, ProviderRequestAccounting, ProviderResponse, ChatMessage, PromptTokenMeasurement } from "./types.ts";
4
2
  import { ProviderError, type ProviderErrorKind } from "./errors.ts";
5
3
  import { emitWarningOnce } from "./warnings.ts";
6
4
  import { assertPromptTokenMeasurement } from "./promptTokens.ts";
@@ -125,32 +123,38 @@ export default class Pool implements Provider {
125
123
  source: `pool:${sources}`,
126
124
  };
127
125
  }
128
- calculateCost(usage: ProviderUsage): number { return this.#backends[0].calculateCost(usage); }
129
- calculateCharge(usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }> {
130
- const backend = this.#backends[0];
131
- return resolveProviderCost(undefined, backend.calculateCharge?.(usage), () => backend.calculateCost(usage)) as Exclude<ProviderCost, { kind: "authoritative" }>;
132
- }
133
-
134
126
  // --- dispatch ---
135
127
 
136
- async generate(args: { messages: ChatMessage[]; workerId: string; primaryWorkerId?: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse> {
128
+ async generate(args: Parameters<Provider["generate"]>[0]): Promise<ProviderResponse> {
137
129
  const { workerId, signal } = args;
138
130
  if (workerId === undefined || workerId.length === 0) throw new Error("Pool.generate: workerId is required - affinity keys on it");
139
131
  const tried = new Set<number>();
132
+ const priorAccounting: ProviderRequestAccounting[] = [];
140
133
  let idx = this.#route(workerId);
141
- let lastErr: unknown;
142
134
  for (;;) {
143
135
  tried.add(idx);
144
136
  try {
145
- return await this.#backends[idx].generate(args);
137
+ const response = await this.#backends[idx].generate(args);
138
+ return priorAccounting.length === 0
139
+ ? response
140
+ : { ...response, accounting: [...priorAccounting, ...response.accounting] };
146
141
  } catch (err) {
147
- lastErr = err;
148
- if (signal?.aborted) throw err; // caller cancellation is never a failover
142
+ if (signal?.aborted) {
143
+ if (err instanceof ProviderError) err.prependAccounting(priorAccounting);
144
+ throw err;
145
+ }
149
146
  // Only a backend-AVAILABILITY failure overflows; auth/quota/content/
150
147
  // malformed fail the same on a peer, so they propagate.
151
- if (!(err instanceof ProviderError) || !OVERFLOW_KINDS.has(err.kind)) throw err;
148
+ if (!(err instanceof ProviderError) || !OVERFLOW_KINDS.has(err.kind)) {
149
+ if (err instanceof ProviderError) err.prependAccounting(priorAccounting);
150
+ throw err;
151
+ }
152
152
  const next = this.#nextUntried(tried);
153
- if (next === null) throw lastErr; // the whole fleet is unavailable
153
+ if (next === null) {
154
+ err.prependAccounting(priorAccounting);
155
+ throw err;
156
+ }
157
+ priorAccounting.push(...err.accounting);
154
158
  this.#affinity.set(workerId, next); // re-stick: the worker's cache moves with it
155
159
  idx = next;
156
160
  }
@@ -14,8 +14,10 @@ const mapOf = (entries: Record<string, string>, skipped: Record<string, string>
14
14
 
15
15
  const fullEnv = Object.freeze({
16
16
  PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000",
17
+ PLURNK_PROVIDERS_OPERATION_TIMEOUT: "2700000",
18
+ PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT: "600000",
17
19
  PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
18
- PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0", PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT: "512", PLURNK_PROVIDERS_PROMPT_CACHE_KEY: "1",
20
+ PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0", PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT: "512", PLURNK_PROVIDERS_CACHE_AFFINITY: "1", PLURNK_PROVIDERS_CACHE_WRITE_POLICY: "stable-system",
19
21
  OPENAI_BASE_URL: "http://x",
20
22
  });
21
23
 
@@ -271,30 +273,33 @@ test("{§deepseek-reasoning-request} #157: direct DeepSeek composes catalog fact
271
273
  assert.deepEqual(request.thinking, { type: "disabled" });
272
274
  assert.equal(provider.contextWindow, 1_000_000);
273
275
  assert.equal(response.assistant.content, "ok");
274
- assert.deepEqual(response.assistant.usage, {
275
- prompt: 10,
276
- completion: 2,
277
- reasoning: 0,
278
- cached: 8,
279
- total: 12,
276
+ assert.deepEqual(response.accounting[0]?.usage, {
277
+ inputTokens: 10,
278
+ outputTokens: 2,
279
+ totalTokens: 12,
280
+ inputTokenDetails: { noCacheTokens: 2, cacheReadTokens: 8 },
281
+ });
282
+ assert.deepEqual(response.accounting[0]?.cost, {
283
+ kind: "estimated",
284
+ amount: { amount: "0.0000008624", currency: "USD" },
285
+ source: "Models.dev catalog rates",
280
286
  });
281
- assert.ok(Math.abs(provider.calculateCost(response.assistant.usage) - 0.0000008624) < 1e-15);
282
287
  mock.restoreAll();
283
288
  });
284
289
 
285
- test("an explicit malformed operator override still fails at its owning contract", async () => {
290
+ test("an explicit malformed cache-affinity override still fails at its owning contract", async () => {
286
291
  await assert.rejects(
287
292
  () => instantiateProvider(
288
293
  "fireworks",
289
294
  {
290
295
  FIREWORKS_API_KEY: "fw",
291
- PLURNK_PROVIDERS_PROMPT_CACHE_KEY: "malformed",
296
+ PLURNK_PROVIDERS_CACHE_AFFINITY: "malformed",
292
297
  },
293
298
  "deepseek-v4-pro",
294
299
  async () => ({}),
295
300
  mapOf({}),
296
301
  ),
297
- /fireworks provider: PLURNK_PROVIDERS_PROMPT_CACHE_KEY must be "0" or "1"/,
302
+ /fireworks provider: PLURNK_PROVIDERS_CACHE_AFFINITY must be "0" or "1"/,
298
303
  );
299
304
  });
300
305
 
@@ -0,0 +1,94 @@
1
+ import assert from "node:assert/strict";
2
+ import test from "node:test";
3
+ import {
4
+ aggregateProviderAccounting,
5
+ plurnkCostNormalizer,
6
+ providerCostNormalizer,
7
+ } from "./accounting.ts";
8
+
9
+ const evidence = ({ providerMetadata, usage, charge }: {
10
+ providerMetadata?: unknown;
11
+ usage?: unknown;
12
+ charge?: unknown;
13
+ }) => ({
14
+ ...(providerMetadata === undefined ? {} : { providerMetadata }),
15
+ ...(usage === undefined ? {} : { usage }),
16
+ ...(charge === undefined ? {} : { charge }),
17
+ response: { id: "response-1" },
18
+ });
19
+
20
+ test("xAI response ticks normalize to a directly charged request", () => {
21
+ const normalize = providerCostNormalizer("@ai-sdk/xai");
22
+ assert.notEqual(normalize, undefined);
23
+ assert.deepEqual(normalize!(evidence({ usage: { cost_in_usd_ticks: 15_493_500 } })), {
24
+ kind: "charged",
25
+ amount: { amount: "15493500", currency: "USDTICK" },
26
+ usdEquivalent: "0.00154935",
27
+ source: "xAI response usage.cost_in_usd_ticks",
28
+ });
29
+ });
30
+
31
+ test("OpenRouter response cost normalizes without rate reconstruction", () => {
32
+ const normalize = providerCostNormalizer("@openrouter/ai-sdk-provider");
33
+ assert.notEqual(normalize, undefined);
34
+ assert.deepEqual(normalize!(evidence({ providerMetadata: { openrouter: { usage: { cost: 3.2e-7 } } } })), {
35
+ kind: "charged",
36
+ amount: { amount: "0.00000032", currency: "USD" },
37
+ source: "OpenRouter response usage.cost",
38
+ });
39
+ });
40
+
41
+ test("DeepInfra's documented response estimate remains estimated", () => {
42
+ const normalize = providerCostNormalizer("@ai-sdk/deepinfra");
43
+ assert.notEqual(normalize, undefined);
44
+ assert.deepEqual(normalize!(evidence({ usage: { estimated_cost: 5.04e-5 } })), {
45
+ kind: "estimated",
46
+ amount: { amount: "0.0000504", currency: "USD" },
47
+ source: "DeepInfra response usage.estimated_cost",
48
+ });
49
+ });
50
+
51
+ test("first-party charged evidence is validated at its adapter boundary", () => {
52
+ const charged = {
53
+ kind: "charged",
54
+ amount: { amount: "0.01", currency: "USD" },
55
+ source: "plurnk endpoint",
56
+ } as const;
57
+ assert.deepEqual(plurnkCostNormalizer(evidence({ charge: charged })), charged);
58
+ assert.equal(plurnkCostNormalizer(evidence({})), undefined);
59
+ });
60
+
61
+ test("response cost normalization is an explicit adapter capability", () => {
62
+ assert.equal(providerCostNormalizer("@ai-sdk/anthropic"), undefined);
63
+ assert.equal(providerCostNormalizer("@ai-sdk/xai")!(evidence({ usage: {} })), undefined);
64
+ assert.throws(
65
+ () => providerCostNormalizer("@ai-sdk/xai")!(evidence({ usage: { cost_in_usd_ticks: "1" } })),
66
+ /cost_in_usd_ticks must be numeric/,
67
+ );
68
+ });
69
+
70
+ test("aggregateProviderAccounting preserves request order and only sums known fields", () => {
71
+ const accounting = aggregateProviderAccounting([
72
+ {
73
+ provider: "provider:a",
74
+ model: "m",
75
+ outcome: "error",
76
+ status: 429,
77
+ cost: { kind: "unknown", reason: "no response accounting" },
78
+ },
79
+ {
80
+ provider: "provider:b",
81
+ model: "m",
82
+ outcome: "response",
83
+ usage: { inputTokens: 2, outputTokens: 3, totalTokens: 5 },
84
+ cost: {
85
+ kind: "charged",
86
+ amount: { amount: "0.25", currency: "USD" },
87
+ source: "provider b",
88
+ },
89
+ },
90
+ ]);
91
+ assert.deepEqual(accounting.requests.map(({ provider }) => provider), ["provider:a", "provider:b"]);
92
+ assert.equal(accounting.usage, null, "an unknown failed request prevents fabricated totals");
93
+ assert.equal(accounting.costUsd, null);
94
+ });
@@ -0,0 +1,190 @@
1
+ import type {
2
+ ChargedCost,
3
+ ProviderAccounting,
4
+ ProviderCost,
5
+ ProviderCostNormalizer,
6
+ ProviderRequestAccounting,
7
+ ProviderUsage,
8
+ } from "./types.ts";
9
+ import {
10
+ sumProviderCostsUsd,
11
+ validateChargedCost,
12
+ validateProviderCost,
13
+ } from "./cost.ts";
14
+ import { validateProviderUsage } from "./usage.ts";
15
+
16
+ const recordOf = (value: unknown): Record<string, unknown> | null =>
17
+ typeof value === "object" && value !== null && !Array.isArray(value)
18
+ ? value as Record<string, unknown>
19
+ : null;
20
+
21
+ const decimalFromNumber = (value: number, subject: string): string => {
22
+ if (!Number.isFinite(value) || value < 0) {
23
+ throw new TypeError(`${subject} must be a finite non-negative number`);
24
+ }
25
+ const source = String(value);
26
+ if (!/[eE]/.test(source)) return source;
27
+ const [coefficient, exponentSource] = source.toLowerCase().split("e");
28
+ const exponent = Number(exponentSource);
29
+ const [integer, fraction = ""] = coefficient!.split(".");
30
+ const digits = `${integer}${fraction}`;
31
+ const point = integer!.length + exponent;
32
+ if (point <= 0) return `0.${"0".repeat(-point)}${digits}`;
33
+ if (point >= digits.length) return `${digits}${"0".repeat(point - digits.length)}`;
34
+ return `${digits.slice(0, point)}.${digits.slice(point)}`;
35
+ };
36
+
37
+ const usdFromTicks = (ticks: number): string => {
38
+ if (!Number.isSafeInteger(ticks) || ticks < 0) {
39
+ throw new TypeError("xAI costInUsdTicks must be a non-negative safe integer");
40
+ }
41
+ const digits = String(ticks).padStart(11, "0");
42
+ const integer = digits.slice(0, -10).replace(/^0+(?=\d)/, "");
43
+ const fraction = digits.slice(-10).replace(/0+$/, "");
44
+ return fraction === "" ? integer : `${integer}.${fraction}`;
45
+ };
46
+
47
+ const xaiCost: ProviderCostNormalizer = ({ usage }) => {
48
+ const wireUsage = recordOf(usage);
49
+ if (wireUsage === null || !("cost_in_usd_ticks" in wireUsage)) return undefined;
50
+ const ticks = wireUsage.cost_in_usd_ticks;
51
+ if (typeof ticks !== "number") {
52
+ throw new TypeError("xAI usage.cost_in_usd_ticks must be numeric");
53
+ }
54
+ return {
55
+ kind: "charged",
56
+ amount: { amount: String(ticks), currency: "USDTICK" },
57
+ usdEquivalent: usdFromTicks(ticks),
58
+ source: "xAI response usage.cost_in_usd_ticks",
59
+ };
60
+ };
61
+
62
+ const openRouterCost: ProviderCostNormalizer = ({ providerMetadata }) => {
63
+ const usage = recordOf(recordOf(recordOf(providerMetadata)?.openrouter)?.usage);
64
+ if (usage === null || !("cost" in usage)) return undefined;
65
+ const cost = usage.cost;
66
+ if (typeof cost !== "number") throw new TypeError("OpenRouter usage.cost must be numeric");
67
+ return {
68
+ kind: "charged",
69
+ amount: { amount: decimalFromNumber(cost, "OpenRouter usage.cost"), currency: "USD" },
70
+ source: "OpenRouter response usage.cost",
71
+ };
72
+ };
73
+
74
+ const deepInfraCost: ProviderCostNormalizer = ({ usage }) => {
75
+ const wireUsage = recordOf(usage);
76
+ if (wireUsage === null || !("estimated_cost" in wireUsage)) return undefined;
77
+ const cost = wireUsage.estimated_cost;
78
+ if (typeof cost !== "number") throw new TypeError("DeepInfra usage.estimated_cost must be numeric");
79
+ return {
80
+ kind: "estimated",
81
+ amount: { amount: decimalFromNumber(cost, "DeepInfra usage.estimated_cost"), currency: "USD" },
82
+ source: "DeepInfra response usage.estimated_cost",
83
+ };
84
+ };
85
+
86
+ // The first-party endpoint owns the direct charged-cost wire field.
87
+ export const plurnkCostNormalizer: ProviderCostNormalizer = ({ charge }) => {
88
+ if (charge === undefined) return undefined;
89
+ return validateChargedCost(charge) as ChargedCost;
90
+ };
91
+
92
+ export const providerCostNormalizer = (
93
+ sdkPackage: string,
94
+ ): ProviderCostNormalizer | undefined => {
95
+ switch (sdkPackage) {
96
+ case "@ai-sdk/xai": return xaiCost;
97
+ case "@ai-sdk/deepinfra": return deepInfraCost;
98
+ case "@openrouter/ai-sdk-provider": return openRouterCost;
99
+ default: return undefined;
100
+ }
101
+ };
102
+
103
+ export const validateProviderRequestAccounting = (
104
+ value: unknown,
105
+ ): ProviderRequestAccounting => {
106
+ const request = recordOf(value);
107
+ if (request === null) throw new TypeError("provider request accounting must be an object");
108
+ if (typeof request.provider !== "string" || request.provider.length === 0) {
109
+ throw new TypeError("provider request accounting.provider must be non-empty");
110
+ }
111
+ if (typeof request.model !== "string" || request.model.length === 0) {
112
+ throw new TypeError("provider request accounting.model must be non-empty");
113
+ }
114
+ if (request.outcome !== "response" && request.outcome !== "error") {
115
+ throw new TypeError("provider request accounting.outcome must be response or error");
116
+ }
117
+ if (request.status !== undefined
118
+ && (!Number.isInteger(request.status) || (request.status as number) < 100 || (request.status as number) > 599)) {
119
+ throw new TypeError("provider request accounting.status must be an HTTP status");
120
+ }
121
+ if (request.usage !== undefined) validateProviderUsage(request.usage as ProviderUsage);
122
+ validateProviderCost(request.cost);
123
+ return value as ProviderRequestAccounting;
124
+ };
125
+
126
+ const sumKnown = (
127
+ requests: readonly ProviderRequestAccounting[],
128
+ read: (usage: ProviderUsage) => number | undefined,
129
+ ): number | undefined => {
130
+ const values = requests.map((request) => request.usage === undefined
131
+ ? undefined
132
+ : read(request.usage));
133
+ return values.some((value) => value === undefined)
134
+ ? undefined
135
+ : (values as number[]).reduce((sum, value) => sum + value, 0);
136
+ };
137
+
138
+ export const aggregateProviderAccounting = (
139
+ values: readonly ProviderRequestAccounting[],
140
+ ): ProviderAccounting => {
141
+ const requests = values.map(validateProviderRequestAccounting);
142
+ if (requests.length === 0) {
143
+ return {
144
+ requests: [],
145
+ usage: {
146
+ inputTokens: 0,
147
+ outputTokens: 0,
148
+ totalTokens: 0,
149
+ inputTokenDetails: {
150
+ noCacheTokens: 0,
151
+ cacheReadTokens: 0,
152
+ cacheWriteTokens: 0,
153
+ },
154
+ outputTokenDetails: { textTokens: 0, reasoningTokens: 0 },
155
+ },
156
+ costUsd: "0",
157
+ };
158
+ }
159
+
160
+ const inputTokens = sumKnown(requests, (usage) => usage.inputTokens);
161
+ const outputTokens = sumKnown(requests, (usage) => usage.outputTokens);
162
+ const totalTokens = sumKnown(requests, (usage) => usage.totalTokens);
163
+ const noCacheTokens = sumKnown(requests, (usage) => usage.inputTokenDetails?.noCacheTokens);
164
+ const cacheReadTokens = sumKnown(requests, (usage) => usage.inputTokenDetails?.cacheReadTokens);
165
+ const cacheWriteTokens = sumKnown(requests, (usage) => usage.inputTokenDetails?.cacheWriteTokens);
166
+ const textTokens = sumKnown(requests, (usage) => usage.outputTokenDetails?.textTokens);
167
+ const reasoningTokens = sumKnown(requests, (usage) => usage.outputTokenDetails?.reasoningTokens);
168
+ const inputTokenDetails = noCacheTokens === undefined
169
+ && cacheReadTokens === undefined && cacheWriteTokens === undefined
170
+ ? undefined
171
+ : { noCacheTokens, cacheReadTokens, cacheWriteTokens };
172
+ const outputTokenDetails = textTokens === undefined && reasoningTokens === undefined
173
+ ? undefined
174
+ : { textTokens, reasoningTokens };
175
+ const usage = inputTokens === undefined && outputTokens === undefined && totalTokens === undefined
176
+ && inputTokenDetails === undefined && outputTokenDetails === undefined
177
+ ? null
178
+ : validateProviderUsage({
179
+ ...(inputTokens === undefined ? {} : { inputTokens }),
180
+ ...(outputTokens === undefined ? {} : { outputTokens }),
181
+ ...(totalTokens === undefined ? {} : { totalTokens }),
182
+ ...(inputTokenDetails === undefined ? {} : { inputTokenDetails }),
183
+ ...(outputTokenDetails === undefined ? {} : { outputTokenDetails }),
184
+ });
185
+ return {
186
+ requests: [...requests],
187
+ usage,
188
+ costUsd: sumProviderCostsUsd(requests.map(({ cost }) => cost)),
189
+ };
190
+ };