@plurnk/plurnk-providers 1.3.4 → 1.3.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,242 @@
1
+ import assert from "node:assert/strict";
2
+ import { describe, it } from "node:test";
3
+ import { createOpenAICompatible } from "@ai-sdk/openai-compatible";
4
+ import { APICallError, streamText } from "ai";
5
+
6
+ const encoder = new TextEncoder();
7
+
8
+ const sseResponse = (...chunks: object[]): Response => new Response(
9
+ new ReadableStream({
10
+ start(controller) {
11
+ for (const chunk of chunks) {
12
+ controller.enqueue(encoder.encode(`data: ${JSON.stringify(chunk)}\n\n`));
13
+ }
14
+ controller.enqueue(encoder.encode("data: [DONE]\n\n"));
15
+ controller.close();
16
+ },
17
+ }),
18
+ { headers: { "content-type": "text/event-stream" } },
19
+ );
20
+
21
+ describe("AI SDK adapter spike", () => {
22
+ it("preserves PLURNK request extensions and complete stream evidence", async () => {
23
+ let requestBody: Record<string, unknown> | undefined;
24
+ const rawChunks = [
25
+ {
26
+ id: "response-1",
27
+ object: "chat.completion.chunk",
28
+ created: 1,
29
+ model: "test-model",
30
+ choices: [{
31
+ index: 0,
32
+ delta: { role: "assistant", reasoning_content: "because " },
33
+ finish_reason: null,
34
+ }],
35
+ },
36
+ {
37
+ id: "response-1",
38
+ object: "chat.completion.chunk",
39
+ created: 1,
40
+ model: "test-model",
41
+ choices: [{
42
+ index: 0,
43
+ delta: { content: "answer" },
44
+ finish_reason: null,
45
+ }],
46
+ },
47
+ {
48
+ id: "response-1",
49
+ object: "chat.completion.chunk",
50
+ created: 1,
51
+ model: "test-model",
52
+ choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
53
+ usage: {
54
+ prompt_tokens: 3,
55
+ completion_tokens: 5,
56
+ total_tokens: 8,
57
+ completion_tokens_details: { reasoning_tokens: 2 },
58
+ },
59
+ },
60
+ ];
61
+ const provider = createOpenAICompatible({
62
+ name: "spike",
63
+ baseURL: "https://example.test/v1",
64
+ apiKey: "test-key",
65
+ includeUsage: true,
66
+ fetch: async (_input, init) => {
67
+ requestBody = JSON.parse(String(init?.body)) as Record<string, unknown>;
68
+ return sseResponse(...rawChunks);
69
+ },
70
+ });
71
+
72
+ const result = streamText({
73
+ model: provider("test-model"),
74
+ messages: [{ role: "user", content: "question" }],
75
+ maxOutputTokens: 64,
76
+ temperature: 0.25,
77
+ maxRetries: 0,
78
+ timeout: { totalMs: 1_000, chunkMs: 500 },
79
+ includeRawChunks: true,
80
+ providerOptions: {
81
+ spike: {
82
+ grammar: "root ::= \"answer\"",
83
+ id_slot: 2,
84
+ enable_thinking: true,
85
+ service_tier: "flex",
86
+ },
87
+ },
88
+ });
89
+ const parts = [];
90
+ for await (const part of result.fullStream) parts.push(part);
91
+
92
+ assert.equal(requestBody?.model, "test-model");
93
+ assert.equal(requestBody?.max_tokens, 64);
94
+ assert.equal(requestBody?.temperature, 0.25);
95
+ assert.equal(requestBody?.grammar, "root ::= \"answer\"");
96
+ assert.equal(requestBody?.id_slot, 2);
97
+ assert.equal(requestBody?.enable_thinking, true);
98
+ assert.equal(requestBody?.service_tier, "flex");
99
+ assert.equal(requestBody?.stream, true);
100
+ assert.deepEqual(requestBody?.stream_options, { include_usage: true });
101
+
102
+ assert.equal(await result.text, "answer");
103
+ assert.equal(await result.reasoningText, "because ");
104
+ assert.equal(await result.finishReason, "stop");
105
+ assert.equal(await result.rawFinishReason, "stop");
106
+ const usage = await result.usage;
107
+ assert.equal(usage.inputTokens, 3);
108
+ assert.equal(usage.outputTokens, 5);
109
+ assert.equal(usage.outputTokenDetails.reasoningTokens, 2);
110
+ assert.equal(usage.outputTokenDetails.textTokens, 3);
111
+ assert.equal(usage.totalTokens, 8);
112
+ assert.deepEqual(
113
+ parts.filter((part) => part.type === "raw").map((part) => part.rawValue),
114
+ rawChunks,
115
+ );
116
+ });
117
+
118
+ it("surfaces typed HTTP failures without retrying", async () => {
119
+ let requests = 0;
120
+ const provider = createOpenAICompatible({
121
+ name: "spike",
122
+ baseURL: "https://example.test/v1",
123
+ apiKey: "test-key",
124
+ fetch: async () => {
125
+ requests += 1;
126
+ return new Response(
127
+ JSON.stringify({ error: { message: "rate limited", type: "rate_limit" } }),
128
+ {
129
+ status: 429,
130
+ headers: {
131
+ "content-type": "application/json",
132
+ "retry-after": "3",
133
+ "x-request-id": "request-1",
134
+ },
135
+ },
136
+ );
137
+ },
138
+ });
139
+ const result = streamText({
140
+ model: provider("test-model"),
141
+ prompt: "question",
142
+ maxRetries: 0,
143
+ onError: () => {},
144
+ });
145
+
146
+ const parts = [];
147
+ for await (const part of result.fullStream) parts.push(part);
148
+ const errorPart = parts.find((part) => part.type === "error");
149
+ assert.ok(errorPart?.type === "error");
150
+ assert.ok(APICallError.isInstance(errorPart.error));
151
+ assert.equal(errorPart.error.statusCode, 429);
152
+ assert.equal(errorPart.error.isRetryable, true);
153
+ assert.equal(errorPart.error.responseHeaders?.["retry-after"], "3");
154
+ assert.equal(errorPart.error.responseHeaders?.["x-request-id"], "request-1");
155
+ assert.match(errorPart.error.responseBody ?? "", /rate limited/);
156
+ assert.equal(requests, 1);
157
+ });
158
+
159
+ it("enforces stream-idle timeout through the transport contract", async () => {
160
+ const provider = createOpenAICompatible({
161
+ name: "spike",
162
+ baseURL: "https://example.test/v1",
163
+ apiKey: "test-key",
164
+ fetch: async () => new Response(
165
+ new ReadableStream({
166
+ start(controller) {
167
+ controller.enqueue(encoder.encode(`data: ${JSON.stringify({
168
+ id: "response-1",
169
+ object: "chat.completion.chunk",
170
+ created: 1,
171
+ model: "test-model",
172
+ choices: [{
173
+ index: 0,
174
+ delta: { role: "assistant", content: "started" },
175
+ finish_reason: null,
176
+ }],
177
+ })}\n\n`));
178
+ setTimeout(() => controller.close(), 100);
179
+ },
180
+ }),
181
+ { headers: { "content-type": "text/event-stream" } },
182
+ ),
183
+ });
184
+ const result = streamText({
185
+ model: provider("test-model"),
186
+ prompt: "question",
187
+ maxRetries: 0,
188
+ timeout: { totalMs: 500, chunkMs: 25 },
189
+ onError: () => {},
190
+ });
191
+
192
+ await assert.rejects(
193
+ () => Promise.resolve(result.text),
194
+ (error: unknown) => {
195
+ assert.match(String(error), /timed out|timeout/i);
196
+ return true;
197
+ },
198
+ );
199
+ });
200
+
201
+ it("propagates caller cancellation into the injected transport", async () => {
202
+ const caller = new AbortController();
203
+ let transportAborted = false;
204
+ const provider = createOpenAICompatible({
205
+ name: "spike",
206
+ baseURL: "https://example.test/v1",
207
+ apiKey: "test-key",
208
+ fetch: async (_input, init) => {
209
+ const transportSignal = init?.signal as AbortSignal;
210
+ return await new Promise<Response>((_resolve, reject) => {
211
+ transportSignal.addEventListener(
212
+ "abort",
213
+ () => {
214
+ transportAborted = true;
215
+ reject(transportSignal.reason);
216
+ },
217
+ { once: true },
218
+ );
219
+ });
220
+ },
221
+ });
222
+ const result = streamText({
223
+ model: provider("test-model"),
224
+ prompt: "question",
225
+ maxRetries: 0,
226
+ abortSignal: caller.signal,
227
+ onError: () => {},
228
+ });
229
+ const partsPromise = (async () => {
230
+ const parts = [];
231
+ for await (const part of result.fullStream) parts.push(part);
232
+ return parts;
233
+ })();
234
+
235
+ await new Promise((resolve) => setTimeout(resolve, 0));
236
+ caller.abort(new Error("caller stopped"));
237
+ const parts = await partsPromise;
238
+
239
+ assert.equal(transportAborted, true);
240
+ assert.ok(parts.some((part) => part.type === "abort"));
241
+ });
242
+ });
package/src/index.ts CHANGED
@@ -40,7 +40,7 @@ export { chatCompletionStream, chatCompletion, OpenAiHttpError, StreamIdleError
40
40
  export type { StreamResponse, EncryptedReasoningItem, ProviderFetch } from "./openaiStream.ts";
41
41
  export { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, requireEnv, reasoningFromEnv, scopeEnvToAlias, dataCaptureFromEnv, contextWindowFromEnv, envelopeFromEnv, resolveReserve } from "./env.ts";
42
42
  export type { Reasoning, ReasoningMode, ReserveSpec } from "./env.ts";
43
- export { normalizeUsage, computeCost } from "./usage.ts";
43
+ export { normalizeUsage, calculateCostUsd } from "./usage.ts";
44
44
  export type { RawUsage, TokenRates } from "./usage.ts";
45
45
  export { ProviderError, classifyProviderError, toProviderError, providerSource } from "./telemetry.ts";
46
46
  export type { TelemetryEvent, ProviderTelemetryKind } from "./telemetry.ts";
@@ -550,7 +550,7 @@ test("bedrock: contextWindow resolves from the catalog via the inference-profile
550
550
  // MECHANISM (non-null, positive), never the literal - a catalog refresh must not break the build.
551
551
  assert.ok(p!.contextWindow !== null && p!.contextWindow > 0, `expected a catalog-resolved window, got ${p!.contextWindow}`);
552
552
  // cost is NOT taken from the native anthropic rate (bedrock marks up) — stays 0
553
- assert.equal(p!.costFor({ prompt: 1_000_000, completion: 1_000_000, reasoning: 0, cached: 0, total: 2_000_000 }), 0);
553
+ assert.equal(p!.calculateCost({ prompt: 1_000_000, completion: 1_000_000, reasoning: 0, cached: 0, total: 2_000_000 }), 0);
554
554
  });
555
555
 
556
556
  test("bedrock: a publisher the catalog lacks (meta) fails hard (cloud, no probe); PLURNK_PROVIDERS_CONTEXT_WINDOW still wins", async () => {
@@ -585,11 +585,11 @@ test("standard provider: a catalog hit fills contextWindow + cost when there's n
585
585
  assert.ok(p !== null);
586
586
  assert.equal(p.contextWindow, info.contextWindow); // catalog window, no probe needed
587
587
  if (info.cost !== undefined) {
588
- // 1M output tokens → outputPer1M USD, in pico-USD (per-1M ×1e6 per token).
589
- const c = p.costFor({ prompt: 0, completion: 1_000_000, reasoning: 0, cached: 0, total: 1_000_000 });
590
- assert.equal(c, Math.round(info.cost.outputPer1M * 1e6 * 1_000_000));
588
+ // One million output tokens cost exactly the catalog's per-million USD rate.
589
+ const c = p.calculateCost({ prompt: 0, completion: 1_000_000, reasoning: 0, cached: 0, total: 1_000_000 });
590
+ assert.equal(c, info.cost.outputPer1M);
591
591
  } else {
592
- assert.equal(p.costFor({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 2 }), 0);
592
+ assert.equal(p.calculateCost({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 2 }), 0);
593
593
  }
594
594
  });
595
595
 
@@ -828,27 +828,17 @@ test("plurnk: a set-but-rejected key (401) surfaces the distinct #537 rejected h
828
828
  assert.equal(seen.filter((s) => s.url.endsWith("/chat/completions")).length, 1); // terminal — never retried, distinct message
829
829
  });
830
830
 
831
- test("plurnk: normalizes the endpoint's balance_pico into meta.balancePico (#23)", async () => {
831
+ test("plurnk: passes currency-explicit balance metadata through (#23)", async () => {
832
832
  mock.method(globalThis, "fetch", async (url: string) => {
833
833
  const u = String(url);
834
834
  if (u.endsWith("/models")) return new Response(JSON.stringify({ data: [{ id: "plurnk", meta: { n_ctx: 49152 } }] }), { status: 200 });
835
835
  // plurnk has detectLlamaServer:false → streams; balance rides as a top-level field on a chunk.
836
- const sse = 'data: {"choices":[{"delta":{"content":"hi"},"finish_reason":"stop"}],"balance_pico":880000000}\n\ndata: [DONE]';
836
+ const sse = 'data: {"choices":[{"delta":{"content":"hi"},"finish_reason":"stop"}],"balance":{"amount":"0.00000088","currency":"XMR"}}\n\ndata: [DONE]';
837
837
  return new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode(sse)); c.close(); } }), { status: 200 });
838
838
  });
839
839
  const p = await standardProviderFromEnv("plurnk", { ...baseEnv, PLURNK_API_KEY: "pk-test" }, "plurnk");
840
840
  const res = await p!.generate({ workerId: "r", messages: [] });
841
- assert.equal(res.meta?.balancePico, 880000000);
842
- mock.restoreAll();
843
- });
844
-
845
- test("a third-party (non-plurnk) provider never NORMALIZES balancePico — only plurnk holds that contract", async () => {
846
- mock.method(globalThis, "fetch", async () =>
847
- new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"hi"}}],"balance_pico":880000000}\n\ndata: [DONE]')); c.close(); } }), { status: 200 }));
848
- const p = await standardProviderFromEnv("groq", { ...baseEnv, GROQ_API_KEY: "k", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
849
- const res = await p!.generate({ workerId: "r", messages: [] });
850
- assert.equal("balancePico" in (res.meta ?? {}), false); // groq has no balanceMetaKey — no normalization
851
- assert.equal(res.meta?.balance_pico, 880000000); // but the raw field still passes through (every-provider meta)
841
+ assert.deepEqual(res.meta?.balance, { amount: "0.00000088", currency: "XMR" });
852
842
  mock.restoreAll();
853
843
  });
854
844
 
@@ -14,7 +14,7 @@ import OpenAICompatProvider, { type ReasoningStyle, type GrammarStyle } from "./
14
14
  import { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, reasoningFromEnv, dataCaptureFromEnv, contextWindowFromEnv, envelopeFromEnv, resolveReserve, type ReserveSpec } from "./env.ts";
15
15
  import { emitWarningOnce } from "./warnings.ts";
16
16
  import { providerSource } from "./telemetry.ts";
17
- import { computeCost } from "./usage.ts";
17
+ import { calculateCostUsd } from "./usage.ts";
18
18
  import { lookup } from "@plurnk/plurnk-models";
19
19
 
20
20
  type StandardProviderSpec = {
@@ -91,9 +91,6 @@ type StandardProviderSpec = {
91
91
  // Default ON for standard providers (OpenAI-standard field, broadly accepted);
92
92
  // set false to opt a backend out (e.g. anthropic's cache_control mechanism).
93
93
  promptCacheKey?: boolean;
94
- // Top-level response field the endpoint reports account balance (pico-USD) in,
95
- // surfaced as ProviderResponse.balancePico (plurnk only, #23). Absent elsewhere.
96
- balanceMetaKey?: string;
97
94
  // RETIRED knob (#27→mimetypes#44): exact client-side tokenizer families were
98
95
  // removed with the tokenizer shed — the var is kept ONLY to fail hard with a
99
96
  // migration pointer when an operator still sets it.
@@ -292,7 +289,7 @@ export const STANDARD_PROVIDERS: Readonly<Record<string, StandardProviderSpec>>
292
289
  apiKeyMessage: "PLURNK_API_KEY not found. Acquire one at https://plurnk.ai . Plurnk also supports local models and alternative cloud provider configurations.",
293
290
  apiKeyRejectedMessage: "PLURNK_API_KEY was rejected by plurnk.ai (invalid or expired). Verify it at https://plurnk.ai .",
294
291
  reasoningStyle: "none", tokenizerEnvVar: "PLURNK_TOKENIZER",
295
- probeNctx: true, detectLlamaServer: false, firstPartyMetadata: true, balanceMetaKey: "balance_pico", suppressTuningFloors: true,
292
+ probeNctx: true, detectLlamaServer: false, firstPartyMetadata: true, suppressTuningFloors: true,
296
293
  },
297
294
  });
298
295
 
@@ -526,7 +523,7 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
526
523
  // window for a known cloud model (groq/deepseek/mistral/…, which don't
527
524
  // probe). A local llama-server model misses the catalog and keeps its
528
525
  // probed n_ctx. Standard providers carry NO live pricing, so the catalog is
529
- // the sole — never shadowing — cost source; per-1M USD → pico-USD/token (×1e6).
526
+ // the sole — never shadowing — cost source, expressed in USD per 1M tokens.
530
527
  // A relay with a catalogContextLookup (bedrock) resolves its window via the
531
528
  // underlying model's publisher and carries NO catalog cost (native rate ≠ relay
532
529
  // rate, #22); everyone else keys the catalog directly on (name, wireModel).
@@ -551,12 +548,12 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
551
548
  );
552
549
  }
553
550
  const cost = fallback?.cost;
554
- const costFor = cost === undefined
551
+ const calculateCost = cost === undefined
555
552
  ? undefined
556
- : (usage: ProviderUsage): number => computeCost(usage, {
557
- input: cost.inputPer1M * 1e6,
558
- output: cost.outputPer1M * 1e6,
559
- cached: (cost.cacheReadPer1M ?? cost.inputPer1M) * 1e6,
553
+ : (usage: ProviderUsage): number => calculateCostUsd(usage, {
554
+ input: cost.inputPer1M,
555
+ output: cost.outputPer1M,
556
+ cached: cost.cacheReadPer1M ?? cost.inputPer1M,
560
557
  });
561
558
 
562
559
  // #507: completion cap — when the catalog reports a maxOutput, use
@@ -610,7 +607,7 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
610
607
  retryDelayMs: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_DELAY, "PLURNK_PROVIDERS_RETRY_DELAY", name),
611
608
  retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", name),
612
609
  reasoningStyle,
613
- costFor,
610
+ calculateCost,
614
611
  source: providerSource(name),
615
612
  grammarStyle,
616
613
  // Optional debug toggle (off by default): validate a transported grammar
@@ -624,7 +621,6 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
624
621
  apiKeyRejectedMessage: spec.apiKeyRejectedMessage,
625
622
  promptCacheKey: spec.promptCacheKey ?? true, // #518: default-on for standard providers (OpenAI-standard field, 6/6 backends verified accept it); per-spec opt-out below
626
623
  serviceTier,
627
- balanceMetaKey: spec.balanceMetaKey,
628
624
  supportsSlotPinning,
629
625
  slotCount,
630
626
  eosText,
package/src/types.ts CHANGED
@@ -69,9 +69,9 @@ export interface ProviderResponse {
69
69
  readonly assistant: ProviderAssistant;
70
70
  readonly assistantRaw: unknown;
71
71
  // Per-turn provider→client metadata bag: the backend's non-standard top-level
72
- // response fields, passed through verbatim, PLUS validated known keys we hold a
73
- // contract for (e.g. `balancePico` — a finite pico-USD number, from the plurnk
74
- // endpoint). The consumer (service) merges this into its Turn metadata and
72
+ // response fields passed through verbatim. Monetary values carry their own
73
+ // amount and currency; the provider does not reinterpret them. The consumer
74
+ // (service) merges this into its Turn metadata and
75
75
  // filters what reaches the client; it reads `meta`, never mines `assistantRaw`.
76
76
  // Absent when the backend reported no extra fields (#23, generalized).
77
77
  readonly meta?: Record<string, unknown>;
@@ -191,9 +191,9 @@ export interface Provider {
191
191
  // `tokenize === undefined` means the backend can't. Exact-counting
192
192
  // consumers (the tokenizer seam) prefer this over any client-side data.
193
193
  tokenize?(text: string): Promise<number[]>;
194
- // Provider-owned cost calculation. Returns pico-USD (1e-12 USD).
194
+ // Provider-owned estimated cost calculation. Returns USD.
195
195
  // Returns 0 for siblings/models with no known rates.
196
- costFor(usage: ProviderUsage): number;
196
+ calculateCost(usage: ProviderUsage): number;
197
197
  }
198
198
 
199
199
  // ProviderAlias moved to @plurnk/plurnk-aliases (the zero-dep parser, #27);
package/src/usage.test.ts CHANGED
@@ -1,6 +1,6 @@
1
1
  import test from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
- import { normalizeUsage, computeCost } from "./usage.ts";
3
+ import { normalizeUsage, calculateCostUsd } from "./usage.ts";
4
4
 
5
5
  // — normalizeUsage —
6
6
 
@@ -115,22 +115,20 @@ test("normalizeUsage: no total reported -> re-split skipped, reasoning stays 0 (
115
115
  assert.deepEqual(u, { prompt: 10, completion: 20, reasoning: 0, cached: 0, total: 30 });
116
116
  });
117
117
 
118
- // — computeCost —
118
+ // — calculateCostUsd —
119
119
 
120
- test("computeCost: bills reasoning at the output rate", () => {
120
+ test("calculateCostUsd: bills reasoning at the USD-per-million output rate", () => {
121
121
  // 100 input, 0 cached, 50 completion + 200 reasoning = 250 output.
122
122
  const usage = { prompt: 100, completion: 50, reasoning: 200, cached: 0, total: 350 };
123
- // input 1 pico/tok, output 10 pico/tok → 100*1 + 250*10 = 2600
124
- assert.equal(computeCost(usage, { input: 1, output: 10, cached: 0 }), 2600);
123
+ assert.equal(calculateCostUsd(usage, { input: 1, output: 10, cached: 0 }), 0.0026);
125
124
  });
126
125
 
127
- test("computeCost: cached prompt billed at the cache rate, remainder at input", () => {
126
+ test("calculateCostUsd: cached prompt billed at the cache rate, remainder at input", () => {
128
127
  const usage = { prompt: 1000, completion: 0, reasoning: 0, cached: 400, total: 1000 };
129
- // 600 non-cached @5 + 400 cached @1 = 3000 + 400 = 3400
130
- assert.equal(computeCost(usage, { input: 5, output: 99, cached: 1 }), 3400);
128
+ assert.equal(calculateCostUsd(usage, { input: 5, output: 99, cached: 1 }), 0.0034);
131
129
  });
132
130
 
133
- test("computeCost: zero rates → 0", () => {
131
+ test("calculateCostUsd: zero rates → 0", () => {
134
132
  const usage = { prompt: 9, completion: 9, reasoning: 9, cached: 9, total: 27 };
135
- assert.equal(computeCost(usage, { input: 0, output: 0, cached: 0 }), 0);
133
+ assert.equal(calculateCostUsd(usage, { input: 0, output: 0, cached: 0 }), 0);
136
134
  });
package/src/usage.ts CHANGED
@@ -69,14 +69,18 @@ export const normalizeUsage = (raw: RawUsage | null | undefined, reasoningText =
69
69
  return { prompt, completion, reasoning, cached, total };
70
70
  };
71
71
 
72
- // Per-token rates in pico-USD (1e-12 USD).
72
+ // Conventional provider pricing: USD per million tokens, matching Models.dev.
73
73
  export type TokenRates = { input: number; output: number; cached: number };
74
74
 
75
75
  // The one cost formula every provider uses: non-cached prompt at the input
76
76
  // rate, cached prompt at the cache rate, and billable output (completion +
77
77
  // reasoning) at the output rate.
78
- export const computeCost = (usage: ProviderUsage, rates: TokenRates): number => {
78
+ export const calculateCostUsd = (usage: ProviderUsage, rates: TokenRates): number => {
79
79
  const nonCachedPrompt = Math.max(0, usage.prompt - usage.cached);
80
80
  const output = usage.completion + usage.reasoning;
81
- return Math.round(nonCachedPrompt * rates.input + usage.cached * rates.cached + output * rates.output);
81
+ return (
82
+ nonCachedPrompt * rates.input
83
+ + usage.cached * rates.cached
84
+ + output * rates.output
85
+ ) / 1_000_000;
82
86
  };