@plurnk/plurnk-providers 1.3.4 → 1.3.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +6 -7
- package/SPEC.md +29 -12
- package/dist/Mock.d.ts +1 -1
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +1 -1
- package/dist/Mock.js.map +1 -1
- package/dist/OpenAICompat.d.ts +2 -3
- package/dist/OpenAICompat.d.ts.map +1 -1
- package/dist/OpenAICompat.js +5 -16
- package/dist/OpenAICompat.js.map +1 -1
- package/dist/Pool.d.ts +1 -1
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +1 -1
- package/dist/Pool.js.map +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -1
- package/dist/index.js.map +1 -1
- package/dist/standardProviders.d.ts +0 -1
- package/dist/standardProviders.d.ts.map +1 -1
- package/dist/standardProviders.js +9 -10
- package/dist/standardProviders.js.map +1 -1
- package/dist/types.d.ts +1 -1
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +1 -1
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +4 -2
- package/dist/usage.js.map +1 -1
- package/package.json +10 -7
- package/src/Mock.test.ts +2 -2
- package/src/Mock.ts +1 -1
- package/src/OpenAICompat.test.ts +10 -26
- package/src/OpenAICompat.ts +6 -17
- package/src/Pool.test.ts +3 -3
- package/src/Pool.ts +1 -1
- package/src/ProviderRegistry.test.ts +1 -1
- package/src/aiSdkAdapter.spike.test.ts +242 -0
- package/src/index.ts +1 -1
- package/src/standardProviders.test.ts +8 -18
- package/src/standardProviders.ts +9 -13
- package/src/types.ts +5 -5
- package/src/usage.test.ts +8 -10
- package/src/usage.ts +7 -3
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
import assert from "node:assert/strict";
|
|
2
|
+
import { describe, it } from "node:test";
|
|
3
|
+
import { createOpenAICompatible } from "@ai-sdk/openai-compatible";
|
|
4
|
+
import { APICallError, streamText } from "ai";
|
|
5
|
+
|
|
6
|
+
const encoder = new TextEncoder();
|
|
7
|
+
|
|
8
|
+
const sseResponse = (...chunks: object[]): Response => new Response(
|
|
9
|
+
new ReadableStream({
|
|
10
|
+
start(controller) {
|
|
11
|
+
for (const chunk of chunks) {
|
|
12
|
+
controller.enqueue(encoder.encode(`data: ${JSON.stringify(chunk)}\n\n`));
|
|
13
|
+
}
|
|
14
|
+
controller.enqueue(encoder.encode("data: [DONE]\n\n"));
|
|
15
|
+
controller.close();
|
|
16
|
+
},
|
|
17
|
+
}),
|
|
18
|
+
{ headers: { "content-type": "text/event-stream" } },
|
|
19
|
+
);
|
|
20
|
+
|
|
21
|
+
describe("AI SDK adapter spike", () => {
|
|
22
|
+
it("preserves PLURNK request extensions and complete stream evidence", async () => {
|
|
23
|
+
let requestBody: Record<string, unknown> | undefined;
|
|
24
|
+
const rawChunks = [
|
|
25
|
+
{
|
|
26
|
+
id: "response-1",
|
|
27
|
+
object: "chat.completion.chunk",
|
|
28
|
+
created: 1,
|
|
29
|
+
model: "test-model",
|
|
30
|
+
choices: [{
|
|
31
|
+
index: 0,
|
|
32
|
+
delta: { role: "assistant", reasoning_content: "because " },
|
|
33
|
+
finish_reason: null,
|
|
34
|
+
}],
|
|
35
|
+
},
|
|
36
|
+
{
|
|
37
|
+
id: "response-1",
|
|
38
|
+
object: "chat.completion.chunk",
|
|
39
|
+
created: 1,
|
|
40
|
+
model: "test-model",
|
|
41
|
+
choices: [{
|
|
42
|
+
index: 0,
|
|
43
|
+
delta: { content: "answer" },
|
|
44
|
+
finish_reason: null,
|
|
45
|
+
}],
|
|
46
|
+
},
|
|
47
|
+
{
|
|
48
|
+
id: "response-1",
|
|
49
|
+
object: "chat.completion.chunk",
|
|
50
|
+
created: 1,
|
|
51
|
+
model: "test-model",
|
|
52
|
+
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
|
|
53
|
+
usage: {
|
|
54
|
+
prompt_tokens: 3,
|
|
55
|
+
completion_tokens: 5,
|
|
56
|
+
total_tokens: 8,
|
|
57
|
+
completion_tokens_details: { reasoning_tokens: 2 },
|
|
58
|
+
},
|
|
59
|
+
},
|
|
60
|
+
];
|
|
61
|
+
const provider = createOpenAICompatible({
|
|
62
|
+
name: "spike",
|
|
63
|
+
baseURL: "https://example.test/v1",
|
|
64
|
+
apiKey: "test-key",
|
|
65
|
+
includeUsage: true,
|
|
66
|
+
fetch: async (_input, init) => {
|
|
67
|
+
requestBody = JSON.parse(String(init?.body)) as Record<string, unknown>;
|
|
68
|
+
return sseResponse(...rawChunks);
|
|
69
|
+
},
|
|
70
|
+
});
|
|
71
|
+
|
|
72
|
+
const result = streamText({
|
|
73
|
+
model: provider("test-model"),
|
|
74
|
+
messages: [{ role: "user", content: "question" }],
|
|
75
|
+
maxOutputTokens: 64,
|
|
76
|
+
temperature: 0.25,
|
|
77
|
+
maxRetries: 0,
|
|
78
|
+
timeout: { totalMs: 1_000, chunkMs: 500 },
|
|
79
|
+
includeRawChunks: true,
|
|
80
|
+
providerOptions: {
|
|
81
|
+
spike: {
|
|
82
|
+
grammar: "root ::= \"answer\"",
|
|
83
|
+
id_slot: 2,
|
|
84
|
+
enable_thinking: true,
|
|
85
|
+
service_tier: "flex",
|
|
86
|
+
},
|
|
87
|
+
},
|
|
88
|
+
});
|
|
89
|
+
const parts = [];
|
|
90
|
+
for await (const part of result.fullStream) parts.push(part);
|
|
91
|
+
|
|
92
|
+
assert.equal(requestBody?.model, "test-model");
|
|
93
|
+
assert.equal(requestBody?.max_tokens, 64);
|
|
94
|
+
assert.equal(requestBody?.temperature, 0.25);
|
|
95
|
+
assert.equal(requestBody?.grammar, "root ::= \"answer\"");
|
|
96
|
+
assert.equal(requestBody?.id_slot, 2);
|
|
97
|
+
assert.equal(requestBody?.enable_thinking, true);
|
|
98
|
+
assert.equal(requestBody?.service_tier, "flex");
|
|
99
|
+
assert.equal(requestBody?.stream, true);
|
|
100
|
+
assert.deepEqual(requestBody?.stream_options, { include_usage: true });
|
|
101
|
+
|
|
102
|
+
assert.equal(await result.text, "answer");
|
|
103
|
+
assert.equal(await result.reasoningText, "because ");
|
|
104
|
+
assert.equal(await result.finishReason, "stop");
|
|
105
|
+
assert.equal(await result.rawFinishReason, "stop");
|
|
106
|
+
const usage = await result.usage;
|
|
107
|
+
assert.equal(usage.inputTokens, 3);
|
|
108
|
+
assert.equal(usage.outputTokens, 5);
|
|
109
|
+
assert.equal(usage.outputTokenDetails.reasoningTokens, 2);
|
|
110
|
+
assert.equal(usage.outputTokenDetails.textTokens, 3);
|
|
111
|
+
assert.equal(usage.totalTokens, 8);
|
|
112
|
+
assert.deepEqual(
|
|
113
|
+
parts.filter((part) => part.type === "raw").map((part) => part.rawValue),
|
|
114
|
+
rawChunks,
|
|
115
|
+
);
|
|
116
|
+
});
|
|
117
|
+
|
|
118
|
+
it("surfaces typed HTTP failures without retrying", async () => {
|
|
119
|
+
let requests = 0;
|
|
120
|
+
const provider = createOpenAICompatible({
|
|
121
|
+
name: "spike",
|
|
122
|
+
baseURL: "https://example.test/v1",
|
|
123
|
+
apiKey: "test-key",
|
|
124
|
+
fetch: async () => {
|
|
125
|
+
requests += 1;
|
|
126
|
+
return new Response(
|
|
127
|
+
JSON.stringify({ error: { message: "rate limited", type: "rate_limit" } }),
|
|
128
|
+
{
|
|
129
|
+
status: 429,
|
|
130
|
+
headers: {
|
|
131
|
+
"content-type": "application/json",
|
|
132
|
+
"retry-after": "3",
|
|
133
|
+
"x-request-id": "request-1",
|
|
134
|
+
},
|
|
135
|
+
},
|
|
136
|
+
);
|
|
137
|
+
},
|
|
138
|
+
});
|
|
139
|
+
const result = streamText({
|
|
140
|
+
model: provider("test-model"),
|
|
141
|
+
prompt: "question",
|
|
142
|
+
maxRetries: 0,
|
|
143
|
+
onError: () => {},
|
|
144
|
+
});
|
|
145
|
+
|
|
146
|
+
const parts = [];
|
|
147
|
+
for await (const part of result.fullStream) parts.push(part);
|
|
148
|
+
const errorPart = parts.find((part) => part.type === "error");
|
|
149
|
+
assert.ok(errorPart?.type === "error");
|
|
150
|
+
assert.ok(APICallError.isInstance(errorPart.error));
|
|
151
|
+
assert.equal(errorPart.error.statusCode, 429);
|
|
152
|
+
assert.equal(errorPart.error.isRetryable, true);
|
|
153
|
+
assert.equal(errorPart.error.responseHeaders?.["retry-after"], "3");
|
|
154
|
+
assert.equal(errorPart.error.responseHeaders?.["x-request-id"], "request-1");
|
|
155
|
+
assert.match(errorPart.error.responseBody ?? "", /rate limited/);
|
|
156
|
+
assert.equal(requests, 1);
|
|
157
|
+
});
|
|
158
|
+
|
|
159
|
+
it("enforces stream-idle timeout through the transport contract", async () => {
|
|
160
|
+
const provider = createOpenAICompatible({
|
|
161
|
+
name: "spike",
|
|
162
|
+
baseURL: "https://example.test/v1",
|
|
163
|
+
apiKey: "test-key",
|
|
164
|
+
fetch: async () => new Response(
|
|
165
|
+
new ReadableStream({
|
|
166
|
+
start(controller) {
|
|
167
|
+
controller.enqueue(encoder.encode(`data: ${JSON.stringify({
|
|
168
|
+
id: "response-1",
|
|
169
|
+
object: "chat.completion.chunk",
|
|
170
|
+
created: 1,
|
|
171
|
+
model: "test-model",
|
|
172
|
+
choices: [{
|
|
173
|
+
index: 0,
|
|
174
|
+
delta: { role: "assistant", content: "started" },
|
|
175
|
+
finish_reason: null,
|
|
176
|
+
}],
|
|
177
|
+
})}\n\n`));
|
|
178
|
+
setTimeout(() => controller.close(), 100);
|
|
179
|
+
},
|
|
180
|
+
}),
|
|
181
|
+
{ headers: { "content-type": "text/event-stream" } },
|
|
182
|
+
),
|
|
183
|
+
});
|
|
184
|
+
const result = streamText({
|
|
185
|
+
model: provider("test-model"),
|
|
186
|
+
prompt: "question",
|
|
187
|
+
maxRetries: 0,
|
|
188
|
+
timeout: { totalMs: 500, chunkMs: 25 },
|
|
189
|
+
onError: () => {},
|
|
190
|
+
});
|
|
191
|
+
|
|
192
|
+
await assert.rejects(
|
|
193
|
+
() => Promise.resolve(result.text),
|
|
194
|
+
(error: unknown) => {
|
|
195
|
+
assert.match(String(error), /timed out|timeout/i);
|
|
196
|
+
return true;
|
|
197
|
+
},
|
|
198
|
+
);
|
|
199
|
+
});
|
|
200
|
+
|
|
201
|
+
it("propagates caller cancellation into the injected transport", async () => {
|
|
202
|
+
const caller = new AbortController();
|
|
203
|
+
let transportAborted = false;
|
|
204
|
+
const provider = createOpenAICompatible({
|
|
205
|
+
name: "spike",
|
|
206
|
+
baseURL: "https://example.test/v1",
|
|
207
|
+
apiKey: "test-key",
|
|
208
|
+
fetch: async (_input, init) => {
|
|
209
|
+
const transportSignal = init?.signal as AbortSignal;
|
|
210
|
+
return await new Promise<Response>((_resolve, reject) => {
|
|
211
|
+
transportSignal.addEventListener(
|
|
212
|
+
"abort",
|
|
213
|
+
() => {
|
|
214
|
+
transportAborted = true;
|
|
215
|
+
reject(transportSignal.reason);
|
|
216
|
+
},
|
|
217
|
+
{ once: true },
|
|
218
|
+
);
|
|
219
|
+
});
|
|
220
|
+
},
|
|
221
|
+
});
|
|
222
|
+
const result = streamText({
|
|
223
|
+
model: provider("test-model"),
|
|
224
|
+
prompt: "question",
|
|
225
|
+
maxRetries: 0,
|
|
226
|
+
abortSignal: caller.signal,
|
|
227
|
+
onError: () => {},
|
|
228
|
+
});
|
|
229
|
+
const partsPromise = (async () => {
|
|
230
|
+
const parts = [];
|
|
231
|
+
for await (const part of result.fullStream) parts.push(part);
|
|
232
|
+
return parts;
|
|
233
|
+
})();
|
|
234
|
+
|
|
235
|
+
await new Promise((resolve) => setTimeout(resolve, 0));
|
|
236
|
+
caller.abort(new Error("caller stopped"));
|
|
237
|
+
const parts = await partsPromise;
|
|
238
|
+
|
|
239
|
+
assert.equal(transportAborted, true);
|
|
240
|
+
assert.ok(parts.some((part) => part.type === "abort"));
|
|
241
|
+
});
|
|
242
|
+
});
|
package/src/index.ts
CHANGED
|
@@ -40,7 +40,7 @@ export { chatCompletionStream, chatCompletion, OpenAiHttpError, StreamIdleError
|
|
|
40
40
|
export type { StreamResponse, EncryptedReasoningItem, ProviderFetch } from "./openaiStream.ts";
|
|
41
41
|
export { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, requireEnv, reasoningFromEnv, scopeEnvToAlias, dataCaptureFromEnv, contextWindowFromEnv, envelopeFromEnv, resolveReserve } from "./env.ts";
|
|
42
42
|
export type { Reasoning, ReasoningMode, ReserveSpec } from "./env.ts";
|
|
43
|
-
export { normalizeUsage,
|
|
43
|
+
export { normalizeUsage, calculateCostUsd } from "./usage.ts";
|
|
44
44
|
export type { RawUsage, TokenRates } from "./usage.ts";
|
|
45
45
|
export { ProviderError, classifyProviderError, toProviderError, providerSource } from "./telemetry.ts";
|
|
46
46
|
export type { TelemetryEvent, ProviderTelemetryKind } from "./telemetry.ts";
|
|
@@ -550,7 +550,7 @@ test("bedrock: contextWindow resolves from the catalog via the inference-profile
|
|
|
550
550
|
// MECHANISM (non-null, positive), never the literal - a catalog refresh must not break the build.
|
|
551
551
|
assert.ok(p!.contextWindow !== null && p!.contextWindow > 0, `expected a catalog-resolved window, got ${p!.contextWindow}`);
|
|
552
552
|
// cost is NOT taken from the native anthropic rate (bedrock marks up) — stays 0
|
|
553
|
-
assert.equal(p!.
|
|
553
|
+
assert.equal(p!.calculateCost({ prompt: 1_000_000, completion: 1_000_000, reasoning: 0, cached: 0, total: 2_000_000 }), 0);
|
|
554
554
|
});
|
|
555
555
|
|
|
556
556
|
test("bedrock: a publisher the catalog lacks (meta) fails hard (cloud, no probe); PLURNK_PROVIDERS_CONTEXT_WINDOW still wins", async () => {
|
|
@@ -585,11 +585,11 @@ test("standard provider: a catalog hit fills contextWindow + cost when there's n
|
|
|
585
585
|
assert.ok(p !== null);
|
|
586
586
|
assert.equal(p.contextWindow, info.contextWindow); // catalog window, no probe needed
|
|
587
587
|
if (info.cost !== undefined) {
|
|
588
|
-
//
|
|
589
|
-
const c = p.
|
|
590
|
-
assert.equal(c,
|
|
588
|
+
// One million output tokens cost exactly the catalog's per-million USD rate.
|
|
589
|
+
const c = p.calculateCost({ prompt: 0, completion: 1_000_000, reasoning: 0, cached: 0, total: 1_000_000 });
|
|
590
|
+
assert.equal(c, info.cost.outputPer1M);
|
|
591
591
|
} else {
|
|
592
|
-
assert.equal(p.
|
|
592
|
+
assert.equal(p.calculateCost({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 2 }), 0);
|
|
593
593
|
}
|
|
594
594
|
});
|
|
595
595
|
|
|
@@ -828,27 +828,17 @@ test("plurnk: a set-but-rejected key (401) surfaces the distinct #537 rejected h
|
|
|
828
828
|
assert.equal(seen.filter((s) => s.url.endsWith("/chat/completions")).length, 1); // terminal — never retried, distinct message
|
|
829
829
|
});
|
|
830
830
|
|
|
831
|
-
test("plurnk:
|
|
831
|
+
test("plurnk: passes currency-explicit balance metadata through (#23)", async () => {
|
|
832
832
|
mock.method(globalThis, "fetch", async (url: string) => {
|
|
833
833
|
const u = String(url);
|
|
834
834
|
if (u.endsWith("/models")) return new Response(JSON.stringify({ data: [{ id: "plurnk", meta: { n_ctx: 49152 } }] }), { status: 200 });
|
|
835
835
|
// plurnk has detectLlamaServer:false → streams; balance rides as a top-level field on a chunk.
|
|
836
|
-
const sse = 'data: {"choices":[{"delta":{"content":"hi"},"finish_reason":"stop"}],"
|
|
836
|
+
const sse = 'data: {"choices":[{"delta":{"content":"hi"},"finish_reason":"stop"}],"balance":{"amount":"0.00000088","currency":"XMR"}}\n\ndata: [DONE]';
|
|
837
837
|
return new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode(sse)); c.close(); } }), { status: 200 });
|
|
838
838
|
});
|
|
839
839
|
const p = await standardProviderFromEnv("plurnk", { ...baseEnv, PLURNK_API_KEY: "pk-test" }, "plurnk");
|
|
840
840
|
const res = await p!.generate({ workerId: "r", messages: [] });
|
|
841
|
-
assert.
|
|
842
|
-
mock.restoreAll();
|
|
843
|
-
});
|
|
844
|
-
|
|
845
|
-
test("a third-party (non-plurnk) provider never NORMALIZES balancePico — only plurnk holds that contract", async () => {
|
|
846
|
-
mock.method(globalThis, "fetch", async () =>
|
|
847
|
-
new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"hi"}}],"balance_pico":880000000}\n\ndata: [DONE]')); c.close(); } }), { status: 200 }));
|
|
848
|
-
const p = await standardProviderFromEnv("groq", { ...baseEnv, GROQ_API_KEY: "k", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
|
|
849
|
-
const res = await p!.generate({ workerId: "r", messages: [] });
|
|
850
|
-
assert.equal("balancePico" in (res.meta ?? {}), false); // groq has no balanceMetaKey — no normalization
|
|
851
|
-
assert.equal(res.meta?.balance_pico, 880000000); // but the raw field still passes through (every-provider meta)
|
|
841
|
+
assert.deepEqual(res.meta?.balance, { amount: "0.00000088", currency: "XMR" });
|
|
852
842
|
mock.restoreAll();
|
|
853
843
|
});
|
|
854
844
|
|
package/src/standardProviders.ts
CHANGED
|
@@ -14,7 +14,7 @@ import OpenAICompatProvider, { type ReasoningStyle, type GrammarStyle } from "./
|
|
|
14
14
|
import { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, reasoningFromEnv, dataCaptureFromEnv, contextWindowFromEnv, envelopeFromEnv, resolveReserve, type ReserveSpec } from "./env.ts";
|
|
15
15
|
import { emitWarningOnce } from "./warnings.ts";
|
|
16
16
|
import { providerSource } from "./telemetry.ts";
|
|
17
|
-
import {
|
|
17
|
+
import { calculateCostUsd } from "./usage.ts";
|
|
18
18
|
import { lookup } from "@plurnk/plurnk-models";
|
|
19
19
|
|
|
20
20
|
type StandardProviderSpec = {
|
|
@@ -91,9 +91,6 @@ type StandardProviderSpec = {
|
|
|
91
91
|
// Default ON for standard providers (OpenAI-standard field, broadly accepted);
|
|
92
92
|
// set false to opt a backend out (e.g. anthropic's cache_control mechanism).
|
|
93
93
|
promptCacheKey?: boolean;
|
|
94
|
-
// Top-level response field the endpoint reports account balance (pico-USD) in,
|
|
95
|
-
// surfaced as ProviderResponse.balancePico (plurnk only, #23). Absent elsewhere.
|
|
96
|
-
balanceMetaKey?: string;
|
|
97
94
|
// RETIRED knob (#27→mimetypes#44): exact client-side tokenizer families were
|
|
98
95
|
// removed with the tokenizer shed — the var is kept ONLY to fail hard with a
|
|
99
96
|
// migration pointer when an operator still sets it.
|
|
@@ -292,7 +289,7 @@ export const STANDARD_PROVIDERS: Readonly<Record<string, StandardProviderSpec>>
|
|
|
292
289
|
apiKeyMessage: "PLURNK_API_KEY not found. Acquire one at https://plurnk.ai . Plurnk also supports local models and alternative cloud provider configurations.",
|
|
293
290
|
apiKeyRejectedMessage: "PLURNK_API_KEY was rejected by plurnk.ai (invalid or expired). Verify it at https://plurnk.ai .",
|
|
294
291
|
reasoningStyle: "none", tokenizerEnvVar: "PLURNK_TOKENIZER",
|
|
295
|
-
probeNctx: true, detectLlamaServer: false, firstPartyMetadata: true,
|
|
292
|
+
probeNctx: true, detectLlamaServer: false, firstPartyMetadata: true, suppressTuningFloors: true,
|
|
296
293
|
},
|
|
297
294
|
});
|
|
298
295
|
|
|
@@ -526,7 +523,7 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
|
|
|
526
523
|
// window for a known cloud model (groq/deepseek/mistral/…, which don't
|
|
527
524
|
// probe). A local llama-server model misses the catalog and keeps its
|
|
528
525
|
// probed n_ctx. Standard providers carry NO live pricing, so the catalog is
|
|
529
|
-
// the sole — never shadowing — cost source
|
|
526
|
+
// the sole — never shadowing — cost source, expressed in USD per 1M tokens.
|
|
530
527
|
// A relay with a catalogContextLookup (bedrock) resolves its window via the
|
|
531
528
|
// underlying model's publisher and carries NO catalog cost (native rate ≠ relay
|
|
532
529
|
// rate, #22); everyone else keys the catalog directly on (name, wireModel).
|
|
@@ -551,12 +548,12 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
|
|
|
551
548
|
);
|
|
552
549
|
}
|
|
553
550
|
const cost = fallback?.cost;
|
|
554
|
-
const
|
|
551
|
+
const calculateCost = cost === undefined
|
|
555
552
|
? undefined
|
|
556
|
-
: (usage: ProviderUsage): number =>
|
|
557
|
-
input: cost.inputPer1M
|
|
558
|
-
output: cost.outputPer1M
|
|
559
|
-
cached:
|
|
553
|
+
: (usage: ProviderUsage): number => calculateCostUsd(usage, {
|
|
554
|
+
input: cost.inputPer1M,
|
|
555
|
+
output: cost.outputPer1M,
|
|
556
|
+
cached: cost.cacheReadPer1M ?? cost.inputPer1M,
|
|
560
557
|
});
|
|
561
558
|
|
|
562
559
|
// #507: completion cap — when the catalog reports a maxOutput, use
|
|
@@ -610,7 +607,7 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
|
|
|
610
607
|
retryDelayMs: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_DELAY, "PLURNK_PROVIDERS_RETRY_DELAY", name),
|
|
611
608
|
retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", name),
|
|
612
609
|
reasoningStyle,
|
|
613
|
-
|
|
610
|
+
calculateCost,
|
|
614
611
|
source: providerSource(name),
|
|
615
612
|
grammarStyle,
|
|
616
613
|
// Optional debug toggle (off by default): validate a transported grammar
|
|
@@ -624,7 +621,6 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
|
|
|
624
621
|
apiKeyRejectedMessage: spec.apiKeyRejectedMessage,
|
|
625
622
|
promptCacheKey: spec.promptCacheKey ?? true, // #518: default-on for standard providers (OpenAI-standard field, 6/6 backends verified accept it); per-spec opt-out below
|
|
626
623
|
serviceTier,
|
|
627
|
-
balanceMetaKey: spec.balanceMetaKey,
|
|
628
624
|
supportsSlotPinning,
|
|
629
625
|
slotCount,
|
|
630
626
|
eosText,
|
package/src/types.ts
CHANGED
|
@@ -69,9 +69,9 @@ export interface ProviderResponse {
|
|
|
69
69
|
readonly assistant: ProviderAssistant;
|
|
70
70
|
readonly assistantRaw: unknown;
|
|
71
71
|
// Per-turn provider→client metadata bag: the backend's non-standard top-level
|
|
72
|
-
// response fields
|
|
73
|
-
//
|
|
74
|
-
//
|
|
72
|
+
// response fields passed through verbatim. Monetary values carry their own
|
|
73
|
+
// amount and currency; the provider does not reinterpret them. The consumer
|
|
74
|
+
// (service) merges this into its Turn metadata and
|
|
75
75
|
// filters what reaches the client; it reads `meta`, never mines `assistantRaw`.
|
|
76
76
|
// Absent when the backend reported no extra fields (#23, generalized).
|
|
77
77
|
readonly meta?: Record<string, unknown>;
|
|
@@ -191,9 +191,9 @@ export interface Provider {
|
|
|
191
191
|
// `tokenize === undefined` means the backend can't. Exact-counting
|
|
192
192
|
// consumers (the tokenizer seam) prefer this over any client-side data.
|
|
193
193
|
tokenize?(text: string): Promise<number[]>;
|
|
194
|
-
// Provider-owned cost calculation. Returns
|
|
194
|
+
// Provider-owned estimated cost calculation. Returns USD.
|
|
195
195
|
// Returns 0 for siblings/models with no known rates.
|
|
196
|
-
|
|
196
|
+
calculateCost(usage: ProviderUsage): number;
|
|
197
197
|
}
|
|
198
198
|
|
|
199
199
|
// ProviderAlias moved to @plurnk/plurnk-aliases (the zero-dep parser, #27);
|
package/src/usage.test.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import test from "node:test";
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
|
-
import { normalizeUsage,
|
|
3
|
+
import { normalizeUsage, calculateCostUsd } from "./usage.ts";
|
|
4
4
|
|
|
5
5
|
// — normalizeUsage —
|
|
6
6
|
|
|
@@ -115,22 +115,20 @@ test("normalizeUsage: no total reported -> re-split skipped, reasoning stays 0 (
|
|
|
115
115
|
assert.deepEqual(u, { prompt: 10, completion: 20, reasoning: 0, cached: 0, total: 30 });
|
|
116
116
|
});
|
|
117
117
|
|
|
118
|
-
// —
|
|
118
|
+
// — calculateCostUsd —
|
|
119
119
|
|
|
120
|
-
test("
|
|
120
|
+
test("calculateCostUsd: bills reasoning at the USD-per-million output rate", () => {
|
|
121
121
|
// 100 input, 0 cached, 50 completion + 200 reasoning = 250 output.
|
|
122
122
|
const usage = { prompt: 100, completion: 50, reasoning: 200, cached: 0, total: 350 };
|
|
123
|
-
|
|
124
|
-
assert.equal(computeCost(usage, { input: 1, output: 10, cached: 0 }), 2600);
|
|
123
|
+
assert.equal(calculateCostUsd(usage, { input: 1, output: 10, cached: 0 }), 0.0026);
|
|
125
124
|
});
|
|
126
125
|
|
|
127
|
-
test("
|
|
126
|
+
test("calculateCostUsd: cached prompt billed at the cache rate, remainder at input", () => {
|
|
128
127
|
const usage = { prompt: 1000, completion: 0, reasoning: 0, cached: 400, total: 1000 };
|
|
129
|
-
|
|
130
|
-
assert.equal(computeCost(usage, { input: 5, output: 99, cached: 1 }), 3400);
|
|
128
|
+
assert.equal(calculateCostUsd(usage, { input: 5, output: 99, cached: 1 }), 0.0034);
|
|
131
129
|
});
|
|
132
130
|
|
|
133
|
-
test("
|
|
131
|
+
test("calculateCostUsd: zero rates → 0", () => {
|
|
134
132
|
const usage = { prompt: 9, completion: 9, reasoning: 9, cached: 9, total: 27 };
|
|
135
|
-
assert.equal(
|
|
133
|
+
assert.equal(calculateCostUsd(usage, { input: 0, output: 0, cached: 0 }), 0);
|
|
136
134
|
});
|
package/src/usage.ts
CHANGED
|
@@ -69,14 +69,18 @@ export const normalizeUsage = (raw: RawUsage | null | undefined, reasoningText =
|
|
|
69
69
|
return { prompt, completion, reasoning, cached, total };
|
|
70
70
|
};
|
|
71
71
|
|
|
72
|
-
//
|
|
72
|
+
// Conventional provider pricing: USD per million tokens, matching Models.dev.
|
|
73
73
|
export type TokenRates = { input: number; output: number; cached: number };
|
|
74
74
|
|
|
75
75
|
// The one cost formula every provider uses: non-cached prompt at the input
|
|
76
76
|
// rate, cached prompt at the cache rate, and billable output (completion +
|
|
77
77
|
// reasoning) at the output rate.
|
|
78
|
-
export const
|
|
78
|
+
export const calculateCostUsd = (usage: ProviderUsage, rates: TokenRates): number => {
|
|
79
79
|
const nonCachedPrompt = Math.max(0, usage.prompt - usage.cached);
|
|
80
80
|
const output = usage.completion + usage.reasoning;
|
|
81
|
-
return
|
|
81
|
+
return (
|
|
82
|
+
nonCachedPrompt * rates.input
|
|
83
|
+
+ usage.cached * rates.cached
|
|
84
|
+
+ output * rates.output
|
|
85
|
+
) / 1_000_000;
|
|
82
86
|
};
|