@plurnk/plurnk-providers 1.4.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +40 -34
- package/README.md +3 -0
- package/SPEC.md +153 -62
- package/dist/AiSdkProvider.d.ts +19 -25
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +353 -120
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +7 -13
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +36 -8
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +2 -21
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +19 -14
- package/dist/Pool.js.map +1 -1
- package/dist/accounting.d.ts +6 -0
- package/dist/accounting.d.ts.map +1 -0
- package/dist/accounting.js +168 -0
- package/dist/accounting.js.map +1 -0
- package/dist/aiSdkTransport.d.ts +11 -3
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +198 -29
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +7 -2
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +32 -26
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +18 -7
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +10 -10
- package/dist/cost.d.ts.map +1 -1
- package/dist/cost.js +88 -43
- package/dist/cost.js.map +1 -1
- package/dist/env.d.ts +5 -7
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +30 -32
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +14 -2
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +60 -2
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +4 -4
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -2
- package/dist/index.js.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/sdkModels.d.ts +6 -0
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +46 -3
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +40 -29
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +21 -4
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +188 -74
- package/dist/usage.js.map +1 -1
- package/package.json +9 -7
- package/src/AiSdkProvider.test.ts +1039 -182
- package/src/AiSdkProvider.ts +428 -141
- package/src/Mock.test.ts +37 -12
- package/src/Mock.ts +46 -12
- package/src/Pool.test.ts +19 -6
- package/src/Pool.ts +20 -16
- package/src/ProviderRegistry.test.ts +16 -11
- package/src/accounting.test.ts +94 -0
- package/src/accounting.ts +190 -0
- package/src/aiSdkTransport.test.ts +42 -49
- package/src/aiSdkTransport.ts +218 -32
- package/src/boundaries.test.ts +2 -0
- package/src/catalogProvider.test.ts +271 -24
- package/src/catalogProvider.ts +44 -28
- package/src/compatibleProvider.test.ts +6 -3
- package/src/compatibleProvider.ts +20 -7
- package/src/cost.test.ts +55 -35
- package/src/cost.ts +110 -54
- package/src/defaults.test.ts +13 -3
- package/src/env.test.ts +50 -26
- package/src/env.ts +43 -42
- package/src/errors.test.ts +47 -2
- package/src/errors.ts +68 -3
- package/src/index.ts +21 -5
- package/src/ollama.test.ts +4 -1
- package/src/ollama.ts +3 -3
- package/src/sdkModels.test.ts +94 -3
- package/src/sdkModels.ts +53 -3
- package/src/types.ts +91 -33
- package/src/usage.test.ts +112 -108
- package/src/usage.ts +233 -84
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
import test, { mock } from "node:test";
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
|
+
import { once } from "node:events";
|
|
4
|
+
import { createServer } from "node:http";
|
|
3
5
|
import { catalogProviderFromEnv } from "./catalogProvider.ts";
|
|
4
6
|
import { resetEmittedWarnings } from "./warnings.ts";
|
|
5
7
|
|
|
@@ -7,6 +9,8 @@ const env = {
|
|
|
7
9
|
OPENAI_API_KEY: "test-key",
|
|
8
10
|
OPENAI_BASE_URL: "https://api.openai.com/v1",
|
|
9
11
|
PLURNK_PROVIDERS_FETCH_TIMEOUT: "1000",
|
|
12
|
+
PLURNK_PROVIDERS_OPERATION_TIMEOUT: "3000",
|
|
13
|
+
PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT: "1000",
|
|
10
14
|
PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
|
|
11
15
|
PLURNK_PROVIDERS_REASONING: "off",
|
|
12
16
|
PLURNK_PROVIDERS_TEMPERATURE: "0.2",
|
|
@@ -16,7 +20,8 @@ const env = {
|
|
|
16
20
|
PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
|
|
17
21
|
PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
|
|
18
22
|
PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT: "512",
|
|
19
|
-
|
|
23
|
+
PLURNK_PROVIDERS_CACHE_AFFINITY: "1",
|
|
24
|
+
PLURNK_PROVIDERS_CACHE_WRITE_POLICY: "stable-system",
|
|
20
25
|
};
|
|
21
26
|
|
|
22
27
|
test.afterEach(() => {
|
|
@@ -31,13 +36,6 @@ test("catalog provider resolves model physics and Models.dev USD rates", () => {
|
|
|
31
36
|
assert.equal(provider?.contextWindow, 1_047_576);
|
|
32
37
|
assert.equal(provider?.reasoningReserve, 16_384);
|
|
33
38
|
assert.equal(provider?.completionReserve, 32_768);
|
|
34
|
-
assert.ok((provider?.calculateCost({
|
|
35
|
-
prompt: 1_000_000,
|
|
36
|
-
completion: 1_000_000,
|
|
37
|
-
reasoning: 0,
|
|
38
|
-
cached: 0,
|
|
39
|
-
total: 2_000_000,
|
|
40
|
-
}) ?? 0) > 0);
|
|
41
39
|
});
|
|
42
40
|
|
|
43
41
|
test("an operator context window caps catalog physics and percentage reserves derive from the cap", () => {
|
|
@@ -96,7 +94,11 @@ test("official AI SDK provider owns the native request while PLURNK owns call se
|
|
|
96
94
|
});
|
|
97
95
|
|
|
98
96
|
assert.equal(result?.assistant.content, "done");
|
|
99
|
-
assert.equal(result?.
|
|
97
|
+
assert.equal(result?.accounting[0]?.usage?.totalTokens, 3);
|
|
98
|
+
assert.deepEqual(result?.accounting[0]?.cost, {
|
|
99
|
+
kind: "unknown",
|
|
100
|
+
reason: "the provider response omitted a token category with a distinct Models.dev rate",
|
|
101
|
+
});
|
|
100
102
|
assert.equal(calls.length, 1);
|
|
101
103
|
assert.equal(calls[0]?.url, "https://api.openai.com/v1/chat/completions");
|
|
102
104
|
assert.equal(calls[0]?.body.model, "gpt-4.1-mini");
|
|
@@ -104,6 +106,227 @@ test("official AI SDK provider owns the native request while PLURNK owns call se
|
|
|
104
106
|
assert.equal(calls[0]?.body.top_p, 0.8);
|
|
105
107
|
assert.equal(calls[0]?.body.seed, 7);
|
|
106
108
|
assert.equal(calls[0]?.body.max_tokens, 64);
|
|
109
|
+
assert.equal(calls[0]?.body.prompt_cache_key, "worker", "the official OpenAI SDK projects the documented affinity key");
|
|
110
|
+
});
|
|
111
|
+
|
|
112
|
+
test("Cerebras explicit reasoning activation needs no operator effort or token budget", async () => {
|
|
113
|
+
let body: Record<string, unknown> | undefined;
|
|
114
|
+
mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
|
|
115
|
+
body = JSON.parse(String(init?.body)) as Record<string, unknown>;
|
|
116
|
+
return new Response([
|
|
117
|
+
`data: ${JSON.stringify({
|
|
118
|
+
id: "chatcmpl-cerebras",
|
|
119
|
+
object: "chat.completion.chunk",
|
|
120
|
+
created: 1,
|
|
121
|
+
model: "gemma-4-31b",
|
|
122
|
+
choices: [{ index: 0, delta: { reasoning: "consider" }, finish_reason: null }],
|
|
123
|
+
})}`,
|
|
124
|
+
`data: ${JSON.stringify({
|
|
125
|
+
id: "chatcmpl-cerebras",
|
|
126
|
+
object: "chat.completion.chunk",
|
|
127
|
+
created: 2,
|
|
128
|
+
model: "gemma-4-31b",
|
|
129
|
+
choices: [{ index: 0, delta: { content: "done" }, finish_reason: "stop" }],
|
|
130
|
+
})}`,
|
|
131
|
+
`data: ${JSON.stringify({
|
|
132
|
+
id: "chatcmpl-cerebras",
|
|
133
|
+
object: "chat.completion.chunk",
|
|
134
|
+
created: 3,
|
|
135
|
+
model: "gemma-4-31b",
|
|
136
|
+
choices: [],
|
|
137
|
+
usage: {
|
|
138
|
+
prompt_tokens: 2,
|
|
139
|
+
completion_tokens: 2,
|
|
140
|
+
total_tokens: 4,
|
|
141
|
+
completion_tokens_details: { reasoning_tokens: 1 },
|
|
142
|
+
},
|
|
143
|
+
})}`,
|
|
144
|
+
"data: [DONE]",
|
|
145
|
+
].join("\n\n"), { headers: { "content-type": "text/event-stream" } });
|
|
146
|
+
});
|
|
147
|
+
|
|
148
|
+
const provider = catalogProviderFromEnv("cerebras", {
|
|
149
|
+
...env,
|
|
150
|
+
CEREBRAS_API_KEY: "test-key",
|
|
151
|
+
PLURNK_PROVIDERS_REASONING: "on",
|
|
152
|
+
}, "gemma-4-31b");
|
|
153
|
+
const result = await provider?.generate({
|
|
154
|
+
workerId: "worker",
|
|
155
|
+
messages: [{ role: "user", content: "hello" }],
|
|
156
|
+
});
|
|
157
|
+
|
|
158
|
+
assert.equal(body?.reasoning_effort, "medium", "the native SDK projects unqualified on to its enabled posture");
|
|
159
|
+
assert.equal("thinking_budget_tokens" in (body ?? {}), false, "activation does not invent a token budget");
|
|
160
|
+
assert.equal(result?.assistant.reasoning, "consider");
|
|
161
|
+
assert.equal(result?.accounting[0]?.usage?.outputTokenDetails?.reasoningTokens, 1);
|
|
162
|
+
});
|
|
163
|
+
|
|
164
|
+
test("Google adaptive reasoning requests and preserves readable thought summaries", async () => {
|
|
165
|
+
const bodies: Array<{
|
|
166
|
+
generationConfig?: { thinkingConfig?: { includeThoughts?: boolean } };
|
|
167
|
+
}> = [];
|
|
168
|
+
mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
|
|
169
|
+
bodies.push(JSON.parse(String(init?.body)) as typeof bodies[number]);
|
|
170
|
+
return new Response(`data: ${JSON.stringify({
|
|
171
|
+
responseId: "response-gemini",
|
|
172
|
+
candidates: [{
|
|
173
|
+
content: {
|
|
174
|
+
role: "model",
|
|
175
|
+
parts: [
|
|
176
|
+
{ text: "consider", thought: true },
|
|
177
|
+
{ text: "done" },
|
|
178
|
+
],
|
|
179
|
+
},
|
|
180
|
+
finishReason: "STOP",
|
|
181
|
+
}],
|
|
182
|
+
usageMetadata: {
|
|
183
|
+
promptTokenCount: 2,
|
|
184
|
+
candidatesTokenCount: 1,
|
|
185
|
+
thoughtsTokenCount: 1,
|
|
186
|
+
totalTokenCount: 4,
|
|
187
|
+
},
|
|
188
|
+
})}\n\n`, {
|
|
189
|
+
headers: { "content-type": "text/event-stream" },
|
|
190
|
+
});
|
|
191
|
+
});
|
|
192
|
+
|
|
193
|
+
const provider = catalogProviderFromEnv("google", {
|
|
194
|
+
...env,
|
|
195
|
+
GEMINI_API_KEY: "test-key",
|
|
196
|
+
PLURNK_PROVIDERS_REASONING: "adaptive",
|
|
197
|
+
}, "gemini-3.7-flash");
|
|
198
|
+
const result = await provider?.generate({
|
|
199
|
+
workerId: "worker",
|
|
200
|
+
messages: [{ role: "user", content: "hello" }],
|
|
201
|
+
});
|
|
202
|
+
|
|
203
|
+
assert.deepEqual(bodies[0]?.generationConfig?.thinkingConfig, {
|
|
204
|
+
includeThoughts: true,
|
|
205
|
+
}, "adaptive leaves thinking depth to Google while requesting its readable summary");
|
|
206
|
+
assert.equal(result?.assistant.reasoning, "consider");
|
|
207
|
+
assert.equal(result?.assistant.content, "done");
|
|
208
|
+
assert.equal(result?.accounting[0]?.usage?.outputTokenDetails?.reasoningTokens, 1);
|
|
209
|
+
|
|
210
|
+
const disabled = catalogProviderFromEnv("google", {
|
|
211
|
+
...env,
|
|
212
|
+
GEMINI_API_KEY: "test-key",
|
|
213
|
+
PLURNK_PROVIDERS_REASONING: "off",
|
|
214
|
+
}, "gemini-3.7-flash");
|
|
215
|
+
await disabled?.generate({
|
|
216
|
+
workerId: "worker",
|
|
217
|
+
messages: [{ role: "user", content: "hello" }],
|
|
218
|
+
});
|
|
219
|
+
assert.equal(
|
|
220
|
+
bodies[1]?.generationConfig?.thinkingConfig?.includeThoughts,
|
|
221
|
+
undefined,
|
|
222
|
+
"off does not request readable thoughts",
|
|
223
|
+
);
|
|
224
|
+
});
|
|
225
|
+
|
|
226
|
+
test("native provider routes project their documented cache controls through the actual SDK request", async (t) => {
|
|
227
|
+
const calls: Array<{ url: string; headers: Headers; body: Record<string, unknown> }> = [];
|
|
228
|
+
mock.method(globalThis, "fetch", async (input: string | URL | Request, init?: RequestInit) => {
|
|
229
|
+
calls.push({
|
|
230
|
+
url: String(input),
|
|
231
|
+
headers: new Headers(init?.headers),
|
|
232
|
+
body: JSON.parse(String(init?.body)) as Record<string, unknown>,
|
|
233
|
+
});
|
|
234
|
+
return new Response([
|
|
235
|
+
`data: ${JSON.stringify({
|
|
236
|
+
id: "response",
|
|
237
|
+
object: "chat.completion.chunk",
|
|
238
|
+
created: 1,
|
|
239
|
+
model: "served",
|
|
240
|
+
choices: [{ index: 0, delta: { content: "ok" }, finish_reason: "stop" }],
|
|
241
|
+
})}`,
|
|
242
|
+
`data: ${JSON.stringify({
|
|
243
|
+
id: "response",
|
|
244
|
+
object: "chat.completion.chunk",
|
|
245
|
+
created: 2,
|
|
246
|
+
model: "served",
|
|
247
|
+
choices: [],
|
|
248
|
+
usage: { prompt_tokens: 2, completion_tokens: 1, total_tokens: 3 },
|
|
249
|
+
})}`,
|
|
250
|
+
"data: [DONE]",
|
|
251
|
+
].join("\n\n"), { headers: { "content-type": "text/event-stream" } });
|
|
252
|
+
});
|
|
253
|
+
|
|
254
|
+
await t.test("DeepInfra native options become prompt_cache_key", async () => {
|
|
255
|
+
const provider = catalogProviderFromEnv("deepinfra", {
|
|
256
|
+
...env,
|
|
257
|
+
DEEPINFRA_API_KEY: "test-key",
|
|
258
|
+
}, "zai-org/GLM-5.2");
|
|
259
|
+
await provider?.generate({
|
|
260
|
+
workerId: "deepinfra-worker",
|
|
261
|
+
messages: [{ role: "user", content: "hello" }],
|
|
262
|
+
});
|
|
263
|
+
assert.equal(calls.at(-1)?.body.prompt_cache_key, "deepinfra-worker");
|
|
264
|
+
});
|
|
265
|
+
|
|
266
|
+
await t.test("OpenRouter carries session affinity and an Anthropic system breakpoint", async () => {
|
|
267
|
+
let call: { headers: Headers; body: Record<string, unknown> } | undefined;
|
|
268
|
+
const server = createServer(async (request, response) => {
|
|
269
|
+
const chunks: Buffer[] = [];
|
|
270
|
+
for await (const chunk of request) chunks.push(Buffer.from(chunk));
|
|
271
|
+
call = {
|
|
272
|
+
headers: new Headers(request.headers as Record<string, string>),
|
|
273
|
+
body: JSON.parse(Buffer.concat(chunks).toString("utf8")) as Record<string, unknown>,
|
|
274
|
+
};
|
|
275
|
+
response.writeHead(200, { "content-type": "text/event-stream" });
|
|
276
|
+
response.end([
|
|
277
|
+
`data: ${JSON.stringify({
|
|
278
|
+
id: "response",
|
|
279
|
+
object: "chat.completion.chunk",
|
|
280
|
+
created: 1,
|
|
281
|
+
model: "served",
|
|
282
|
+
choices: [{ index: 0, delta: { content: "ok" }, finish_reason: "stop" }],
|
|
283
|
+
})}`,
|
|
284
|
+
`data: ${JSON.stringify({
|
|
285
|
+
id: "response",
|
|
286
|
+
object: "chat.completion.chunk",
|
|
287
|
+
created: 2,
|
|
288
|
+
model: "served",
|
|
289
|
+
choices: [],
|
|
290
|
+
usage: { prompt_tokens: 2, completion_tokens: 1, total_tokens: 3 },
|
|
291
|
+
})}`,
|
|
292
|
+
"data: [DONE]",
|
|
293
|
+
].join("\n\n"));
|
|
294
|
+
});
|
|
295
|
+
server.listen(0, "127.0.0.1");
|
|
296
|
+
await once(server, "listening");
|
|
297
|
+
t.after(() => new Promise<void>((resolve, reject) => {
|
|
298
|
+
server.close((error) => error === undefined ? resolve() : reject(error));
|
|
299
|
+
}));
|
|
300
|
+
const address = server.address();
|
|
301
|
+
if (address === null || typeof address === "string") {
|
|
302
|
+
throw new Error("OpenRouter request-capture server did not bind a TCP address");
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
const provider = catalogProviderFromEnv("openrouter", {
|
|
306
|
+
...env,
|
|
307
|
+
OPENROUTER_API_KEY: "test-key",
|
|
308
|
+
}, "anthropic/claude-sonnet-4.6", `http://127.0.0.1:${address.port}/api/v1`);
|
|
309
|
+
await provider?.generate({
|
|
310
|
+
workerId: "openrouter-worker",
|
|
311
|
+
messages: [
|
|
312
|
+
{ role: "system", content: "stable system packet" },
|
|
313
|
+
{ role: "user", content: "changing user packet" },
|
|
314
|
+
],
|
|
315
|
+
});
|
|
316
|
+
assert.equal(call?.headers.get("x-session-id"), "openrouter-worker");
|
|
317
|
+
assert.deepEqual((call?.body.messages as unknown[] | undefined)?.[0], {
|
|
318
|
+
role: "system",
|
|
319
|
+
content: [{
|
|
320
|
+
type: "text",
|
|
321
|
+
text: "stable system packet",
|
|
322
|
+
cache_control: { type: "ephemeral" },
|
|
323
|
+
}],
|
|
324
|
+
});
|
|
325
|
+
assert.deepEqual((call?.body.messages as unknown[] | undefined)?.[1], {
|
|
326
|
+
role: "user",
|
|
327
|
+
content: "changing user packet",
|
|
328
|
+
});
|
|
329
|
+
});
|
|
107
330
|
});
|
|
108
331
|
|
|
109
332
|
test("cataloged unknown model fails unless its context is explicit", () => {
|
|
@@ -119,22 +342,46 @@ test("cataloged unknown model fails unless its context is explicit", () => {
|
|
|
119
342
|
assert.equal(provider?.contextWindow, 8192);
|
|
120
343
|
});
|
|
121
344
|
|
|
122
|
-
test("
|
|
123
|
-
|
|
345
|
+
test("Models.dev is the only fallback rate table", async () => {
|
|
346
|
+
mock.method(globalThis, "fetch", async () => new Response([
|
|
347
|
+
`data: ${JSON.stringify({
|
|
348
|
+
id: "response",
|
|
349
|
+
model: "served",
|
|
350
|
+
choices: [{ delta: { content: "ok" }, finish_reason: "stop" }],
|
|
351
|
+
})}`,
|
|
352
|
+
`data: ${JSON.stringify({
|
|
353
|
+
id: "response",
|
|
354
|
+
model: "served",
|
|
355
|
+
choices: [],
|
|
356
|
+
usage: {
|
|
357
|
+
prompt_tokens: 1_000,
|
|
358
|
+
prompt_tokens_details: { cached_tokens: 400 },
|
|
359
|
+
completion_tokens: 100,
|
|
360
|
+
total_tokens: 1_150,
|
|
361
|
+
},
|
|
362
|
+
})}`,
|
|
363
|
+
"data: [DONE]",
|
|
364
|
+
].join("\n\n"), { headers: { "content-type": "text/event-stream" } }));
|
|
365
|
+
const cataloged = catalogProviderFromEnv("deepseek", {
|
|
366
|
+
...env,
|
|
367
|
+
DEEPSEEK_API_KEY: "test-key",
|
|
368
|
+
}, "deepseek-v4-flash");
|
|
369
|
+
assert.notEqual(cataloged, null);
|
|
370
|
+
const catalogedResponse = await cataloged!.generate({ workerId: "cataloged", messages: [] });
|
|
371
|
+
assert.deepEqual(catalogedResponse.accounting[0]?.cost, {
|
|
372
|
+
kind: "estimated",
|
|
373
|
+
amount: { amount: "0.00012712", currency: "USD" },
|
|
374
|
+
source: "Models.dev catalog rates",
|
|
375
|
+
});
|
|
376
|
+
|
|
377
|
+
const uncataloged = catalogProviderFromEnv("xai", {
|
|
124
378
|
...env,
|
|
125
379
|
XAI_API_KEY: "test-key",
|
|
126
380
|
PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
cached: 250_000,
|
|
134
|
-
completion: 500_000,
|
|
135
|
-
reasoning: 500_000,
|
|
136
|
-
total: 2_000_000,
|
|
137
|
-
};
|
|
138
|
-
assert.equal(catalogProviderFromEnv("xai", priced, "not-in-the-catalog")?.calculateCost(usage), 9.625);
|
|
139
|
-
assert.equal(catalogProviderFromEnv("openai", priced, "gpt-4.1-mini")?.calculateCost(usage), 9.625);
|
|
381
|
+
}, "not-in-the-catalog");
|
|
382
|
+
const uncatalogedResponse = await uncataloged!.generate({ workerId: "uncataloged", messages: [] });
|
|
383
|
+
assert.deepEqual(uncatalogedResponse.accounting[0]?.cost, {
|
|
384
|
+
kind: "unknown",
|
|
385
|
+
reason: "Models.dev has no complete rate for this model",
|
|
386
|
+
});
|
|
140
387
|
});
|
package/src/catalogProvider.ts
CHANGED
|
@@ -6,22 +6,23 @@ import {
|
|
|
6
6
|
envelopeFromEnv,
|
|
7
7
|
parseRequiredFloat,
|
|
8
8
|
parseRequiredInt,
|
|
9
|
-
|
|
9
|
+
parseTimeoutMs,
|
|
10
|
+
cacheAffinityFromEnv,
|
|
11
|
+
cacheWritePolicyFromEnv,
|
|
10
12
|
reasoningFromEnv,
|
|
11
13
|
reasoningResponseStyleFromEnv,
|
|
12
14
|
resolveReserve,
|
|
13
|
-
tokenRatesFromEnv,
|
|
14
15
|
type ReserveSpec,
|
|
15
16
|
} from "./env.ts";
|
|
16
17
|
import AiSdkProvider, { type ReasoningStyle } from "./AiSdkProvider.ts";
|
|
17
18
|
import { configuredProviderInfo, createSdkModel } from "./sdkModels.ts";
|
|
18
19
|
import { providerSource } from "./notices.ts";
|
|
19
|
-
import type { Provider,
|
|
20
|
-
import {
|
|
20
|
+
import type { Provider, ProviderCostNormalizer } from "./types.ts";
|
|
21
|
+
import { estimateProviderCost } from "./cost.ts";
|
|
21
22
|
import { emitWarningOnce } from "./warnings.ts";
|
|
22
23
|
import type { LanguageModel } from "ai";
|
|
24
|
+
import type { AiSdkProviderOptions, CacheAffinity } from "./AiSdkProvider.ts";
|
|
23
25
|
import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
|
|
24
|
-
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
25
26
|
|
|
26
27
|
const reasoningStyleFromEnv = (
|
|
27
28
|
env: NodeJS.ProcessEnv,
|
|
@@ -45,24 +46,32 @@ export const providerFromSdkModel = ({
|
|
|
45
46
|
env,
|
|
46
47
|
model,
|
|
47
48
|
languageModel,
|
|
49
|
+
normalizeCost,
|
|
48
50
|
url,
|
|
49
51
|
headers,
|
|
50
52
|
contextWindow,
|
|
51
53
|
info,
|
|
52
54
|
attributions,
|
|
55
|
+
cacheAffinity,
|
|
56
|
+
systemCacheProviderOptions,
|
|
57
|
+
reasoningResponseProviderOptions,
|
|
53
58
|
}: {
|
|
54
59
|
name: string;
|
|
55
60
|
env: NodeJS.ProcessEnv;
|
|
56
61
|
model: string;
|
|
57
62
|
languageModel?: LanguageModel;
|
|
63
|
+
normalizeCost?: ProviderCostNormalizer;
|
|
58
64
|
url?: string;
|
|
59
65
|
headers?: Readonly<Record<string, string>>;
|
|
60
66
|
contextWindow: number;
|
|
61
67
|
info?: ModelInfo;
|
|
62
68
|
attributions?: (context: PluginAttributionContext) => PluginAttribution;
|
|
69
|
+
cacheAffinity?: CacheAffinity;
|
|
70
|
+
systemCacheProviderOptions?: AiSdkProviderOptions;
|
|
71
|
+
reasoningResponseProviderOptions?: AiSdkProviderOptions;
|
|
63
72
|
}): Provider => {
|
|
64
73
|
emitWarningOnce(
|
|
65
|
-
`${name} provider:
|
|
74
|
+
`${name} provider: request-level prompt counting is a chars/2 estimate; hard context-envelope admission fails closed without exact or bounded evidence`,
|
|
66
75
|
"PLURNK_PROMPT_COUNT_ESTIMATE",
|
|
67
76
|
);
|
|
68
77
|
|
|
@@ -82,36 +91,34 @@ export const providerFromSdkModel = ({
|
|
|
82
91
|
? configuredReasoning
|
|
83
92
|
: { tokens: Math.round(completionTokens / 2) };
|
|
84
93
|
|
|
85
|
-
const configuredRates = tokenRatesFromEnv(env, name);
|
|
86
94
|
const catalogCost = info?.cost;
|
|
87
|
-
const rates =
|
|
95
|
+
const rates = catalogCost === undefined ? null : {
|
|
88
96
|
input: catalogCost.inputPer1M,
|
|
89
97
|
output: catalogCost.outputPer1M,
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
kind: "estimated",
|
|
102
|
-
usd: String(calculateCostUsd(usage, rates)),
|
|
103
|
-
source: rateSource,
|
|
104
|
-
});
|
|
98
|
+
...(catalogCost.cacheReadPer1M === undefined
|
|
99
|
+
? {}
|
|
100
|
+
: { cacheRead: catalogCost.cacheReadPer1M }),
|
|
101
|
+
...(catalogCost.cacheWritePer1M === undefined
|
|
102
|
+
? {}
|
|
103
|
+
: { cacheWrite: catalogCost.cacheWritePer1M }),
|
|
104
|
+
};
|
|
105
|
+
const estimateCost = (usage: Parameters<typeof estimateProviderCost>[0]) =>
|
|
106
|
+
estimateProviderCost(usage, rates, "Models.dev catalog rates");
|
|
107
|
+
const affinityEnabled = cacheAffinityFromEnv(env, name);
|
|
108
|
+
const cacheWritePolicy = cacheWritePolicyFromEnv(env, name);
|
|
105
109
|
|
|
106
110
|
return new AiSdkProvider({
|
|
107
111
|
model,
|
|
108
112
|
...(attributions === undefined ? {} : { attributions }),
|
|
109
113
|
...(languageModel === undefined ? {} : { languageModel }),
|
|
114
|
+
...(normalizeCost === undefined ? {} : { normalizeCost }),
|
|
110
115
|
...(url === undefined ? {} : { url }),
|
|
111
116
|
...(headers === undefined ? {} : { headers: { ...headers } }),
|
|
112
117
|
contextWindow,
|
|
113
|
-
fetchTimeoutMs:
|
|
114
|
-
|
|
118
|
+
fetchTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", name),
|
|
119
|
+
operationTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_OPERATION_TIMEOUT, "PLURNK_PROVIDERS_OPERATION_TIMEOUT", name),
|
|
120
|
+
firstContentTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT, "PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT", name),
|
|
121
|
+
streamIdleTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", name),
|
|
115
122
|
reasoning,
|
|
116
123
|
reasoningResponseStyle: reasoningResponseStyleFromEnv(env, name),
|
|
117
124
|
temperature: parseRequiredFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", name, 0),
|
|
@@ -122,10 +129,15 @@ export const providerFromSdkModel = ({
|
|
|
122
129
|
retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", name),
|
|
123
130
|
errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", name),
|
|
124
131
|
reasoningStyle: reasoningStyleFromEnv(env, name),
|
|
125
|
-
|
|
132
|
+
...(affinityEnabled && cacheAffinity !== undefined ? { cacheAffinity } : {}),
|
|
133
|
+
...(cacheWritePolicy === "stable-system" && systemCacheProviderOptions !== undefined
|
|
134
|
+
? { systemCacheProviderOptions }
|
|
135
|
+
: {}),
|
|
136
|
+
...(reasoningResponseProviderOptions === undefined
|
|
137
|
+
? {}
|
|
138
|
+
: { reasoningResponseProviderOptions }),
|
|
126
139
|
serviceTier: env.PLURNK_PROVIDERS_SERVICE_TIER,
|
|
127
|
-
|
|
128
|
-
calculateCharge,
|
|
140
|
+
estimateCost,
|
|
129
141
|
source: providerSource(name),
|
|
130
142
|
gbnfDebug: env.PLURNK_PROVIDERS_GBNF_DEBUG !== undefined
|
|
131
143
|
&& env.PLURNK_PROVIDERS_GBNF_DEBUG !== ""
|
|
@@ -165,8 +177,12 @@ export const catalogProviderFromEnv = (
|
|
|
165
177
|
env,
|
|
166
178
|
model: wireModel,
|
|
167
179
|
languageModel: sdk.languageModel,
|
|
180
|
+
normalizeCost: sdk.normalizeCost,
|
|
168
181
|
url: sdk.compatible?.url,
|
|
169
182
|
headers: sdk.compatible?.headers,
|
|
183
|
+
cacheAffinity: sdk.cacheAffinity,
|
|
184
|
+
systemCacheProviderOptions: sdk.systemCacheProviderOptions,
|
|
185
|
+
reasoningResponseProviderOptions: sdk.reasoningResponseProviderOptions,
|
|
170
186
|
contextWindow,
|
|
171
187
|
info,
|
|
172
188
|
});
|
|
@@ -5,6 +5,8 @@ import { compatibleProviderFromEnv } from "./compatibleProvider.ts";
|
|
|
5
5
|
const env = {
|
|
6
6
|
OPENAI_BASE_URL: "http://local.test/v1",
|
|
7
7
|
PLURNK_PROVIDERS_FETCH_TIMEOUT: "1000",
|
|
8
|
+
PLURNK_PROVIDERS_OPERATION_TIMEOUT: "3000",
|
|
9
|
+
PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT: "1000",
|
|
8
10
|
PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
|
|
9
11
|
PLURNK_PROVIDERS_REASONING: "off",
|
|
10
12
|
PLURNK_PROVIDERS_TEMPERATURE: "0.2",
|
|
@@ -16,12 +18,13 @@ const env = {
|
|
|
16
18
|
PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT: "512",
|
|
17
19
|
PLURNK_PROVIDERS_PROBE_ATTEMPTS: "1",
|
|
18
20
|
PLURNK_PROVIDERS_PROBE_DELAY: "0",
|
|
19
|
-
|
|
21
|
+
PLURNK_PROVIDERS_CACHE_AFFINITY: "1",
|
|
22
|
+
PLURNK_PROVIDERS_CACHE_WRITE_POLICY: "stable-system",
|
|
20
23
|
};
|
|
21
24
|
|
|
22
25
|
test.afterEach(() => mock.restoreAll());
|
|
23
26
|
|
|
24
|
-
test("compatible
|
|
27
|
+
test("an undifferentiated compatible endpoint receives no guessed prompt-cache field", async () => {
|
|
25
28
|
let body: Record<string, unknown> | undefined;
|
|
26
29
|
mock.method(globalThis, "fetch", async (input: string | URL | Request, init?: RequestInit) => {
|
|
27
30
|
if (String(input).endsWith("/models")) {
|
|
@@ -41,7 +44,7 @@ test("compatible endpoints preserve configured prompt-cache affinity", async ()
|
|
|
41
44
|
messages: [{ role: "user", content: "hello" }],
|
|
42
45
|
});
|
|
43
46
|
|
|
44
|
-
assert.equal(body
|
|
47
|
+
assert.equal("prompt_cache_key" in (body ?? {}), false);
|
|
45
48
|
});
|
|
46
49
|
|
|
47
50
|
test("the server-wide DRY-off floor emits no DRY request fields", async () => {
|
|
@@ -8,11 +8,14 @@ import {
|
|
|
8
8
|
parseOptionalInt,
|
|
9
9
|
parseRequiredFloat,
|
|
10
10
|
parseRequiredInt,
|
|
11
|
-
|
|
11
|
+
parseTimeoutMs,
|
|
12
|
+
cacheAffinityFromEnv,
|
|
13
|
+
cacheWritePolicyFromEnv,
|
|
12
14
|
reasoningFromEnv,
|
|
13
15
|
reasoningResponseStyleFromEnv,
|
|
14
16
|
} from "./env.ts";
|
|
15
17
|
import { providerSource } from "./notices.ts";
|
|
18
|
+
import { plurnkCostNormalizer } from "./accounting.ts";
|
|
16
19
|
import type { Provider } from "./types.ts";
|
|
17
20
|
import { emitWarningOnce } from "./warnings.ts";
|
|
18
21
|
|
|
@@ -49,7 +52,10 @@ const probeModels = async (
|
|
|
49
52
|
): Promise<EndpointProbe> => {
|
|
50
53
|
const modelsUrl = url.replace(/\/chat\/completions$/, "/models");
|
|
51
54
|
try {
|
|
52
|
-
const response = await fetch(modelsUrl, {
|
|
55
|
+
const response = await fetch(modelsUrl, {
|
|
56
|
+
headers,
|
|
57
|
+
...(timeout > 0 ? { signal: AbortSignal.timeout(timeout) } : {}),
|
|
58
|
+
});
|
|
53
59
|
if (!response.ok) return { nCtx: null, llamaServer: false, servedModel: null, failed: true };
|
|
54
60
|
const data = await response.json() as {
|
|
55
61
|
data?: Array<{ id?: string; n_ctx?: number; meta?: { n_ctx?: number } }>;
|
|
@@ -93,7 +99,7 @@ const probeProps = async (
|
|
|
93
99
|
try {
|
|
94
100
|
const response = await fetch(url.replace(/\/v1\/chat\/completions$/, "/props"), {
|
|
95
101
|
headers,
|
|
96
|
-
signal: AbortSignal.timeout(timeout),
|
|
102
|
+
...(timeout > 0 ? { signal: AbortSignal.timeout(timeout) } : {}),
|
|
97
103
|
});
|
|
98
104
|
if (!response.ok) return { slotCount: null, eosText: null };
|
|
99
105
|
const data = await response.json() as { total_slots?: number; eos_token?: string };
|
|
@@ -112,12 +118,17 @@ export const compatibleProviderFromEnv = async (
|
|
|
112
118
|
model: string,
|
|
113
119
|
baseUrlOverride?: string,
|
|
114
120
|
): Promise<Provider> => {
|
|
121
|
+
// The knobs remain universal and fail hard when malformed, but this local /
|
|
122
|
+
// first-party compatible route declares no vendor cache projection. llama-server
|
|
123
|
+
// already owns slot affinity and the first-party endpoint receives worker metadata.
|
|
124
|
+
cacheAffinityFromEnv(env, provider);
|
|
125
|
+
cacheWritePolicyFromEnv(env, provider);
|
|
115
126
|
const url = chatUrl(provider, env, baseUrlOverride);
|
|
116
127
|
const apiKey = provider === "openai" ? env.OPENAI_API_KEY : env.PLURNK_API_KEY;
|
|
117
128
|
const headers: Record<string, string> = apiKey === undefined || apiKey.length === 0
|
|
118
129
|
? {}
|
|
119
130
|
: { Authorization: `Bearer ${apiKey}` };
|
|
120
|
-
const timeout =
|
|
131
|
+
const timeout = parseTimeoutMs(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", provider);
|
|
121
132
|
const attempts = parseRequiredInt(env.PLURNK_PROVIDERS_PROBE_ATTEMPTS, "PLURNK_PROVIDERS_PROBE_ATTEMPTS", provider);
|
|
122
133
|
const probe = await probeModelsRetrying(
|
|
123
134
|
url,
|
|
@@ -165,7 +176,7 @@ export const compatibleProviderFromEnv = async (
|
|
|
165
176
|
|
|
166
177
|
if (!llamaServer) {
|
|
167
178
|
emitWarningOnce(
|
|
168
|
-
`${provider} provider:
|
|
179
|
+
`${provider} provider: request-level prompt counting is a chars/2 estimate; hard context-envelope admission fails closed without exact or bounded evidence`,
|
|
169
180
|
"PLURNK_PROMPT_COUNT_ESTIMATE",
|
|
170
181
|
);
|
|
171
182
|
}
|
|
@@ -176,7 +187,9 @@ export const compatibleProviderFromEnv = async (
|
|
|
176
187
|
headers,
|
|
177
188
|
contextWindow,
|
|
178
189
|
fetchTimeoutMs: timeout,
|
|
179
|
-
|
|
190
|
+
operationTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_OPERATION_TIMEOUT, "PLURNK_PROVIDERS_OPERATION_TIMEOUT", provider),
|
|
191
|
+
firstContentTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT, "PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT", provider),
|
|
192
|
+
streamIdleTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", provider),
|
|
180
193
|
reasoning: reasoningFromEnv(env, provider),
|
|
181
194
|
reasoningResponseStyle: reasoningResponseStyleFromEnv(env, provider),
|
|
182
195
|
reasoningStyle,
|
|
@@ -192,7 +205,6 @@ export const compatibleProviderFromEnv = async (
|
|
|
192
205
|
tuningFloors: provider !== "plurnk",
|
|
193
206
|
retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", provider),
|
|
194
207
|
errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", provider),
|
|
195
|
-
promptCacheKey: promptCacheKeyFromEnv(env, provider),
|
|
196
208
|
source: providerSource(provider),
|
|
197
209
|
grammarStyle,
|
|
198
210
|
gbnfDebug: env.PLURNK_PROVIDERS_GBNF_DEBUG !== undefined
|
|
@@ -200,6 +212,7 @@ export const compatibleProviderFromEnv = async (
|
|
|
200
212
|
&& env.PLURNK_PROVIDERS_GBNF_DEBUG !== "0",
|
|
201
213
|
...dataCaptureFromEnv(env, provider),
|
|
202
214
|
firstPartyMetadata: provider === "plurnk",
|
|
215
|
+
normalizeCost: provider === "plurnk" ? plurnkCostNormalizer : undefined,
|
|
203
216
|
apiKeyRejectedMessage: provider === "plurnk"
|
|
204
217
|
? "PLURNK_API_KEY was rejected by plurnk.ai (invalid or expired)."
|
|
205
218
|
: undefined,
|