@plurnk/plurnk-providers 1.5.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +36 -22
- package/SPEC.md +133 -59
- package/dist/AiSdkProvider.d.ts +19 -26
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +318 -106
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +4 -9
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +36 -9
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +2 -21
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +19 -14
- package/dist/Pool.js.map +1 -1
- package/dist/accounting.d.ts +5 -2
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js +100 -16
- package/dist/accounting.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +9 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +160 -62
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +7 -3
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +30 -24
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +18 -7
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +10 -10
- package/dist/cost.d.ts.map +1 -1
- package/dist/cost.js +90 -42
- package/dist/cost.js.map +1 -1
- package/dist/env.d.ts +5 -1
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +30 -10
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +14 -2
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +58 -2
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +4 -4
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -2
- package/dist/index.js.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/sdkModels.d.ts +6 -2
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +38 -5
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +33 -31
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +21 -5
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +164 -83
- package/dist/usage.js.map +1 -1
- package/package.json +7 -6
- package/src/AiSdkProvider.test.ts +788 -191
- package/src/AiSdkProvider.ts +381 -124
- package/src/Mock.test.ts +37 -12
- package/src/Mock.ts +45 -14
- package/src/Pool.test.ts +19 -6
- package/src/Pool.ts +20 -16
- package/src/ProviderRegistry.test.ts +16 -11
- package/src/accounting.test.ts +58 -22
- package/src/accounting.ts +120 -18
- package/src/aiSdkTransport.test.ts +42 -49
- package/src/aiSdkTransport.ts +174 -62
- package/src/boundaries.test.ts +1 -0
- package/src/catalogProvider.test.ts +258 -22
- package/src/catalogProvider.ts +42 -27
- package/src/compatibleProvider.test.ts +6 -3
- package/src/compatibleProvider.ts +20 -7
- package/src/cost.test.ts +55 -36
- package/src/cost.ts +111 -50
- package/src/defaults.test.ts +13 -3
- package/src/env.test.ts +54 -5
- package/src/env.ts +43 -18
- package/src/errors.test.ts +47 -2
- package/src/errors.ts +67 -3
- package/src/index.ts +21 -5
- package/src/ollama.test.ts +4 -1
- package/src/ollama.ts +3 -3
- package/src/sdkModels.test.ts +76 -4
- package/src/sdkModels.ts +45 -7
- package/src/types.ts +77 -38
- package/src/usage.test.ts +112 -116
- package/src/usage.ts +209 -93
|
@@ -1,13 +1,16 @@
|
|
|
1
1
|
import test, { mock } from "node:test";
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
|
+
import { once } from "node:events";
|
|
4
|
+
import { createServer } from "node:http";
|
|
3
5
|
import { catalogProviderFromEnv } from "./catalogProvider.ts";
|
|
4
|
-
import { providerCostFor } from "./cost.ts";
|
|
5
6
|
import { resetEmittedWarnings } from "./warnings.ts";
|
|
6
7
|
|
|
7
8
|
const env = {
|
|
8
9
|
OPENAI_API_KEY: "test-key",
|
|
9
10
|
OPENAI_BASE_URL: "https://api.openai.com/v1",
|
|
10
11
|
PLURNK_PROVIDERS_FETCH_TIMEOUT: "1000",
|
|
12
|
+
PLURNK_PROVIDERS_OPERATION_TIMEOUT: "3000",
|
|
13
|
+
PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT: "1000",
|
|
11
14
|
PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
|
|
12
15
|
PLURNK_PROVIDERS_REASONING: "off",
|
|
13
16
|
PLURNK_PROVIDERS_TEMPERATURE: "0.2",
|
|
@@ -17,7 +20,8 @@ const env = {
|
|
|
17
20
|
PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
|
|
18
21
|
PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
|
|
19
22
|
PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT: "512",
|
|
20
|
-
|
|
23
|
+
PLURNK_PROVIDERS_CACHE_AFFINITY: "1",
|
|
24
|
+
PLURNK_PROVIDERS_CACHE_WRITE_POLICY: "stable-system",
|
|
21
25
|
};
|
|
22
26
|
|
|
23
27
|
test.afterEach(() => {
|
|
@@ -32,13 +36,6 @@ test("catalog provider resolves model physics and Models.dev USD rates", () => {
|
|
|
32
36
|
assert.equal(provider?.contextWindow, 1_047_576);
|
|
33
37
|
assert.equal(provider?.reasoningReserve, 16_384);
|
|
34
38
|
assert.equal(provider?.completionReserve, 32_768);
|
|
35
|
-
assert.ok((provider?.calculateCost({
|
|
36
|
-
prompt: 1_000_000,
|
|
37
|
-
completion: 1_000_000,
|
|
38
|
-
reasoning: 0,
|
|
39
|
-
cached: 0,
|
|
40
|
-
total: 2_000_000,
|
|
41
|
-
}) ?? 0) > 0);
|
|
42
39
|
});
|
|
43
40
|
|
|
44
41
|
test("an operator context window caps catalog physics and percentage reserves derive from the cap", () => {
|
|
@@ -97,7 +94,11 @@ test("official AI SDK provider owns the native request while PLURNK owns call se
|
|
|
97
94
|
});
|
|
98
95
|
|
|
99
96
|
assert.equal(result?.assistant.content, "done");
|
|
100
|
-
assert.equal(result?.
|
|
97
|
+
assert.equal(result?.accounting[0]?.usage?.totalTokens, 3);
|
|
98
|
+
assert.deepEqual(result?.accounting[0]?.cost, {
|
|
99
|
+
kind: "unknown",
|
|
100
|
+
reason: "the provider response omitted a token category with a distinct Models.dev rate",
|
|
101
|
+
});
|
|
101
102
|
assert.equal(calls.length, 1);
|
|
102
103
|
assert.equal(calls[0]?.url, "https://api.openai.com/v1/chat/completions");
|
|
103
104
|
assert.equal(calls[0]?.body.model, "gpt-4.1-mini");
|
|
@@ -105,6 +106,227 @@ test("official AI SDK provider owns the native request while PLURNK owns call se
|
|
|
105
106
|
assert.equal(calls[0]?.body.top_p, 0.8);
|
|
106
107
|
assert.equal(calls[0]?.body.seed, 7);
|
|
107
108
|
assert.equal(calls[0]?.body.max_tokens, 64);
|
|
109
|
+
assert.equal(calls[0]?.body.prompt_cache_key, "worker", "the official OpenAI SDK projects the documented affinity key");
|
|
110
|
+
});
|
|
111
|
+
|
|
112
|
+
test("Cerebras explicit reasoning activation needs no operator effort or token budget", async () => {
|
|
113
|
+
let body: Record<string, unknown> | undefined;
|
|
114
|
+
mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
|
|
115
|
+
body = JSON.parse(String(init?.body)) as Record<string, unknown>;
|
|
116
|
+
return new Response([
|
|
117
|
+
`data: ${JSON.stringify({
|
|
118
|
+
id: "chatcmpl-cerebras",
|
|
119
|
+
object: "chat.completion.chunk",
|
|
120
|
+
created: 1,
|
|
121
|
+
model: "gemma-4-31b",
|
|
122
|
+
choices: [{ index: 0, delta: { reasoning: "consider" }, finish_reason: null }],
|
|
123
|
+
})}`,
|
|
124
|
+
`data: ${JSON.stringify({
|
|
125
|
+
id: "chatcmpl-cerebras",
|
|
126
|
+
object: "chat.completion.chunk",
|
|
127
|
+
created: 2,
|
|
128
|
+
model: "gemma-4-31b",
|
|
129
|
+
choices: [{ index: 0, delta: { content: "done" }, finish_reason: "stop" }],
|
|
130
|
+
})}`,
|
|
131
|
+
`data: ${JSON.stringify({
|
|
132
|
+
id: "chatcmpl-cerebras",
|
|
133
|
+
object: "chat.completion.chunk",
|
|
134
|
+
created: 3,
|
|
135
|
+
model: "gemma-4-31b",
|
|
136
|
+
choices: [],
|
|
137
|
+
usage: {
|
|
138
|
+
prompt_tokens: 2,
|
|
139
|
+
completion_tokens: 2,
|
|
140
|
+
total_tokens: 4,
|
|
141
|
+
completion_tokens_details: { reasoning_tokens: 1 },
|
|
142
|
+
},
|
|
143
|
+
})}`,
|
|
144
|
+
"data: [DONE]",
|
|
145
|
+
].join("\n\n"), { headers: { "content-type": "text/event-stream" } });
|
|
146
|
+
});
|
|
147
|
+
|
|
148
|
+
const provider = catalogProviderFromEnv("cerebras", {
|
|
149
|
+
...env,
|
|
150
|
+
CEREBRAS_API_KEY: "test-key",
|
|
151
|
+
PLURNK_PROVIDERS_REASONING: "on",
|
|
152
|
+
}, "gemma-4-31b");
|
|
153
|
+
const result = await provider?.generate({
|
|
154
|
+
workerId: "worker",
|
|
155
|
+
messages: [{ role: "user", content: "hello" }],
|
|
156
|
+
});
|
|
157
|
+
|
|
158
|
+
assert.equal(body?.reasoning_effort, "medium", "the native SDK projects unqualified on to its enabled posture");
|
|
159
|
+
assert.equal("thinking_budget_tokens" in (body ?? {}), false, "activation does not invent a token budget");
|
|
160
|
+
assert.equal(result?.assistant.reasoning, "consider");
|
|
161
|
+
assert.equal(result?.accounting[0]?.usage?.outputTokenDetails?.reasoningTokens, 1);
|
|
162
|
+
});
|
|
163
|
+
|
|
164
|
+
test("Google adaptive reasoning requests and preserves readable thought summaries", async () => {
|
|
165
|
+
const bodies: Array<{
|
|
166
|
+
generationConfig?: { thinkingConfig?: { includeThoughts?: boolean } };
|
|
167
|
+
}> = [];
|
|
168
|
+
mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
|
|
169
|
+
bodies.push(JSON.parse(String(init?.body)) as typeof bodies[number]);
|
|
170
|
+
return new Response(`data: ${JSON.stringify({
|
|
171
|
+
responseId: "response-gemini",
|
|
172
|
+
candidates: [{
|
|
173
|
+
content: {
|
|
174
|
+
role: "model",
|
|
175
|
+
parts: [
|
|
176
|
+
{ text: "consider", thought: true },
|
|
177
|
+
{ text: "done" },
|
|
178
|
+
],
|
|
179
|
+
},
|
|
180
|
+
finishReason: "STOP",
|
|
181
|
+
}],
|
|
182
|
+
usageMetadata: {
|
|
183
|
+
promptTokenCount: 2,
|
|
184
|
+
candidatesTokenCount: 1,
|
|
185
|
+
thoughtsTokenCount: 1,
|
|
186
|
+
totalTokenCount: 4,
|
|
187
|
+
},
|
|
188
|
+
})}\n\n`, {
|
|
189
|
+
headers: { "content-type": "text/event-stream" },
|
|
190
|
+
});
|
|
191
|
+
});
|
|
192
|
+
|
|
193
|
+
const provider = catalogProviderFromEnv("google", {
|
|
194
|
+
...env,
|
|
195
|
+
GEMINI_API_KEY: "test-key",
|
|
196
|
+
PLURNK_PROVIDERS_REASONING: "adaptive",
|
|
197
|
+
}, "gemini-3.7-flash");
|
|
198
|
+
const result = await provider?.generate({
|
|
199
|
+
workerId: "worker",
|
|
200
|
+
messages: [{ role: "user", content: "hello" }],
|
|
201
|
+
});
|
|
202
|
+
|
|
203
|
+
assert.deepEqual(bodies[0]?.generationConfig?.thinkingConfig, {
|
|
204
|
+
includeThoughts: true,
|
|
205
|
+
}, "adaptive leaves thinking depth to Google while requesting its readable summary");
|
|
206
|
+
assert.equal(result?.assistant.reasoning, "consider");
|
|
207
|
+
assert.equal(result?.assistant.content, "done");
|
|
208
|
+
assert.equal(result?.accounting[0]?.usage?.outputTokenDetails?.reasoningTokens, 1);
|
|
209
|
+
|
|
210
|
+
const disabled = catalogProviderFromEnv("google", {
|
|
211
|
+
...env,
|
|
212
|
+
GEMINI_API_KEY: "test-key",
|
|
213
|
+
PLURNK_PROVIDERS_REASONING: "off",
|
|
214
|
+
}, "gemini-3.7-flash");
|
|
215
|
+
await disabled?.generate({
|
|
216
|
+
workerId: "worker",
|
|
217
|
+
messages: [{ role: "user", content: "hello" }],
|
|
218
|
+
});
|
|
219
|
+
assert.equal(
|
|
220
|
+
bodies[1]?.generationConfig?.thinkingConfig?.includeThoughts,
|
|
221
|
+
undefined,
|
|
222
|
+
"off does not request readable thoughts",
|
|
223
|
+
);
|
|
224
|
+
});
|
|
225
|
+
|
|
226
|
+
test("native provider routes project their documented cache controls through the actual SDK request", async (t) => {
|
|
227
|
+
const calls: Array<{ url: string; headers: Headers; body: Record<string, unknown> }> = [];
|
|
228
|
+
mock.method(globalThis, "fetch", async (input: string | URL | Request, init?: RequestInit) => {
|
|
229
|
+
calls.push({
|
|
230
|
+
url: String(input),
|
|
231
|
+
headers: new Headers(init?.headers),
|
|
232
|
+
body: JSON.parse(String(init?.body)) as Record<string, unknown>,
|
|
233
|
+
});
|
|
234
|
+
return new Response([
|
|
235
|
+
`data: ${JSON.stringify({
|
|
236
|
+
id: "response",
|
|
237
|
+
object: "chat.completion.chunk",
|
|
238
|
+
created: 1,
|
|
239
|
+
model: "served",
|
|
240
|
+
choices: [{ index: 0, delta: { content: "ok" }, finish_reason: "stop" }],
|
|
241
|
+
})}`,
|
|
242
|
+
`data: ${JSON.stringify({
|
|
243
|
+
id: "response",
|
|
244
|
+
object: "chat.completion.chunk",
|
|
245
|
+
created: 2,
|
|
246
|
+
model: "served",
|
|
247
|
+
choices: [],
|
|
248
|
+
usage: { prompt_tokens: 2, completion_tokens: 1, total_tokens: 3 },
|
|
249
|
+
})}`,
|
|
250
|
+
"data: [DONE]",
|
|
251
|
+
].join("\n\n"), { headers: { "content-type": "text/event-stream" } });
|
|
252
|
+
});
|
|
253
|
+
|
|
254
|
+
await t.test("DeepInfra native options become prompt_cache_key", async () => {
|
|
255
|
+
const provider = catalogProviderFromEnv("deepinfra", {
|
|
256
|
+
...env,
|
|
257
|
+
DEEPINFRA_API_KEY: "test-key",
|
|
258
|
+
}, "zai-org/GLM-5.2");
|
|
259
|
+
await provider?.generate({
|
|
260
|
+
workerId: "deepinfra-worker",
|
|
261
|
+
messages: [{ role: "user", content: "hello" }],
|
|
262
|
+
});
|
|
263
|
+
assert.equal(calls.at(-1)?.body.prompt_cache_key, "deepinfra-worker");
|
|
264
|
+
});
|
|
265
|
+
|
|
266
|
+
await t.test("OpenRouter carries session affinity and an Anthropic system breakpoint", async () => {
|
|
267
|
+
let call: { headers: Headers; body: Record<string, unknown> } | undefined;
|
|
268
|
+
const server = createServer(async (request, response) => {
|
|
269
|
+
const chunks: Buffer[] = [];
|
|
270
|
+
for await (const chunk of request) chunks.push(Buffer.from(chunk));
|
|
271
|
+
call = {
|
|
272
|
+
headers: new Headers(request.headers as Record<string, string>),
|
|
273
|
+
body: JSON.parse(Buffer.concat(chunks).toString("utf8")) as Record<string, unknown>,
|
|
274
|
+
};
|
|
275
|
+
response.writeHead(200, { "content-type": "text/event-stream" });
|
|
276
|
+
response.end([
|
|
277
|
+
`data: ${JSON.stringify({
|
|
278
|
+
id: "response",
|
|
279
|
+
object: "chat.completion.chunk",
|
|
280
|
+
created: 1,
|
|
281
|
+
model: "served",
|
|
282
|
+
choices: [{ index: 0, delta: { content: "ok" }, finish_reason: "stop" }],
|
|
283
|
+
})}`,
|
|
284
|
+
`data: ${JSON.stringify({
|
|
285
|
+
id: "response",
|
|
286
|
+
object: "chat.completion.chunk",
|
|
287
|
+
created: 2,
|
|
288
|
+
model: "served",
|
|
289
|
+
choices: [],
|
|
290
|
+
usage: { prompt_tokens: 2, completion_tokens: 1, total_tokens: 3 },
|
|
291
|
+
})}`,
|
|
292
|
+
"data: [DONE]",
|
|
293
|
+
].join("\n\n"));
|
|
294
|
+
});
|
|
295
|
+
server.listen(0, "127.0.0.1");
|
|
296
|
+
await once(server, "listening");
|
|
297
|
+
t.after(() => new Promise<void>((resolve, reject) => {
|
|
298
|
+
server.close((error) => error === undefined ? resolve() : reject(error));
|
|
299
|
+
}));
|
|
300
|
+
const address = server.address();
|
|
301
|
+
if (address === null || typeof address === "string") {
|
|
302
|
+
throw new Error("OpenRouter request-capture server did not bind a TCP address");
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
const provider = catalogProviderFromEnv("openrouter", {
|
|
306
|
+
...env,
|
|
307
|
+
OPENROUTER_API_KEY: "test-key",
|
|
308
|
+
}, "anthropic/claude-sonnet-4.6", `http://127.0.0.1:${address.port}/api/v1`);
|
|
309
|
+
await provider?.generate({
|
|
310
|
+
workerId: "openrouter-worker",
|
|
311
|
+
messages: [
|
|
312
|
+
{ role: "system", content: "stable system packet" },
|
|
313
|
+
{ role: "user", content: "changing user packet" },
|
|
314
|
+
],
|
|
315
|
+
});
|
|
316
|
+
assert.equal(call?.headers.get("x-session-id"), "openrouter-worker");
|
|
317
|
+
assert.deepEqual((call?.body.messages as unknown[] | undefined)?.[0], {
|
|
318
|
+
role: "system",
|
|
319
|
+
content: [{
|
|
320
|
+
type: "text",
|
|
321
|
+
text: "stable system packet",
|
|
322
|
+
cache_control: { type: "ephemeral" },
|
|
323
|
+
}],
|
|
324
|
+
});
|
|
325
|
+
assert.deepEqual((call?.body.messages as unknown[] | undefined)?.[1], {
|
|
326
|
+
role: "user",
|
|
327
|
+
content: "changing user packet",
|
|
328
|
+
});
|
|
329
|
+
});
|
|
108
330
|
});
|
|
109
331
|
|
|
110
332
|
test("cataloged unknown model fails unless its context is explicit", () => {
|
|
@@ -120,22 +342,35 @@ test("cataloged unknown model fails unless its context is explicit", () => {
|
|
|
120
342
|
assert.equal(provider?.contextWindow, 8192);
|
|
121
343
|
});
|
|
122
344
|
|
|
123
|
-
test("Models.dev is the only fallback rate table", () => {
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
345
|
+
test("Models.dev is the only fallback rate table", async () => {
|
|
346
|
+
mock.method(globalThis, "fetch", async () => new Response([
|
|
347
|
+
`data: ${JSON.stringify({
|
|
348
|
+
id: "response",
|
|
349
|
+
model: "served",
|
|
350
|
+
choices: [{ delta: { content: "ok" }, finish_reason: "stop" }],
|
|
351
|
+
})}`,
|
|
352
|
+
`data: ${JSON.stringify({
|
|
353
|
+
id: "response",
|
|
354
|
+
model: "served",
|
|
355
|
+
choices: [],
|
|
356
|
+
usage: {
|
|
357
|
+
prompt_tokens: 1_000,
|
|
358
|
+
prompt_tokens_details: { cached_tokens: 400 },
|
|
359
|
+
completion_tokens: 100,
|
|
360
|
+
total_tokens: 1_150,
|
|
361
|
+
},
|
|
362
|
+
})}`,
|
|
363
|
+
"data: [DONE]",
|
|
364
|
+
].join("\n\n"), { headers: { "content-type": "text/event-stream" } }));
|
|
131
365
|
const cataloged = catalogProviderFromEnv("deepseek", {
|
|
132
366
|
...env,
|
|
133
367
|
DEEPSEEK_API_KEY: "test-key",
|
|
134
368
|
}, "deepseek-v4-flash");
|
|
135
369
|
assert.notEqual(cataloged, null);
|
|
136
|
-
|
|
370
|
+
const catalogedResponse = await cataloged!.generate({ workerId: "cataloged", messages: [] });
|
|
371
|
+
assert.deepEqual(catalogedResponse.accounting[0]?.cost, {
|
|
137
372
|
kind: "estimated",
|
|
138
|
-
|
|
373
|
+
amount: { amount: "0.00012712", currency: "USD" },
|
|
139
374
|
source: "Models.dev catalog rates",
|
|
140
375
|
});
|
|
141
376
|
|
|
@@ -144,8 +379,9 @@ test("Models.dev is the only fallback rate table", () => {
|
|
|
144
379
|
XAI_API_KEY: "test-key",
|
|
145
380
|
PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
|
|
146
381
|
}, "not-in-the-catalog");
|
|
147
|
-
|
|
382
|
+
const uncatalogedResponse = await uncataloged!.generate({ workerId: "uncataloged", messages: [] });
|
|
383
|
+
assert.deepEqual(uncatalogedResponse.accounting[0]?.cost, {
|
|
148
384
|
kind: "unknown",
|
|
149
|
-
reason: "
|
|
385
|
+
reason: "Models.dev has no complete rate for this model",
|
|
150
386
|
});
|
|
151
387
|
});
|
package/src/catalogProvider.ts
CHANGED
|
@@ -6,7 +6,9 @@ import {
|
|
|
6
6
|
envelopeFromEnv,
|
|
7
7
|
parseRequiredFloat,
|
|
8
8
|
parseRequiredInt,
|
|
9
|
-
|
|
9
|
+
parseTimeoutMs,
|
|
10
|
+
cacheAffinityFromEnv,
|
|
11
|
+
cacheWritePolicyFromEnv,
|
|
10
12
|
reasoningFromEnv,
|
|
11
13
|
reasoningResponseStyleFromEnv,
|
|
12
14
|
resolveReserve,
|
|
@@ -15,12 +17,12 @@ import {
|
|
|
15
17
|
import AiSdkProvider, { type ReasoningStyle } from "./AiSdkProvider.ts";
|
|
16
18
|
import { configuredProviderInfo, createSdkModel } from "./sdkModels.ts";
|
|
17
19
|
import { providerSource } from "./notices.ts";
|
|
18
|
-
import type {
|
|
19
|
-
import {
|
|
20
|
+
import type { Provider, ProviderCostNormalizer } from "./types.ts";
|
|
21
|
+
import { estimateProviderCost } from "./cost.ts";
|
|
20
22
|
import { emitWarningOnce } from "./warnings.ts";
|
|
21
23
|
import type { LanguageModel } from "ai";
|
|
24
|
+
import type { AiSdkProviderOptions, CacheAffinity } from "./AiSdkProvider.ts";
|
|
22
25
|
import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
|
|
23
|
-
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
24
26
|
|
|
25
27
|
const reasoningStyleFromEnv = (
|
|
26
28
|
env: NodeJS.ProcessEnv,
|
|
@@ -44,26 +46,32 @@ export const providerFromSdkModel = ({
|
|
|
44
46
|
env,
|
|
45
47
|
model,
|
|
46
48
|
languageModel,
|
|
47
|
-
|
|
49
|
+
normalizeCost,
|
|
48
50
|
url,
|
|
49
51
|
headers,
|
|
50
52
|
contextWindow,
|
|
51
53
|
info,
|
|
52
54
|
attributions,
|
|
55
|
+
cacheAffinity,
|
|
56
|
+
systemCacheProviderOptions,
|
|
57
|
+
reasoningResponseProviderOptions,
|
|
53
58
|
}: {
|
|
54
59
|
name: string;
|
|
55
60
|
env: NodeJS.ProcessEnv;
|
|
56
61
|
model: string;
|
|
57
62
|
languageModel?: LanguageModel;
|
|
58
|
-
|
|
63
|
+
normalizeCost?: ProviderCostNormalizer;
|
|
59
64
|
url?: string;
|
|
60
65
|
headers?: Readonly<Record<string, string>>;
|
|
61
66
|
contextWindow: number;
|
|
62
67
|
info?: ModelInfo;
|
|
63
68
|
attributions?: (context: PluginAttributionContext) => PluginAttribution;
|
|
69
|
+
cacheAffinity?: CacheAffinity;
|
|
70
|
+
systemCacheProviderOptions?: AiSdkProviderOptions;
|
|
71
|
+
reasoningResponseProviderOptions?: AiSdkProviderOptions;
|
|
64
72
|
}): Provider => {
|
|
65
73
|
emitWarningOnce(
|
|
66
|
-
`${name} provider:
|
|
74
|
+
`${name} provider: request-level prompt counting is a chars/2 estimate; hard context-envelope admission fails closed without exact or bounded evidence`,
|
|
67
75
|
"PLURNK_PROMPT_COUNT_ESTIMATE",
|
|
68
76
|
);
|
|
69
77
|
|
|
@@ -87,31 +95,30 @@ export const providerFromSdkModel = ({
|
|
|
87
95
|
const rates = catalogCost === undefined ? null : {
|
|
88
96
|
input: catalogCost.inputPer1M,
|
|
89
97
|
output: catalogCost.outputPer1M,
|
|
90
|
-
|
|
98
|
+
...(catalogCost.cacheReadPer1M === undefined
|
|
99
|
+
? {}
|
|
100
|
+
: { cacheRead: catalogCost.cacheReadPer1M }),
|
|
101
|
+
...(catalogCost.cacheWritePer1M === undefined
|
|
102
|
+
? {}
|
|
103
|
+
: { cacheWrite: catalogCost.cacheWritePer1M }),
|
|
91
104
|
};
|
|
92
|
-
const
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
const
|
|
96
|
-
? () => ({ kind: "unknown", reason: "the response reported no cost and Models.dev has no rate for this model" })
|
|
97
|
-
: rates.input === 0 && rates.output === 0 && rates.cached === 0
|
|
98
|
-
? () => ({ kind: "free", source: "Models.dev catalog rates" })
|
|
99
|
-
: (usage: ProviderUsage) => ({
|
|
100
|
-
kind: "estimated",
|
|
101
|
-
usd: calculateCostUsdDecimal(usage, rates),
|
|
102
|
-
source: "Models.dev catalog rates",
|
|
103
|
-
});
|
|
105
|
+
const estimateCost = (usage: Parameters<typeof estimateProviderCost>[0]) =>
|
|
106
|
+
estimateProviderCost(usage, rates, "Models.dev catalog rates");
|
|
107
|
+
const affinityEnabled = cacheAffinityFromEnv(env, name);
|
|
108
|
+
const cacheWritePolicy = cacheWritePolicyFromEnv(env, name);
|
|
104
109
|
|
|
105
110
|
return new AiSdkProvider({
|
|
106
111
|
model,
|
|
107
112
|
...(attributions === undefined ? {} : { attributions }),
|
|
108
113
|
...(languageModel === undefined ? {} : { languageModel }),
|
|
109
|
-
...(
|
|
114
|
+
...(normalizeCost === undefined ? {} : { normalizeCost }),
|
|
110
115
|
...(url === undefined ? {} : { url }),
|
|
111
116
|
...(headers === undefined ? {} : { headers: { ...headers } }),
|
|
112
117
|
contextWindow,
|
|
113
|
-
fetchTimeoutMs:
|
|
114
|
-
|
|
118
|
+
fetchTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", name),
|
|
119
|
+
operationTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_OPERATION_TIMEOUT, "PLURNK_PROVIDERS_OPERATION_TIMEOUT", name),
|
|
120
|
+
firstContentTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT, "PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT", name),
|
|
121
|
+
streamIdleTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", name),
|
|
115
122
|
reasoning,
|
|
116
123
|
reasoningResponseStyle: reasoningResponseStyleFromEnv(env, name),
|
|
117
124
|
temperature: parseRequiredFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", name, 0),
|
|
@@ -122,10 +129,15 @@ export const providerFromSdkModel = ({
|
|
|
122
129
|
retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", name),
|
|
123
130
|
errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", name),
|
|
124
131
|
reasoningStyle: reasoningStyleFromEnv(env, name),
|
|
125
|
-
|
|
132
|
+
...(affinityEnabled && cacheAffinity !== undefined ? { cacheAffinity } : {}),
|
|
133
|
+
...(cacheWritePolicy === "stable-system" && systemCacheProviderOptions !== undefined
|
|
134
|
+
? { systemCacheProviderOptions }
|
|
135
|
+
: {}),
|
|
136
|
+
...(reasoningResponseProviderOptions === undefined
|
|
137
|
+
? {}
|
|
138
|
+
: { reasoningResponseProviderOptions }),
|
|
126
139
|
serviceTier: env.PLURNK_PROVIDERS_SERVICE_TIER,
|
|
127
|
-
|
|
128
|
-
calculateCharge,
|
|
140
|
+
estimateCost,
|
|
129
141
|
source: providerSource(name),
|
|
130
142
|
gbnfDebug: env.PLURNK_PROVIDERS_GBNF_DEBUG !== undefined
|
|
131
143
|
&& env.PLURNK_PROVIDERS_GBNF_DEBUG !== ""
|
|
@@ -165,9 +177,12 @@ export const catalogProviderFromEnv = (
|
|
|
165
177
|
env,
|
|
166
178
|
model: wireModel,
|
|
167
179
|
languageModel: sdk.languageModel,
|
|
168
|
-
|
|
180
|
+
normalizeCost: sdk.normalizeCost,
|
|
169
181
|
url: sdk.compatible?.url,
|
|
170
182
|
headers: sdk.compatible?.headers,
|
|
183
|
+
cacheAffinity: sdk.cacheAffinity,
|
|
184
|
+
systemCacheProviderOptions: sdk.systemCacheProviderOptions,
|
|
185
|
+
reasoningResponseProviderOptions: sdk.reasoningResponseProviderOptions,
|
|
171
186
|
contextWindow,
|
|
172
187
|
info,
|
|
173
188
|
});
|
|
@@ -5,6 +5,8 @@ import { compatibleProviderFromEnv } from "./compatibleProvider.ts";
|
|
|
5
5
|
const env = {
|
|
6
6
|
OPENAI_BASE_URL: "http://local.test/v1",
|
|
7
7
|
PLURNK_PROVIDERS_FETCH_TIMEOUT: "1000",
|
|
8
|
+
PLURNK_PROVIDERS_OPERATION_TIMEOUT: "3000",
|
|
9
|
+
PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT: "1000",
|
|
8
10
|
PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
|
|
9
11
|
PLURNK_PROVIDERS_REASONING: "off",
|
|
10
12
|
PLURNK_PROVIDERS_TEMPERATURE: "0.2",
|
|
@@ -16,12 +18,13 @@ const env = {
|
|
|
16
18
|
PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT: "512",
|
|
17
19
|
PLURNK_PROVIDERS_PROBE_ATTEMPTS: "1",
|
|
18
20
|
PLURNK_PROVIDERS_PROBE_DELAY: "0",
|
|
19
|
-
|
|
21
|
+
PLURNK_PROVIDERS_CACHE_AFFINITY: "1",
|
|
22
|
+
PLURNK_PROVIDERS_CACHE_WRITE_POLICY: "stable-system",
|
|
20
23
|
};
|
|
21
24
|
|
|
22
25
|
test.afterEach(() => mock.restoreAll());
|
|
23
26
|
|
|
24
|
-
test("compatible
|
|
27
|
+
test("an undifferentiated compatible endpoint receives no guessed prompt-cache field", async () => {
|
|
25
28
|
let body: Record<string, unknown> | undefined;
|
|
26
29
|
mock.method(globalThis, "fetch", async (input: string | URL | Request, init?: RequestInit) => {
|
|
27
30
|
if (String(input).endsWith("/models")) {
|
|
@@ -41,7 +44,7 @@ test("compatible endpoints preserve configured prompt-cache affinity", async ()
|
|
|
41
44
|
messages: [{ role: "user", content: "hello" }],
|
|
42
45
|
});
|
|
43
46
|
|
|
44
|
-
assert.equal(body
|
|
47
|
+
assert.equal("prompt_cache_key" in (body ?? {}), false);
|
|
45
48
|
});
|
|
46
49
|
|
|
47
50
|
test("the server-wide DRY-off floor emits no DRY request fields", async () => {
|
|
@@ -8,11 +8,14 @@ import {
|
|
|
8
8
|
parseOptionalInt,
|
|
9
9
|
parseRequiredFloat,
|
|
10
10
|
parseRequiredInt,
|
|
11
|
-
|
|
11
|
+
parseTimeoutMs,
|
|
12
|
+
cacheAffinityFromEnv,
|
|
13
|
+
cacheWritePolicyFromEnv,
|
|
12
14
|
reasoningFromEnv,
|
|
13
15
|
reasoningResponseStyleFromEnv,
|
|
14
16
|
} from "./env.ts";
|
|
15
17
|
import { providerSource } from "./notices.ts";
|
|
18
|
+
import { plurnkCostNormalizer } from "./accounting.ts";
|
|
16
19
|
import type { Provider } from "./types.ts";
|
|
17
20
|
import { emitWarningOnce } from "./warnings.ts";
|
|
18
21
|
|
|
@@ -49,7 +52,10 @@ const probeModels = async (
|
|
|
49
52
|
): Promise<EndpointProbe> => {
|
|
50
53
|
const modelsUrl = url.replace(/\/chat\/completions$/, "/models");
|
|
51
54
|
try {
|
|
52
|
-
const response = await fetch(modelsUrl, {
|
|
55
|
+
const response = await fetch(modelsUrl, {
|
|
56
|
+
headers,
|
|
57
|
+
...(timeout > 0 ? { signal: AbortSignal.timeout(timeout) } : {}),
|
|
58
|
+
});
|
|
53
59
|
if (!response.ok) return { nCtx: null, llamaServer: false, servedModel: null, failed: true };
|
|
54
60
|
const data = await response.json() as {
|
|
55
61
|
data?: Array<{ id?: string; n_ctx?: number; meta?: { n_ctx?: number } }>;
|
|
@@ -93,7 +99,7 @@ const probeProps = async (
|
|
|
93
99
|
try {
|
|
94
100
|
const response = await fetch(url.replace(/\/v1\/chat\/completions$/, "/props"), {
|
|
95
101
|
headers,
|
|
96
|
-
signal: AbortSignal.timeout(timeout),
|
|
102
|
+
...(timeout > 0 ? { signal: AbortSignal.timeout(timeout) } : {}),
|
|
97
103
|
});
|
|
98
104
|
if (!response.ok) return { slotCount: null, eosText: null };
|
|
99
105
|
const data = await response.json() as { total_slots?: number; eos_token?: string };
|
|
@@ -112,12 +118,17 @@ export const compatibleProviderFromEnv = async (
|
|
|
112
118
|
model: string,
|
|
113
119
|
baseUrlOverride?: string,
|
|
114
120
|
): Promise<Provider> => {
|
|
121
|
+
// The knobs remain universal and fail hard when malformed, but this local /
|
|
122
|
+
// first-party compatible route declares no vendor cache projection. llama-server
|
|
123
|
+
// already owns slot affinity and the first-party endpoint receives worker metadata.
|
|
124
|
+
cacheAffinityFromEnv(env, provider);
|
|
125
|
+
cacheWritePolicyFromEnv(env, provider);
|
|
115
126
|
const url = chatUrl(provider, env, baseUrlOverride);
|
|
116
127
|
const apiKey = provider === "openai" ? env.OPENAI_API_KEY : env.PLURNK_API_KEY;
|
|
117
128
|
const headers: Record<string, string> = apiKey === undefined || apiKey.length === 0
|
|
118
129
|
? {}
|
|
119
130
|
: { Authorization: `Bearer ${apiKey}` };
|
|
120
|
-
const timeout =
|
|
131
|
+
const timeout = parseTimeoutMs(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", provider);
|
|
121
132
|
const attempts = parseRequiredInt(env.PLURNK_PROVIDERS_PROBE_ATTEMPTS, "PLURNK_PROVIDERS_PROBE_ATTEMPTS", provider);
|
|
122
133
|
const probe = await probeModelsRetrying(
|
|
123
134
|
url,
|
|
@@ -165,7 +176,7 @@ export const compatibleProviderFromEnv = async (
|
|
|
165
176
|
|
|
166
177
|
if (!llamaServer) {
|
|
167
178
|
emitWarningOnce(
|
|
168
|
-
`${provider} provider:
|
|
179
|
+
`${provider} provider: request-level prompt counting is a chars/2 estimate; hard context-envelope admission fails closed without exact or bounded evidence`,
|
|
169
180
|
"PLURNK_PROMPT_COUNT_ESTIMATE",
|
|
170
181
|
);
|
|
171
182
|
}
|
|
@@ -176,7 +187,9 @@ export const compatibleProviderFromEnv = async (
|
|
|
176
187
|
headers,
|
|
177
188
|
contextWindow,
|
|
178
189
|
fetchTimeoutMs: timeout,
|
|
179
|
-
|
|
190
|
+
operationTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_OPERATION_TIMEOUT, "PLURNK_PROVIDERS_OPERATION_TIMEOUT", provider),
|
|
191
|
+
firstContentTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT, "PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT", provider),
|
|
192
|
+
streamIdleTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", provider),
|
|
180
193
|
reasoning: reasoningFromEnv(env, provider),
|
|
181
194
|
reasoningResponseStyle: reasoningResponseStyleFromEnv(env, provider),
|
|
182
195
|
reasoningStyle,
|
|
@@ -192,7 +205,6 @@ export const compatibleProviderFromEnv = async (
|
|
|
192
205
|
tuningFloors: provider !== "plurnk",
|
|
193
206
|
retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", provider),
|
|
194
207
|
errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", provider),
|
|
195
|
-
promptCacheKey: promptCacheKeyFromEnv(env, provider),
|
|
196
208
|
source: providerSource(provider),
|
|
197
209
|
grammarStyle,
|
|
198
210
|
gbnfDebug: env.PLURNK_PROVIDERS_GBNF_DEBUG !== undefined
|
|
@@ -200,6 +212,7 @@ export const compatibleProviderFromEnv = async (
|
|
|
200
212
|
&& env.PLURNK_PROVIDERS_GBNF_DEBUG !== "0",
|
|
201
213
|
...dataCaptureFromEnv(env, provider),
|
|
202
214
|
firstPartyMetadata: provider === "plurnk",
|
|
215
|
+
normalizeCost: provider === "plurnk" ? plurnkCostNormalizer : undefined,
|
|
203
216
|
apiKeyRejectedMessage: provider === "plurnk"
|
|
204
217
|
? "PLURNK_API_KEY was rejected by plurnk.ai (invalid or expired)."
|
|
205
218
|
: undefined,
|