@plurnk/plurnk-providers 1.5.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. package/.env.defaults +36 -22
  2. package/SPEC.md +133 -59
  3. package/dist/AiSdkProvider.d.ts +19 -26
  4. package/dist/AiSdkProvider.d.ts.map +1 -1
  5. package/dist/AiSdkProvider.js +318 -106
  6. package/dist/AiSdkProvider.js.map +1 -1
  7. package/dist/Mock.d.ts +4 -9
  8. package/dist/Mock.d.ts.map +1 -1
  9. package/dist/Mock.js +36 -9
  10. package/dist/Mock.js.map +1 -1
  11. package/dist/Pool.d.ts +2 -21
  12. package/dist/Pool.d.ts.map +1 -1
  13. package/dist/Pool.js +19 -14
  14. package/dist/Pool.js.map +1 -1
  15. package/dist/accounting.d.ts +5 -2
  16. package/dist/accounting.d.ts.map +1 -1
  17. package/dist/accounting.js +100 -16
  18. package/dist/accounting.js.map +1 -1
  19. package/dist/aiSdkTransport.d.ts +9 -2
  20. package/dist/aiSdkTransport.d.ts.map +1 -1
  21. package/dist/aiSdkTransport.js +160 -62
  22. package/dist/aiSdkTransport.js.map +1 -1
  23. package/dist/catalogProvider.d.ts +7 -3
  24. package/dist/catalogProvider.d.ts.map +1 -1
  25. package/dist/catalogProvider.js +30 -24
  26. package/dist/catalogProvider.js.map +1 -1
  27. package/dist/compatibleProvider.d.ts.map +1 -1
  28. package/dist/compatibleProvider.js +18 -7
  29. package/dist/compatibleProvider.js.map +1 -1
  30. package/dist/cost.d.ts +10 -10
  31. package/dist/cost.d.ts.map +1 -1
  32. package/dist/cost.js +90 -42
  33. package/dist/cost.js.map +1 -1
  34. package/dist/env.d.ts +5 -1
  35. package/dist/env.d.ts.map +1 -1
  36. package/dist/env.js +30 -10
  37. package/dist/env.js.map +1 -1
  38. package/dist/errors.d.ts +14 -2
  39. package/dist/errors.d.ts.map +1 -1
  40. package/dist/errors.js +58 -2
  41. package/dist/errors.js.map +1 -1
  42. package/dist/index.d.ts +4 -4
  43. package/dist/index.d.ts.map +1 -1
  44. package/dist/index.js +3 -2
  45. package/dist/index.js.map +1 -1
  46. package/dist/ollama.js +3 -3
  47. package/dist/ollama.js.map +1 -1
  48. package/dist/sdkModels.d.ts +6 -2
  49. package/dist/sdkModels.d.ts.map +1 -1
  50. package/dist/sdkModels.js +38 -5
  51. package/dist/sdkModels.js.map +1 -1
  52. package/dist/types.d.ts +33 -31
  53. package/dist/types.d.ts.map +1 -1
  54. package/dist/usage.d.ts +21 -5
  55. package/dist/usage.d.ts.map +1 -1
  56. package/dist/usage.js +164 -83
  57. package/dist/usage.js.map +1 -1
  58. package/package.json +7 -6
  59. package/src/AiSdkProvider.test.ts +788 -191
  60. package/src/AiSdkProvider.ts +381 -124
  61. package/src/Mock.test.ts +37 -12
  62. package/src/Mock.ts +45 -14
  63. package/src/Pool.test.ts +19 -6
  64. package/src/Pool.ts +20 -16
  65. package/src/ProviderRegistry.test.ts +16 -11
  66. package/src/accounting.test.ts +58 -22
  67. package/src/accounting.ts +120 -18
  68. package/src/aiSdkTransport.test.ts +42 -49
  69. package/src/aiSdkTransport.ts +174 -62
  70. package/src/boundaries.test.ts +1 -0
  71. package/src/catalogProvider.test.ts +258 -22
  72. package/src/catalogProvider.ts +42 -27
  73. package/src/compatibleProvider.test.ts +6 -3
  74. package/src/compatibleProvider.ts +20 -7
  75. package/src/cost.test.ts +55 -36
  76. package/src/cost.ts +111 -50
  77. package/src/defaults.test.ts +13 -3
  78. package/src/env.test.ts +54 -5
  79. package/src/env.ts +43 -18
  80. package/src/errors.test.ts +47 -2
  81. package/src/errors.ts +67 -3
  82. package/src/index.ts +21 -5
  83. package/src/ollama.test.ts +4 -1
  84. package/src/ollama.ts +3 -3
  85. package/src/sdkModels.test.ts +76 -4
  86. package/src/sdkModels.ts +45 -7
  87. package/src/types.ts +77 -38
  88. package/src/usage.test.ts +112 -116
  89. package/src/usage.ts +209 -93
@@ -1,13 +1,16 @@
1
1
  import test, { mock } from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
+ import { once } from "node:events";
4
+ import { createServer } from "node:http";
3
5
  import { catalogProviderFromEnv } from "./catalogProvider.ts";
4
- import { providerCostFor } from "./cost.ts";
5
6
  import { resetEmittedWarnings } from "./warnings.ts";
6
7
 
7
8
  const env = {
8
9
  OPENAI_API_KEY: "test-key",
9
10
  OPENAI_BASE_URL: "https://api.openai.com/v1",
10
11
  PLURNK_PROVIDERS_FETCH_TIMEOUT: "1000",
12
+ PLURNK_PROVIDERS_OPERATION_TIMEOUT: "3000",
13
+ PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT: "1000",
11
14
  PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
12
15
  PLURNK_PROVIDERS_REASONING: "off",
13
16
  PLURNK_PROVIDERS_TEMPERATURE: "0.2",
@@ -17,7 +20,8 @@ const env = {
17
20
  PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
18
21
  PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
19
22
  PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT: "512",
20
- PLURNK_PROVIDERS_PROMPT_CACHE_KEY: "1",
23
+ PLURNK_PROVIDERS_CACHE_AFFINITY: "1",
24
+ PLURNK_PROVIDERS_CACHE_WRITE_POLICY: "stable-system",
21
25
  };
22
26
 
23
27
  test.afterEach(() => {
@@ -32,13 +36,6 @@ test("catalog provider resolves model physics and Models.dev USD rates", () => {
32
36
  assert.equal(provider?.contextWindow, 1_047_576);
33
37
  assert.equal(provider?.reasoningReserve, 16_384);
34
38
  assert.equal(provider?.completionReserve, 32_768);
35
- assert.ok((provider?.calculateCost({
36
- prompt: 1_000_000,
37
- completion: 1_000_000,
38
- reasoning: 0,
39
- cached: 0,
40
- total: 2_000_000,
41
- }) ?? 0) > 0);
42
39
  });
43
40
 
44
41
  test("an operator context window caps catalog physics and percentage reserves derive from the cap", () => {
@@ -97,7 +94,11 @@ test("official AI SDK provider owns the native request while PLURNK owns call se
97
94
  });
98
95
 
99
96
  assert.equal(result?.assistant.content, "done");
100
- assert.equal(result?.assistant.usage.total, 3);
97
+ assert.equal(result?.accounting[0]?.usage?.totalTokens, 3);
98
+ assert.deepEqual(result?.accounting[0]?.cost, {
99
+ kind: "unknown",
100
+ reason: "the provider response omitted a token category with a distinct Models.dev rate",
101
+ });
101
102
  assert.equal(calls.length, 1);
102
103
  assert.equal(calls[0]?.url, "https://api.openai.com/v1/chat/completions");
103
104
  assert.equal(calls[0]?.body.model, "gpt-4.1-mini");
@@ -105,6 +106,227 @@ test("official AI SDK provider owns the native request while PLURNK owns call se
105
106
  assert.equal(calls[0]?.body.top_p, 0.8);
106
107
  assert.equal(calls[0]?.body.seed, 7);
107
108
  assert.equal(calls[0]?.body.max_tokens, 64);
109
+ assert.equal(calls[0]?.body.prompt_cache_key, "worker", "the official OpenAI SDK projects the documented affinity key");
110
+ });
111
+
112
+ test("Cerebras explicit reasoning activation needs no operator effort or token budget", async () => {
113
+ let body: Record<string, unknown> | undefined;
114
+ mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
115
+ body = JSON.parse(String(init?.body)) as Record<string, unknown>;
116
+ return new Response([
117
+ `data: ${JSON.stringify({
118
+ id: "chatcmpl-cerebras",
119
+ object: "chat.completion.chunk",
120
+ created: 1,
121
+ model: "gemma-4-31b",
122
+ choices: [{ index: 0, delta: { reasoning: "consider" }, finish_reason: null }],
123
+ })}`,
124
+ `data: ${JSON.stringify({
125
+ id: "chatcmpl-cerebras",
126
+ object: "chat.completion.chunk",
127
+ created: 2,
128
+ model: "gemma-4-31b",
129
+ choices: [{ index: 0, delta: { content: "done" }, finish_reason: "stop" }],
130
+ })}`,
131
+ `data: ${JSON.stringify({
132
+ id: "chatcmpl-cerebras",
133
+ object: "chat.completion.chunk",
134
+ created: 3,
135
+ model: "gemma-4-31b",
136
+ choices: [],
137
+ usage: {
138
+ prompt_tokens: 2,
139
+ completion_tokens: 2,
140
+ total_tokens: 4,
141
+ completion_tokens_details: { reasoning_tokens: 1 },
142
+ },
143
+ })}`,
144
+ "data: [DONE]",
145
+ ].join("\n\n"), { headers: { "content-type": "text/event-stream" } });
146
+ });
147
+
148
+ const provider = catalogProviderFromEnv("cerebras", {
149
+ ...env,
150
+ CEREBRAS_API_KEY: "test-key",
151
+ PLURNK_PROVIDERS_REASONING: "on",
152
+ }, "gemma-4-31b");
153
+ const result = await provider?.generate({
154
+ workerId: "worker",
155
+ messages: [{ role: "user", content: "hello" }],
156
+ });
157
+
158
+ assert.equal(body?.reasoning_effort, "medium", "the native SDK projects unqualified on to its enabled posture");
159
+ assert.equal("thinking_budget_tokens" in (body ?? {}), false, "activation does not invent a token budget");
160
+ assert.equal(result?.assistant.reasoning, "consider");
161
+ assert.equal(result?.accounting[0]?.usage?.outputTokenDetails?.reasoningTokens, 1);
162
+ });
163
+
164
+ test("Google adaptive reasoning requests and preserves readable thought summaries", async () => {
165
+ const bodies: Array<{
166
+ generationConfig?: { thinkingConfig?: { includeThoughts?: boolean } };
167
+ }> = [];
168
+ mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
169
+ bodies.push(JSON.parse(String(init?.body)) as typeof bodies[number]);
170
+ return new Response(`data: ${JSON.stringify({
171
+ responseId: "response-gemini",
172
+ candidates: [{
173
+ content: {
174
+ role: "model",
175
+ parts: [
176
+ { text: "consider", thought: true },
177
+ { text: "done" },
178
+ ],
179
+ },
180
+ finishReason: "STOP",
181
+ }],
182
+ usageMetadata: {
183
+ promptTokenCount: 2,
184
+ candidatesTokenCount: 1,
185
+ thoughtsTokenCount: 1,
186
+ totalTokenCount: 4,
187
+ },
188
+ })}\n\n`, {
189
+ headers: { "content-type": "text/event-stream" },
190
+ });
191
+ });
192
+
193
+ const provider = catalogProviderFromEnv("google", {
194
+ ...env,
195
+ GEMINI_API_KEY: "test-key",
196
+ PLURNK_PROVIDERS_REASONING: "adaptive",
197
+ }, "gemini-3.7-flash");
198
+ const result = await provider?.generate({
199
+ workerId: "worker",
200
+ messages: [{ role: "user", content: "hello" }],
201
+ });
202
+
203
+ assert.deepEqual(bodies[0]?.generationConfig?.thinkingConfig, {
204
+ includeThoughts: true,
205
+ }, "adaptive leaves thinking depth to Google while requesting its readable summary");
206
+ assert.equal(result?.assistant.reasoning, "consider");
207
+ assert.equal(result?.assistant.content, "done");
208
+ assert.equal(result?.accounting[0]?.usage?.outputTokenDetails?.reasoningTokens, 1);
209
+
210
+ const disabled = catalogProviderFromEnv("google", {
211
+ ...env,
212
+ GEMINI_API_KEY: "test-key",
213
+ PLURNK_PROVIDERS_REASONING: "off",
214
+ }, "gemini-3.7-flash");
215
+ await disabled?.generate({
216
+ workerId: "worker",
217
+ messages: [{ role: "user", content: "hello" }],
218
+ });
219
+ assert.equal(
220
+ bodies[1]?.generationConfig?.thinkingConfig?.includeThoughts,
221
+ undefined,
222
+ "off does not request readable thoughts",
223
+ );
224
+ });
225
+
226
+ test("native provider routes project their documented cache controls through the actual SDK request", async (t) => {
227
+ const calls: Array<{ url: string; headers: Headers; body: Record<string, unknown> }> = [];
228
+ mock.method(globalThis, "fetch", async (input: string | URL | Request, init?: RequestInit) => {
229
+ calls.push({
230
+ url: String(input),
231
+ headers: new Headers(init?.headers),
232
+ body: JSON.parse(String(init?.body)) as Record<string, unknown>,
233
+ });
234
+ return new Response([
235
+ `data: ${JSON.stringify({
236
+ id: "response",
237
+ object: "chat.completion.chunk",
238
+ created: 1,
239
+ model: "served",
240
+ choices: [{ index: 0, delta: { content: "ok" }, finish_reason: "stop" }],
241
+ })}`,
242
+ `data: ${JSON.stringify({
243
+ id: "response",
244
+ object: "chat.completion.chunk",
245
+ created: 2,
246
+ model: "served",
247
+ choices: [],
248
+ usage: { prompt_tokens: 2, completion_tokens: 1, total_tokens: 3 },
249
+ })}`,
250
+ "data: [DONE]",
251
+ ].join("\n\n"), { headers: { "content-type": "text/event-stream" } });
252
+ });
253
+
254
+ await t.test("DeepInfra native options become prompt_cache_key", async () => {
255
+ const provider = catalogProviderFromEnv("deepinfra", {
256
+ ...env,
257
+ DEEPINFRA_API_KEY: "test-key",
258
+ }, "zai-org/GLM-5.2");
259
+ await provider?.generate({
260
+ workerId: "deepinfra-worker",
261
+ messages: [{ role: "user", content: "hello" }],
262
+ });
263
+ assert.equal(calls.at(-1)?.body.prompt_cache_key, "deepinfra-worker");
264
+ });
265
+
266
+ await t.test("OpenRouter carries session affinity and an Anthropic system breakpoint", async () => {
267
+ let call: { headers: Headers; body: Record<string, unknown> } | undefined;
268
+ const server = createServer(async (request, response) => {
269
+ const chunks: Buffer[] = [];
270
+ for await (const chunk of request) chunks.push(Buffer.from(chunk));
271
+ call = {
272
+ headers: new Headers(request.headers as Record<string, string>),
273
+ body: JSON.parse(Buffer.concat(chunks).toString("utf8")) as Record<string, unknown>,
274
+ };
275
+ response.writeHead(200, { "content-type": "text/event-stream" });
276
+ response.end([
277
+ `data: ${JSON.stringify({
278
+ id: "response",
279
+ object: "chat.completion.chunk",
280
+ created: 1,
281
+ model: "served",
282
+ choices: [{ index: 0, delta: { content: "ok" }, finish_reason: "stop" }],
283
+ })}`,
284
+ `data: ${JSON.stringify({
285
+ id: "response",
286
+ object: "chat.completion.chunk",
287
+ created: 2,
288
+ model: "served",
289
+ choices: [],
290
+ usage: { prompt_tokens: 2, completion_tokens: 1, total_tokens: 3 },
291
+ })}`,
292
+ "data: [DONE]",
293
+ ].join("\n\n"));
294
+ });
295
+ server.listen(0, "127.0.0.1");
296
+ await once(server, "listening");
297
+ t.after(() => new Promise<void>((resolve, reject) => {
298
+ server.close((error) => error === undefined ? resolve() : reject(error));
299
+ }));
300
+ const address = server.address();
301
+ if (address === null || typeof address === "string") {
302
+ throw new Error("OpenRouter request-capture server did not bind a TCP address");
303
+ }
304
+
305
+ const provider = catalogProviderFromEnv("openrouter", {
306
+ ...env,
307
+ OPENROUTER_API_KEY: "test-key",
308
+ }, "anthropic/claude-sonnet-4.6", `http://127.0.0.1:${address.port}/api/v1`);
309
+ await provider?.generate({
310
+ workerId: "openrouter-worker",
311
+ messages: [
312
+ { role: "system", content: "stable system packet" },
313
+ { role: "user", content: "changing user packet" },
314
+ ],
315
+ });
316
+ assert.equal(call?.headers.get("x-session-id"), "openrouter-worker");
317
+ assert.deepEqual((call?.body.messages as unknown[] | undefined)?.[0], {
318
+ role: "system",
319
+ content: [{
320
+ type: "text",
321
+ text: "stable system packet",
322
+ cache_control: { type: "ephemeral" },
323
+ }],
324
+ });
325
+ assert.deepEqual((call?.body.messages as unknown[] | undefined)?.[1], {
326
+ role: "user",
327
+ content: "changing user packet",
328
+ });
329
+ });
108
330
  });
109
331
 
110
332
  test("cataloged unknown model fails unless its context is explicit", () => {
@@ -120,22 +342,35 @@ test("cataloged unknown model fails unless its context is explicit", () => {
120
342
  assert.equal(provider?.contextWindow, 8192);
121
343
  });
122
344
 
123
- test("Models.dev is the only fallback rate table", () => {
124
- const usage = {
125
- prompt: 1_000,
126
- cached: 400,
127
- completion: 100,
128
- reasoning: 50,
129
- total: 1_150,
130
- };
345
+ test("Models.dev is the only fallback rate table", async () => {
346
+ mock.method(globalThis, "fetch", async () => new Response([
347
+ `data: ${JSON.stringify({
348
+ id: "response",
349
+ model: "served",
350
+ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }],
351
+ })}`,
352
+ `data: ${JSON.stringify({
353
+ id: "response",
354
+ model: "served",
355
+ choices: [],
356
+ usage: {
357
+ prompt_tokens: 1_000,
358
+ prompt_tokens_details: { cached_tokens: 400 },
359
+ completion_tokens: 100,
360
+ total_tokens: 1_150,
361
+ },
362
+ })}`,
363
+ "data: [DONE]",
364
+ ].join("\n\n"), { headers: { "content-type": "text/event-stream" } }));
131
365
  const cataloged = catalogProviderFromEnv("deepseek", {
132
366
  ...env,
133
367
  DEEPSEEK_API_KEY: "test-key",
134
368
  }, "deepseek-v4-flash");
135
369
  assert.notEqual(cataloged, null);
136
- assert.deepEqual(providerCostFor(cataloged!, usage), {
370
+ const catalogedResponse = await cataloged!.generate({ workerId: "cataloged", messages: [] });
371
+ assert.deepEqual(catalogedResponse.accounting[0]?.cost, {
137
372
  kind: "estimated",
138
- usd: "0.00012712",
373
+ amount: { amount: "0.00012712", currency: "USD" },
139
374
  source: "Models.dev catalog rates",
140
375
  });
141
376
 
@@ -144,8 +379,9 @@ test("Models.dev is the only fallback rate table", () => {
144
379
  XAI_API_KEY: "test-key",
145
380
  PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
146
381
  }, "not-in-the-catalog");
147
- assert.deepEqual(providerCostFor(uncataloged!, usage), {
382
+ const uncatalogedResponse = await uncataloged!.generate({ workerId: "uncataloged", messages: [] });
383
+ assert.deepEqual(uncatalogedResponse.accounting[0]?.cost, {
148
384
  kind: "unknown",
149
- reason: "the response reported no cost and Models.dev has no rate for this model",
385
+ reason: "Models.dev has no complete rate for this model",
150
386
  });
151
387
  });
@@ -6,7 +6,9 @@ import {
6
6
  envelopeFromEnv,
7
7
  parseRequiredFloat,
8
8
  parseRequiredInt,
9
- promptCacheKeyFromEnv,
9
+ parseTimeoutMs,
10
+ cacheAffinityFromEnv,
11
+ cacheWritePolicyFromEnv,
10
12
  reasoningFromEnv,
11
13
  reasoningResponseStyleFromEnv,
12
14
  resolveReserve,
@@ -15,12 +17,12 @@ import {
15
17
  import AiSdkProvider, { type ReasoningStyle } from "./AiSdkProvider.ts";
16
18
  import { configuredProviderInfo, createSdkModel } from "./sdkModels.ts";
17
19
  import { providerSource } from "./notices.ts";
18
- import type { AuthoritativeChargeNormalizer, Provider, ProviderUsage } from "./types.ts";
19
- import { calculateCostUsd, calculateCostUsdDecimal } from "./usage.ts";
20
+ import type { Provider, ProviderCostNormalizer } from "./types.ts";
21
+ import { estimateProviderCost } from "./cost.ts";
20
22
  import { emitWarningOnce } from "./warnings.ts";
21
23
  import type { LanguageModel } from "ai";
24
+ import type { AiSdkProviderOptions, CacheAffinity } from "./AiSdkProvider.ts";
22
25
  import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
23
- import type { ProviderCost } from "@plurnk/plurnk-contracts";
24
26
 
25
27
  const reasoningStyleFromEnv = (
26
28
  env: NodeJS.ProcessEnv,
@@ -44,26 +46,32 @@ export const providerFromSdkModel = ({
44
46
  env,
45
47
  model,
46
48
  languageModel,
47
- normalizeCharge,
49
+ normalizeCost,
48
50
  url,
49
51
  headers,
50
52
  contextWindow,
51
53
  info,
52
54
  attributions,
55
+ cacheAffinity,
56
+ systemCacheProviderOptions,
57
+ reasoningResponseProviderOptions,
53
58
  }: {
54
59
  name: string;
55
60
  env: NodeJS.ProcessEnv;
56
61
  model: string;
57
62
  languageModel?: LanguageModel;
58
- normalizeCharge?: AuthoritativeChargeNormalizer;
63
+ normalizeCost?: ProviderCostNormalizer;
59
64
  url?: string;
60
65
  headers?: Readonly<Record<string, string>>;
61
66
  contextWindow: number;
62
67
  info?: ModelInfo;
63
68
  attributions?: (context: PluginAttributionContext) => PluginAttribution;
69
+ cacheAffinity?: CacheAffinity;
70
+ systemCacheProviderOptions?: AiSdkProviderOptions;
71
+ reasoningResponseProviderOptions?: AiSdkProviderOptions;
64
72
  }): Provider => {
65
73
  emitWarningOnce(
66
- `${name} provider: physical prompt counting is a chars/2 estimate; over-policy recovery fails closed without exact or bounded request evidence`,
74
+ `${name} provider: request-level prompt counting is a chars/2 estimate; hard context-envelope admission fails closed without exact or bounded evidence`,
67
75
  "PLURNK_PROMPT_COUNT_ESTIMATE",
68
76
  );
69
77
 
@@ -87,31 +95,30 @@ export const providerFromSdkModel = ({
87
95
  const rates = catalogCost === undefined ? null : {
88
96
  input: catalogCost.inputPer1M,
89
97
  output: catalogCost.outputPer1M,
90
- cached: catalogCost.cacheReadPer1M ?? catalogCost.inputPer1M,
98
+ ...(catalogCost.cacheReadPer1M === undefined
99
+ ? {}
100
+ : { cacheRead: catalogCost.cacheReadPer1M }),
101
+ ...(catalogCost.cacheWritePer1M === undefined
102
+ ? {}
103
+ : { cacheWrite: catalogCost.cacheWritePer1M }),
91
104
  };
92
- const calculateCost = rates === null
93
- ? undefined
94
- : (usage: ProviderUsage): number => calculateCostUsd(usage, rates);
95
- const calculateCharge: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }> = rates === null
96
- ? () => ({ kind: "unknown", reason: "the response reported no cost and Models.dev has no rate for this model" })
97
- : rates.input === 0 && rates.output === 0 && rates.cached === 0
98
- ? () => ({ kind: "free", source: "Models.dev catalog rates" })
99
- : (usage: ProviderUsage) => ({
100
- kind: "estimated",
101
- usd: calculateCostUsdDecimal(usage, rates),
102
- source: "Models.dev catalog rates",
103
- });
105
+ const estimateCost = (usage: Parameters<typeof estimateProviderCost>[0]) =>
106
+ estimateProviderCost(usage, rates, "Models.dev catalog rates");
107
+ const affinityEnabled = cacheAffinityFromEnv(env, name);
108
+ const cacheWritePolicy = cacheWritePolicyFromEnv(env, name);
104
109
 
105
110
  return new AiSdkProvider({
106
111
  model,
107
112
  ...(attributions === undefined ? {} : { attributions }),
108
113
  ...(languageModel === undefined ? {} : { languageModel }),
109
- ...(normalizeCharge === undefined ? {} : { normalizeCharge }),
114
+ ...(normalizeCost === undefined ? {} : { normalizeCost }),
110
115
  ...(url === undefined ? {} : { url }),
111
116
  ...(headers === undefined ? {} : { headers: { ...headers } }),
112
117
  contextWindow,
113
- fetchTimeoutMs: parseRequiredInt(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", name),
114
- streamIdleTimeoutMs: parseRequiredInt(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", name),
118
+ fetchTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", name),
119
+ operationTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_OPERATION_TIMEOUT, "PLURNK_PROVIDERS_OPERATION_TIMEOUT", name),
120
+ firstContentTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT, "PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT", name),
121
+ streamIdleTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", name),
115
122
  reasoning,
116
123
  reasoningResponseStyle: reasoningResponseStyleFromEnv(env, name),
117
124
  temperature: parseRequiredFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", name, 0),
@@ -122,10 +129,15 @@ export const providerFromSdkModel = ({
122
129
  retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", name),
123
130
  errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", name),
124
131
  reasoningStyle: reasoningStyleFromEnv(env, name),
125
- promptCacheKey: url === undefined ? false : promptCacheKeyFromEnv(env, name),
132
+ ...(affinityEnabled && cacheAffinity !== undefined ? { cacheAffinity } : {}),
133
+ ...(cacheWritePolicy === "stable-system" && systemCacheProviderOptions !== undefined
134
+ ? { systemCacheProviderOptions }
135
+ : {}),
136
+ ...(reasoningResponseProviderOptions === undefined
137
+ ? {}
138
+ : { reasoningResponseProviderOptions }),
126
139
  serviceTier: env.PLURNK_PROVIDERS_SERVICE_TIER,
127
- calculateCost,
128
- calculateCharge,
140
+ estimateCost,
129
141
  source: providerSource(name),
130
142
  gbnfDebug: env.PLURNK_PROVIDERS_GBNF_DEBUG !== undefined
131
143
  && env.PLURNK_PROVIDERS_GBNF_DEBUG !== ""
@@ -165,9 +177,12 @@ export const catalogProviderFromEnv = (
165
177
  env,
166
178
  model: wireModel,
167
179
  languageModel: sdk.languageModel,
168
- normalizeCharge: sdk.normalizeCharge,
180
+ normalizeCost: sdk.normalizeCost,
169
181
  url: sdk.compatible?.url,
170
182
  headers: sdk.compatible?.headers,
183
+ cacheAffinity: sdk.cacheAffinity,
184
+ systemCacheProviderOptions: sdk.systemCacheProviderOptions,
185
+ reasoningResponseProviderOptions: sdk.reasoningResponseProviderOptions,
171
186
  contextWindow,
172
187
  info,
173
188
  });
@@ -5,6 +5,8 @@ import { compatibleProviderFromEnv } from "./compatibleProvider.ts";
5
5
  const env = {
6
6
  OPENAI_BASE_URL: "http://local.test/v1",
7
7
  PLURNK_PROVIDERS_FETCH_TIMEOUT: "1000",
8
+ PLURNK_PROVIDERS_OPERATION_TIMEOUT: "3000",
9
+ PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT: "1000",
8
10
  PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
9
11
  PLURNK_PROVIDERS_REASONING: "off",
10
12
  PLURNK_PROVIDERS_TEMPERATURE: "0.2",
@@ -16,12 +18,13 @@ const env = {
16
18
  PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT: "512",
17
19
  PLURNK_PROVIDERS_PROBE_ATTEMPTS: "1",
18
20
  PLURNK_PROVIDERS_PROBE_DELAY: "0",
19
- PLURNK_PROVIDERS_PROMPT_CACHE_KEY: "1",
21
+ PLURNK_PROVIDERS_CACHE_AFFINITY: "1",
22
+ PLURNK_PROVIDERS_CACHE_WRITE_POLICY: "stable-system",
20
23
  };
21
24
 
22
25
  test.afterEach(() => mock.restoreAll());
23
26
 
24
- test("compatible endpoints preserve configured prompt-cache affinity", async () => {
27
+ test("an undifferentiated compatible endpoint receives no guessed prompt-cache field", async () => {
25
28
  let body: Record<string, unknown> | undefined;
26
29
  mock.method(globalThis, "fetch", async (input: string | URL | Request, init?: RequestInit) => {
27
30
  if (String(input).endsWith("/models")) {
@@ -41,7 +44,7 @@ test("compatible endpoints preserve configured prompt-cache affinity", async ()
41
44
  messages: [{ role: "user", content: "hello" }],
42
45
  });
43
46
 
44
- assert.equal(body?.prompt_cache_key, "worker-affinity");
47
+ assert.equal("prompt_cache_key" in (body ?? {}), false);
45
48
  });
46
49
 
47
50
  test("the server-wide DRY-off floor emits no DRY request fields", async () => {
@@ -8,11 +8,14 @@ import {
8
8
  parseOptionalInt,
9
9
  parseRequiredFloat,
10
10
  parseRequiredInt,
11
- promptCacheKeyFromEnv,
11
+ parseTimeoutMs,
12
+ cacheAffinityFromEnv,
13
+ cacheWritePolicyFromEnv,
12
14
  reasoningFromEnv,
13
15
  reasoningResponseStyleFromEnv,
14
16
  } from "./env.ts";
15
17
  import { providerSource } from "./notices.ts";
18
+ import { plurnkCostNormalizer } from "./accounting.ts";
16
19
  import type { Provider } from "./types.ts";
17
20
  import { emitWarningOnce } from "./warnings.ts";
18
21
 
@@ -49,7 +52,10 @@ const probeModels = async (
49
52
  ): Promise<EndpointProbe> => {
50
53
  const modelsUrl = url.replace(/\/chat\/completions$/, "/models");
51
54
  try {
52
- const response = await fetch(modelsUrl, { headers, signal: AbortSignal.timeout(timeout) });
55
+ const response = await fetch(modelsUrl, {
56
+ headers,
57
+ ...(timeout > 0 ? { signal: AbortSignal.timeout(timeout) } : {}),
58
+ });
53
59
  if (!response.ok) return { nCtx: null, llamaServer: false, servedModel: null, failed: true };
54
60
  const data = await response.json() as {
55
61
  data?: Array<{ id?: string; n_ctx?: number; meta?: { n_ctx?: number } }>;
@@ -93,7 +99,7 @@ const probeProps = async (
93
99
  try {
94
100
  const response = await fetch(url.replace(/\/v1\/chat\/completions$/, "/props"), {
95
101
  headers,
96
- signal: AbortSignal.timeout(timeout),
102
+ ...(timeout > 0 ? { signal: AbortSignal.timeout(timeout) } : {}),
97
103
  });
98
104
  if (!response.ok) return { slotCount: null, eosText: null };
99
105
  const data = await response.json() as { total_slots?: number; eos_token?: string };
@@ -112,12 +118,17 @@ export const compatibleProviderFromEnv = async (
112
118
  model: string,
113
119
  baseUrlOverride?: string,
114
120
  ): Promise<Provider> => {
121
+ // The knobs remain universal and fail hard when malformed, but this local /
122
+ // first-party compatible route declares no vendor cache projection. llama-server
123
+ // already owns slot affinity and the first-party endpoint receives worker metadata.
124
+ cacheAffinityFromEnv(env, provider);
125
+ cacheWritePolicyFromEnv(env, provider);
115
126
  const url = chatUrl(provider, env, baseUrlOverride);
116
127
  const apiKey = provider === "openai" ? env.OPENAI_API_KEY : env.PLURNK_API_KEY;
117
128
  const headers: Record<string, string> = apiKey === undefined || apiKey.length === 0
118
129
  ? {}
119
130
  : { Authorization: `Bearer ${apiKey}` };
120
- const timeout = parseRequiredInt(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", provider);
131
+ const timeout = parseTimeoutMs(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", provider);
121
132
  const attempts = parseRequiredInt(env.PLURNK_PROVIDERS_PROBE_ATTEMPTS, "PLURNK_PROVIDERS_PROBE_ATTEMPTS", provider);
122
133
  const probe = await probeModelsRetrying(
123
134
  url,
@@ -165,7 +176,7 @@ export const compatibleProviderFromEnv = async (
165
176
 
166
177
  if (!llamaServer) {
167
178
  emitWarningOnce(
168
- `${provider} provider: physical prompt counting is a chars/2 estimate; over-policy recovery fails closed without exact or bounded request evidence`,
179
+ `${provider} provider: request-level prompt counting is a chars/2 estimate; hard context-envelope admission fails closed without exact or bounded evidence`,
169
180
  "PLURNK_PROMPT_COUNT_ESTIMATE",
170
181
  );
171
182
  }
@@ -176,7 +187,9 @@ export const compatibleProviderFromEnv = async (
176
187
  headers,
177
188
  contextWindow,
178
189
  fetchTimeoutMs: timeout,
179
- streamIdleTimeoutMs: parseRequiredInt(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", provider),
190
+ operationTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_OPERATION_TIMEOUT, "PLURNK_PROVIDERS_OPERATION_TIMEOUT", provider),
191
+ firstContentTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT, "PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT", provider),
192
+ streamIdleTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", provider),
180
193
  reasoning: reasoningFromEnv(env, provider),
181
194
  reasoningResponseStyle: reasoningResponseStyleFromEnv(env, provider),
182
195
  reasoningStyle,
@@ -192,7 +205,6 @@ export const compatibleProviderFromEnv = async (
192
205
  tuningFloors: provider !== "plurnk",
193
206
  retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", provider),
194
207
  errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", provider),
195
- promptCacheKey: promptCacheKeyFromEnv(env, provider),
196
208
  source: providerSource(provider),
197
209
  grammarStyle,
198
210
  gbnfDebug: env.PLURNK_PROVIDERS_GBNF_DEBUG !== undefined
@@ -200,6 +212,7 @@ export const compatibleProviderFromEnv = async (
200
212
  && env.PLURNK_PROVIDERS_GBNF_DEBUG !== "0",
201
213
  ...dataCaptureFromEnv(env, provider),
202
214
  firstPartyMetadata: provider === "plurnk",
215
+ normalizeCost: provider === "plurnk" ? plurnkCostNormalizer : undefined,
203
216
  apiKeyRejectedMessage: provider === "plurnk"
204
217
  ? "PLURNK_API_KEY was rejected by plurnk.ai (invalid or expired)."
205
218
  : undefined,