@plurnk/plurnk-providers 1.3.4 → 1.3.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. package/.env.defaults +41 -52
  2. package/README.md +44 -53
  3. package/SPEC.md +215 -354
  4. package/dist/AiSdkProvider.d.ts +78 -0
  5. package/dist/AiSdkProvider.d.ts.map +1 -0
  6. package/dist/AiSdkProvider.js +591 -0
  7. package/dist/AiSdkProvider.js.map +1 -0
  8. package/dist/Mock.d.ts +1 -1
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +1 -1
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/OpenAICompat.d.ts +3 -5
  13. package/dist/OpenAICompat.d.ts.map +1 -1
  14. package/dist/OpenAICompat.js +44 -133
  15. package/dist/OpenAICompat.js.map +1 -1
  16. package/dist/Pool.d.ts +1 -1
  17. package/dist/Pool.d.ts.map +1 -1
  18. package/dist/Pool.js +1 -1
  19. package/dist/Pool.js.map +1 -1
  20. package/dist/ProviderRegistry.d.ts.map +1 -1
  21. package/dist/ProviderRegistry.js +37 -24
  22. package/dist/ProviderRegistry.js.map +1 -1
  23. package/dist/aiSdkTransport.d.ts +52 -0
  24. package/dist/aiSdkTransport.d.ts.map +1 -0
  25. package/dist/aiSdkTransport.js +294 -0
  26. package/dist/aiSdkTransport.js.map +1 -0
  27. package/dist/catalogProvider.d.ts +15 -0
  28. package/dist/catalogProvider.d.ts.map +1 -0
  29. package/dist/catalogProvider.js +103 -0
  30. package/dist/catalogProvider.js.map +1 -0
  31. package/dist/compatibleProvider.d.ts +3 -0
  32. package/dist/compatibleProvider.d.ts.map +1 -0
  33. package/dist/compatibleProvider.js +146 -0
  34. package/dist/compatibleProvider.js.map +1 -0
  35. package/dist/discover.d.ts.map +1 -1
  36. package/dist/discover.js.map +1 -1
  37. package/dist/env.d.ts +1 -0
  38. package/dist/env.d.ts.map +1 -1
  39. package/dist/env.js +13 -6
  40. package/dist/env.js.map +1 -1
  41. package/dist/index.d.ts +5 -7
  42. package/dist/index.d.ts.map +1 -1
  43. package/dist/index.js +4 -8
  44. package/dist/index.js.map +1 -1
  45. package/dist/ollama.d.ts +3 -0
  46. package/dist/ollama.d.ts.map +1 -0
  47. package/dist/ollama.js +39 -0
  48. package/dist/ollama.js.map +1 -0
  49. package/dist/openai.d.ts +2 -4
  50. package/dist/openai.d.ts.map +1 -1
  51. package/dist/openai.js +1 -2
  52. package/dist/openai.js.map +1 -1
  53. package/dist/sdkModels.d.ts +13 -0
  54. package/dist/sdkModels.d.ts.map +1 -0
  55. package/dist/sdkModels.js +153 -0
  56. package/dist/sdkModels.js.map +1 -0
  57. package/dist/standardProviders.d.ts +0 -1
  58. package/dist/standardProviders.d.ts.map +1 -1
  59. package/dist/standardProviders.js +9 -11
  60. package/dist/standardProviders.js.map +1 -1
  61. package/dist/telemetry.d.ts.map +1 -1
  62. package/dist/telemetry.js +20 -9
  63. package/dist/telemetry.js.map +1 -1
  64. package/dist/types.d.ts +4 -3
  65. package/dist/types.d.ts.map +1 -1
  66. package/dist/usage.d.ts +1 -1
  67. package/dist/usage.d.ts.map +1 -1
  68. package/dist/usage.js +4 -2
  69. package/dist/usage.js.map +1 -1
  70. package/package.json +19 -8
  71. package/src/{OpenAICompat.test.ts → AiSdkProvider.test.ts} +202 -177
  72. package/src/{OpenAICompat.ts → AiSdkProvider.ts} +89 -144
  73. package/src/Mock.test.ts +3 -3
  74. package/src/Mock.ts +1 -1
  75. package/src/Pool.test.ts +3 -3
  76. package/src/Pool.ts +1 -1
  77. package/src/ProviderRegistry.test.ts +40 -27
  78. package/src/ProviderRegistry.ts +35 -24
  79. package/src/aiSdkTransport.test.ts +253 -0
  80. package/src/aiSdkTransport.ts +369 -0
  81. package/src/boundaries.test.ts +2 -2
  82. package/src/catalogProvider.test.ts +100 -0
  83. package/src/catalogProvider.ts +151 -0
  84. package/src/compatibleProvider.test.ts +44 -0
  85. package/src/compatibleProvider.ts +205 -0
  86. package/src/discover.test.ts +12 -12
  87. package/src/discover.ts +3 -6
  88. package/src/env.ts +14 -6
  89. package/src/index.ts +7 -11
  90. package/src/ollama.ts +63 -0
  91. package/src/openai.ts +2 -8
  92. package/src/sdkModels.test.ts +47 -0
  93. package/src/sdkModels.ts +194 -0
  94. package/src/telemetry.test.ts +17 -10
  95. package/src/telemetry.ts +22 -14
  96. package/src/types.ts +10 -13
  97. package/src/usage.test.ts +8 -10
  98. package/src/usage.ts +7 -3
  99. package/src/openaiStream.ts +0 -310
  100. package/src/standardProviders.test.ts +0 -949
  101. package/src/standardProviders.ts +0 -635
@@ -1,13 +1,37 @@
1
1
  import test, { mock } from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
- import OpenAICompatProvider, { effortFromBudget } from "./OpenAICompat.ts";
4
- import { OpenAiHttpError } from "./openaiStream.ts";
3
+ import AiSdkProvider, { effortFromBudget } from "./AiSdkProvider.ts";
5
4
  import { ProviderError } from "./telemetry.ts";
6
5
 
7
6
  // Build a fake fetch returning a one-chunk SSE stream, capturing the request
8
7
  // so tests can assert what the spine sent on the wire.
9
8
  const sseStream = (chunks: unknown[]) => {
10
- const lines = [...chunks.map((c) => `data: ${JSON.stringify(c)}`), "data: [DONE]"].join("\n\n");
9
+ const normalized = chunks.map((value, index) => {
10
+ const chunk = value as Record<string, any>;
11
+ const usage = chunk.usage !== undefined
12
+ ? {
13
+ ...chunk.usage,
14
+ ...(chunk.usage.cached_tokens !== undefined
15
+ ? {
16
+ prompt_tokens_details: {
17
+ cached_tokens: chunk.usage.cached_tokens,
18
+ },
19
+ }
20
+ : {}),
21
+ }
22
+ : undefined;
23
+ if (usage !== undefined) delete usage.cached_tokens;
24
+ return {
25
+ id: "test-completion",
26
+ object: "chat.completion.chunk",
27
+ created: index + 1,
28
+ model: "m",
29
+ ...chunk,
30
+ ...(chunk.choices === undefined && usage !== undefined ? { choices: [] } : {}),
31
+ ...(usage !== undefined ? { usage } : {}),
32
+ };
33
+ });
34
+ const lines = [...normalized.map((c) => `data: ${JSON.stringify(c)}`), "data: [DONE]"].join("\n\n");
11
35
  return new ReadableStream({
12
36
  start(controller) {
13
37
  controller.enqueue(new TextEncoder().encode(lines));
@@ -43,7 +67,6 @@ const injectedBase = {
43
67
  fetchTimeoutMs: 5000,
44
68
  temperature: 0.2,
45
69
  repeatPenalty: 1.15,
46
- retryDelayMs: 1,
47
70
  retryAttempts: 0,
48
71
  reasoning: { mode: "off" as const, budget: null },
49
72
  };
@@ -65,9 +88,9 @@ test("#608: per-instance fetch owns streaming and buffered requests", async () =
65
88
  }), { status: 200, headers: { "Content-Type": "application/json" } });
66
89
  };
67
90
 
68
- const streamed = await new OpenAICompatProvider({ ...injectedBase, fetch: streamingFetch, rawBody: true })
91
+ const streamed = await new AiSdkProvider({ ...injectedBase, fetch: streamingFetch, rawBody: true })
69
92
  .generate({ workerId: "stream", messages: [{ role: "user", content: "hello" }] });
70
- const buffered = await new OpenAICompatProvider({ ...injectedBase, fetch: bufferedFetch, streaming: false })
93
+ const buffered = await new AiSdkProvider({ ...injectedBase, fetch: bufferedFetch, streaming: false })
71
94
  .generate({ workerId: "buffer", messages: [{ role: "user", content: "hello" }] });
72
95
 
73
96
  assert.equal(streamed.assistant.content, "streamed");
@@ -83,18 +106,20 @@ test("#608: per-instance fetch owns streaming and buffered requests", async () =
83
106
  });
84
107
 
85
108
  test("#608: caller cancellation and provider timeout reach an injected fetch", async () => {
86
- const pendingFetch: typeof globalThis.fetch = async (_input, init) =>
87
- new Promise((_resolve, reject) => {
109
+ const pendingFetch: typeof globalThis.fetch = async (_input, init) => {
110
+ init?.signal?.throwIfAborted();
111
+ return new Promise((_resolve, reject) => {
88
112
  const signal = init?.signal;
89
113
  signal?.addEventListener("abort", () => reject(signal.reason), { once: true });
90
114
  });
115
+ };
91
116
  const caller = new AbortController();
92
- const callerProvider = new OpenAICompatProvider({ ...injectedBase, fetch: pendingFetch });
117
+ const callerProvider = new AiSdkProvider({ ...injectedBase, fetch: pendingFetch });
93
118
  const callerRequest = callerProvider.generate({ workerId: "cancel", messages: [], signal: caller.signal });
94
119
  caller.abort(new Error("operator cancelled"));
95
120
  await assert.rejects(callerRequest, /operator cancelled/);
96
121
 
97
- const timeoutProvider = new OpenAICompatProvider({ ...injectedBase, fetch: pendingFetch, fetchTimeoutMs: 1 });
122
+ const timeoutProvider = new AiSdkProvider({ ...injectedBase, fetch: pendingFetch, fetchTimeoutMs: 1 });
98
123
  await assert.rejects(
99
124
  timeoutProvider.generate({ workerId: "timeout", messages: [] }),
100
125
  (error: ProviderError) => error.kind === "network_failure",
@@ -116,7 +141,7 @@ test("#608: per-instance fetch owns tokenization and retry attempts", async () =
116
141
  { choices: [], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } },
117
142
  ]), { status: 200 });
118
143
  };
119
- const provider = new OpenAICompatProvider({
144
+ const provider = new AiSdkProvider({
120
145
  ...injectedBase,
121
146
  fetch: providerFetch,
122
147
  retryAttempts: 1,
@@ -135,7 +160,13 @@ test("#608: per-instance fetch owns tokenization and retry attempts", async () =
135
160
  // Sequenced fetch mock for retry tests: each entry is one HTTP response. A 200
136
161
  // streams its chunks; any other status returns that error (with an optional
137
162
  // retry-after header). The last entry repeats once the script runs out.
138
- type ScriptedResponse = { status: number; chunks?: unknown[]; retryAfter?: number | string; body?: string };
163
+ type ScriptedResponse = {
164
+ status: number;
165
+ chunks?: unknown[];
166
+ retryAfter?: number | string;
167
+ shouldRetry?: boolean;
168
+ body?: string;
169
+ };
139
170
  const installFetchScript = (responses: ScriptedResponse[]) => {
140
171
  const calls: { url: string; init: RequestInit }[] = [];
141
172
  let i = 0;
@@ -144,8 +175,15 @@ const installFetchScript = (responses: ScriptedResponse[]) => {
144
175
  const r = responses[Math.min(i, responses.length - 1)];
145
176
  i++;
146
177
  if (r.status === 200) return new Response(sseStream(r.chunks ?? []), { status: 200 });
147
- const headers = r.retryAfter !== undefined ? { "retry-after": String(r.retryAfter) } : {};
148
- return new Response(r.body ?? "err", { status: r.status, headers });
178
+ const headers = {
179
+ "content-type": "application/json",
180
+ ...(r.retryAfter !== undefined ? { "retry-after": String(r.retryAfter) } : {}),
181
+ ...(r.shouldRetry !== undefined ? { "x-should-retry": String(r.shouldRetry) } : {}),
182
+ };
183
+ return new Response(
184
+ r.body ?? JSON.stringify({ error: { message: `HTTP ${r.status}` } }),
185
+ { status: r.status, headers },
186
+ );
149
187
  });
150
188
  return calls;
151
189
  };
@@ -164,33 +202,25 @@ test("effortFromBudget: maps budget to tiers", () => {
164
202
  assert.equal(effortFromBudget(4001), "high");
165
203
  });
166
204
 
167
- test("#543: OpenAiHttpError distills a non-JSON (edge/CDN HTML) body and drops the OpenAI prefix", () => {
168
- const cf = new OpenAiHttpError(524, "<!DOCTYPE html><html><body>Error code 524</body></html>", null);
169
- assert.equal(cf.message, "524 origin timeout"); // distilled: no raw HTML, no "OpenAI" prefix
170
- assert.ok(cf.body.length > 20); // raw body retained on the field for forensics
171
- const api = new OpenAiHttpError(400, '{"error":{"message":"bad param"}}', null);
172
- assert.match(api.message, /^OpenAI 400 - \{/); // JSON API error passes through verbatim
173
- });
174
-
175
205
  test("#543: a 524 Cloudflare edge timeout fails fast - not retried despite retryAttempts", async () => {
176
206
  const calls = installFetchScript([{ status: 524, retryAfter: 120 }]);
177
- const p = new OpenAICompatProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
207
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
178
208
  await assert.rejects(p.generate({ workerId: "r", messages: [] }));
179
209
  await flush();
180
210
  assert.equal(calls.length, 1); // edge code: one attempt, no retry despite retryAttempts: 3
181
211
  mock.restoreAll();
182
212
  });
183
213
 
184
- test("#548: a 422 grammar_invalid is transient — retried on the budget, surfaces as grammar_invalid", async () => {
214
+ test("#548: a 422 grammar_invalid is a failed exchange, not transport replay policy", async () => {
185
215
  const body = JSON.stringify({ error: { message: "non-conforming emission rejected: ...", type: "grammar_invalid" } });
186
216
  const calls = installFetchScript([{ status: 422, body }]);
187
- const p = new OpenAICompatProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 2 });
217
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 2 });
188
218
  await assert.rejects(
189
219
  p.generate({ workerId: "r", messages: [] }),
190
220
  (e: unknown) => e instanceof ProviderError && e.kind === "grammar_invalid",
191
221
  );
192
222
  await flush();
193
- assert.equal(calls.length, 3); // initial + 2 retries: rode the bounded budget, unlike a terminal 422
223
+ assert.equal(calls.length, 1);
194
224
  mock.restoreAll();
195
225
  });
196
226
 
@@ -199,7 +229,7 @@ test("an SSE error frame is a failed exchange, not an empty completion", async (
199
229
  status: 422,
200
230
  error: { message: "non-conforming emission rejected", type: "grammar_invalid" },
201
231
  }]);
202
- const p = new OpenAICompatProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
232
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
203
233
  await assert.rejects(
204
234
  p.generate({ workerId: "r", messages: [] }),
205
235
  (e: unknown) => e instanceof ProviderError && e.kind === "grammar_invalid",
@@ -209,46 +239,46 @@ test("an SSE error frame is a failed exchange, not an empty completion", async (
209
239
 
210
240
  test("#539: a trailing eos_token (--special EOG leak) is stripped from content", async () => {
211
241
  installFetchJson({ model: "m", choices: [{ message: { content: "the answer<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
212
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
242
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
213
243
  const res = await p.generate({ workerId: "r", messages: [] });
214
244
  assert.equal(res.assistant.content, "the answer"); // trailing <eos> gone; packet + verdict see clean bytes
215
245
  });
216
246
 
217
247
  test("#539: without a probed eos_token the content passes through untouched", async () => {
218
248
  installFetchJson({ model: "m", choices: [{ message: { content: "keeps <eos> literally" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
219
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
249
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
220
250
  const res = await p.generate({ workerId: "r", messages: [] });
221
251
  assert.equal(res.assistant.content, "keeps <eos> literally"); // no eosText (a cloud backend) -> no strip
222
252
  });
223
253
 
224
254
  test("#539: only the TRAILING eos_token is stripped; a quoted one mid-body survives", async () => {
225
255
  installFetchJson({ model: "m", choices: [{ message: { content: "quotes <eos> in the body<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 } });
226
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
256
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
227
257
  const res = await p.generate({ workerId: "r", messages: [] });
228
258
  assert.equal(res.assistant.content, "quotes <eos> in the body"); // only the tail goes
229
259
  });
230
260
 
231
261
  test("identity getters and defaults", () => {
232
- const p = new OpenAICompatProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
262
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
233
263
  assert.equal(p.model, "m");
234
264
  assert.equal(p.contextWindow, null); // default
235
265
  assert.equal(p.countTokens(""), 0);
236
266
  assert.equal(p.countTokens("four"), 2); // default heuristic ceil(4/2) upper bound
237
- assert.equal(p.costFor({ prompt: 9, completion: 9, reasoning: 0, cached: 0, total: 18 }), 0); // default free
267
+ assert.equal(p.calculateCost({ prompt: 9, completion: 9, reasoning: 0, cached: 0, total: 18 }), 0); // default free
238
268
  });
239
269
 
240
- test("injected countTokens and costFor are used", () => {
241
- const p = new OpenAICompatProvider({
242
- model: "m", url: "http://x", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
270
+ test("injected countTokens and calculateCost are used", () => {
271
+ const p = new AiSdkProvider({
272
+ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
243
273
  countTokens: (t) => t.length,
244
- costFor: (u) => u.total * 2,
274
+ calculateCost: (u) => u.total * 2,
245
275
  });
246
276
  assert.equal(p.countTokens("abc"), 3);
247
- assert.equal(p.costFor({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 5 }), 10);
277
+ assert.equal(p.calculateCost({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 5 }), 10);
248
278
  });
249
279
 
250
280
  test("generate maps a streamed response into ProviderResponse", async () => {
251
- const p = new OpenAICompatProvider({ model: "req-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
281
+ const p = new AiSdkProvider({ model: "req-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
252
282
  installFetch([
253
283
  { model: "wire-model", choices: [{ delta: { content: "hel" } }] },
254
284
  { choices: [{ delta: { content: "lo" }, finish_reason: "stop" }] },
@@ -263,31 +293,42 @@ test("generate maps a streamed response into ProviderResponse", async () => {
263
293
  assert.notEqual(assistantRaw, undefined);
264
294
  });
265
295
 
266
- test("generate normalizes an out-of-set finish_reason to null", async () => {
267
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
296
+ test("generate surfaces and normalizes an out-of-set finish_reason", async () => {
297
+ const warnings: Array<{ message: string; code?: string }> = [];
298
+ mock.method(process, "emitWarning", (message: string | Error, options?: string | { code?: string }) => {
299
+ warnings.push({
300
+ message: String(message),
301
+ ...(typeof options === "object" && options.code !== undefined ? { code: options.code } : {}),
302
+ });
303
+ });
304
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
268
305
  installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "function_call" }] }]);
269
306
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
270
307
  assert.equal(assistant.finishReason, null);
308
+ assert.deepEqual(warnings, [{
309
+ message: 'unrecognized finish_reason "function_call"; treated as no-signal (finishReason=null). If it denotes a token-cap hit, core\'s length-cap detection will miss it.',
310
+ code: "PLURNK_FINISH_REASON_UNKNOWN",
311
+ }]);
271
312
  });
272
313
 
273
314
  test("generate translates a backend cap synonym to canonical length (#425)", async () => {
274
315
  // gemini shouts MAX_TOKENS, anthropic says max_tokens -- both must reach core as
275
316
  // "length" so its truncation check (=== "length") is a cross-backend invariant.
276
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
317
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
277
318
  installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "MAX_TOKENS" }] }]);
278
319
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
279
320
  assert.equal(assistant.finishReason, "length");
280
321
  });
281
322
 
282
323
  test("generate translates end_turn to canonical stop (#425)", async () => {
283
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
324
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
284
325
  installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "end_turn" }] }]);
285
326
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
286
327
  assert.equal(assistant.finishReason, "stop");
287
328
  });
288
329
 
289
330
  test("generate aggregates reasoning deltas under multiple field names", async () => {
290
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
331
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
291
332
  installFetch([{ choices: [{ delta: { reasoning_content: "be", thinking: "cause" } }] }]);
292
333
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
293
334
  assert.equal(assistant.reasoning, "because");
@@ -303,7 +344,7 @@ test("#482 sealed relay reasoning (non-streamed): encrypted reasoning_details su
303
344
  { type: "reasoning.text", text: "never surfaced here" },
304
345
  ],
305
346
  }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
306
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
347
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
307
348
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
308
349
  // item shape: wire `id` preserved, subtype from position (#482 widening)
309
350
  assert.deepEqual(assistant.reasoningEncrypted, [{ id: "rs_1", subtype: "message", encrypted: [{ data: "gAAAAABqBLOB", format: "openai-responses-v1" }] }]);
@@ -316,14 +357,14 @@ test("#482 widening: distinct wire ids stay distinct items (a single-object shap
316
357
  { type: "reasoning.encrypted", data: "AAA", format: "openai-responses-v1", id: "rs_1" },
317
358
  { type: "reasoning.encrypted", data: "BBB", format: "openai-responses-v1", id: "rs_2" },
318
359
  ] }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
319
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
360
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
320
361
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
321
362
  assert.equal(assistant.reasoningEncrypted?.length, 2);
322
363
  assert.deepEqual(assistant.reasoningEncrypted?.map((i) => i.id), ["rs_1", "rs_2"]);
323
364
  });
324
365
 
325
366
  test("#482 sealed relay reasoning (streamed): chunked blob concatenates per entry index", async () => {
326
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
367
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
327
368
  installFetch([
328
369
  { choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "gAAAA", format: "openai-responses-v1", id: "rs_1", index: 0 }] } }] },
329
370
  { choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "BqXYZ", id: "rs_1", index: 0 }] } }] },
@@ -335,20 +376,20 @@ test("#482 sealed relay reasoning (streamed): chunked blob concatenates per entr
335
376
  });
336
377
 
337
378
  test("reasoningStyle 'think' gates on budget != 0 (magnitude irrelevant for native)", async () => {
338
- const on = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
379
+ const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
339
380
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
340
381
  await on.generate({ workerId: "r", messages: [] });
341
382
  assert.equal(JSON.parse(calls[0].init.body as string).think, true);
342
383
 
343
384
  mock.restoreAll();
344
- const off = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
385
+ const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
345
386
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
346
387
  await off.generate({ workerId: "r", messages: [] });
347
388
  assert.equal("think" in JSON.parse(calls[0].init.body as string), false);
348
389
  });
349
390
 
350
391
  test("reasoningStyle 'effort' sends a reasoning_effort tier from the budget", async () => {
351
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "on", budget: 5000 }, retryAttempts: 0, reasoningStyle: "effort" });
392
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: 5000 }, retryAttempts: 0, reasoningStyle: "effort" });
352
393
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
353
394
  await p.generate({ workerId: "r", messages: [] });
354
395
  assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, "high");
@@ -359,7 +400,7 @@ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS (#403 —
359
400
  // 400s reasoning_effort='adaptive' for non-MiniMax models (wire-verified,
360
401
  // #403): adaptive = the backend's own default posture = omission.
361
402
  for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "on", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "on"; budget: number | null }, string | null]>) {
362
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
403
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
363
404
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
364
405
  await p.generate({ workerId: "r", messages: [] });
365
406
  const body = JSON.parse(calls[0].init.body as string);
@@ -370,7 +411,7 @@ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS (#403 —
370
411
  });
371
412
 
372
413
  test("the family temperature default rides every request; caller sampling overrides it (#30)", async () => {
373
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
414
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
374
415
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
375
416
  await p.generate({ workerId: "r", messages: [] });
376
417
  assert.equal(JSON.parse(calls[0].init.body as string).temperature, 0.2);
@@ -387,9 +428,9 @@ test("the family temperature default rides every request; caller sampling overri
387
428
  });
388
429
 
389
430
  test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box default; never on cloud", async () => {
390
- const base = { model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off" as const, budget: null }, retryAttempts: 0 };
431
+ const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off" as const, budget: null }, retryAttempts: 0 };
391
432
  // set + llamacpp -> the loop-breakers ride the wire
392
- const p = new OpenAICompatProvider({ ...base, grammarStyle: "llamacpp", dryMultiplier: 0.8, dryBase: 1.75, dryAllowedLength: 2, repeatLastN: 512 });
433
+ const p = new AiSdkProvider({ ...base, grammarStyle: "llamacpp", dryMultiplier: 0.8, dryBase: 1.75, dryAllowedLength: 2, repeatLastN: 512 });
393
434
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
394
435
  await p.generate({ workerId: "r", messages: [] });
395
436
  let body = JSON.parse(calls[0].init.body as string);
@@ -400,7 +441,7 @@ test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves th
400
441
  assert.equal(body.repeat_penalty, 1.15); // repeat_penalty always rides the llamacpp path
401
442
  mock.restoreAll();
402
443
  // unset -> no dry_*/repeat_last_n on the wire (box keeps its own defaults)
403
- const p2 = new OpenAICompatProvider({ ...base, grammarStyle: "llamacpp" });
444
+ const p2 = new AiSdkProvider({ ...base, grammarStyle: "llamacpp" });
404
445
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
405
446
  await p2.generate({ workerId: "r", messages: [] });
406
447
  body = JSON.parse(calls[0].init.body as string);
@@ -408,7 +449,7 @@ test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves th
408
449
  assert.equal("repeat_last_n" in body, false);
409
450
  mock.restoreAll();
410
451
  // DRY is a llama.cpp sampler: a cloud ("none") provider never emits it, even if configured
411
- const p3 = new OpenAICompatProvider({ ...base, grammarStyle: "none", dryMultiplier: 0.8, repeatLastN: 512 });
452
+ const p3 = new AiSdkProvider({ ...base, grammarStyle: "none", dryMultiplier: 0.8, repeatLastN: 512 });
412
453
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
413
454
  await p3.generate({ workerId: "r", messages: [] });
414
455
  body = JSON.parse(calls[0].init.body as string);
@@ -418,7 +459,7 @@ test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves th
418
459
  });
419
460
 
420
461
  test("llamacpp grammar path: temperature default + the managed repeat-penalty floor", async () => {
421
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
462
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
422
463
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
423
464
  await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
424
465
  const body = JSON.parse(calls[0].init.body as string);
@@ -428,13 +469,13 @@ test("llamacpp grammar path: temperature default + the managed repeat-penalty fl
428
469
 
429
470
  test("#426: the repeat penalty rides EVERY request rail-off, keyed per backend (cloud degeneration guard)", async () => {
430
471
  // llama.cpp with NO grammar carries its key too (unconstrained local is guarded)
431
- const llama = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
472
+ const llama = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
432
473
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
433
474
  await llama.generate({ workerId: "r", messages: [] });
434
475
  assert.equal(JSON.parse(calls[0].init.body as string).repeat_penalty, 1.15);
435
476
  mock.restoreAll();
436
477
  // a `none`-style cloud backend WITH a frequency penalty gets frequency_penalty (OpenAI-standard, #426)
437
- const cloud = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
478
+ const cloud = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
438
479
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
439
480
  await cloud.generate({ workerId: "r", messages: [] });
440
481
  const cloudBody = JSON.parse(calls[0].init.body as string);
@@ -443,14 +484,14 @@ test("#426: the repeat penalty rides EVERY request rail-off, keyed per backend (
443
484
  assert.equal("repeat_penalty" in cloudBody, false);
444
485
  mock.restoreAll();
445
486
  // frequencyPenalty unset (default 0) opts out cleanly - sends nothing (an out-of-date plugin runs unguarded, never breaks)
446
- const bare = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
487
+ const bare = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
447
488
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
448
489
  await bare.generate({ workerId: "r", messages: [] });
449
490
  assert.equal("frequency_penalty" in JSON.parse(calls[0].init.body as string), false);
450
491
  });
451
492
 
452
493
  test("sampling passthrough forwards caller params; managed + reserved keys win", async () => {
453
- const p = new OpenAICompatProvider({ model: "managed-model", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
494
+ const p = new AiSdkProvider({ model: "managed-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
454
495
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
455
496
  await p.generate({
456
497
  workerId: "r",
@@ -473,7 +514,7 @@ test("sampling passthrough forwards caller params; managed + reserved keys win",
473
514
  });
474
515
 
475
516
  test("#477 sampling passthrough guards contract invariants: n/tools/caps stripped, platform knobs pass", async () => {
476
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
517
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
477
518
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
478
519
  await p.generate({
479
520
  workerId: "r",
@@ -500,7 +541,7 @@ test("#477 sampling passthrough guards contract invariants: n/tools/caps strippe
500
541
  test("#488 postmortem: intent maps IDENTICALLY under a transported grammar — sanctioned channel coexists with rails", async () => {
501
542
  // The brief rails-win-the-channel clamp is REVERTED: closing the channel starved a
502
543
  // reasoning-tuned model into escaping mid-content (unconstrained, discarded, billed).
503
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
544
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
504
545
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
505
546
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
506
547
  const body = JSON.parse(calls[0].init.body as string);
@@ -514,7 +555,7 @@ test("#488 postmortem: intent maps IDENTICALLY under a transported grammar — s
514
555
  test("#488 channel-escape detector: billed completion tokens vastly beyond visible channels attach grammar_unenforced", async () => {
515
556
  // The run105 shape: tiny visible content, no reasoning, thousands billed — the decode
516
557
  // escaped into a discarded reasoning block, unconstrained.
517
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
558
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
518
559
  installFetch([
519
560
  { choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
520
561
  { usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
@@ -528,7 +569,7 @@ test("#488 channel-escape detector: billed completion tokens vastly beyond visib
528
569
  });
529
570
 
530
571
  test("#488 loud state absent on grammarless calls; no escape event without a transported grammar", async () => {
531
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
572
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
532
573
  installFetch([
533
574
  { choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
534
575
  { usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
@@ -539,33 +580,33 @@ test("#488 loud state absent on grammarless calls; no escape event without a tra
539
580
  });
540
581
 
541
582
  test("reasoningStyle 'template' always emits enable_thinking mirroring budget != 0 — explicit false, never omitted", async () => {
542
- const on = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
583
+ const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
543
584
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
544
585
  await on.generate({ workerId: "r", messages: [] });
545
586
  assert.deepEqual(JSON.parse(calls[0].init.body as string).chat_template_kwargs, { enable_thinking: true });
546
587
 
547
588
  mock.restoreAll();
548
- const off = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
589
+ const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
549
590
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
550
591
  await off.generate({ workerId: "r", messages: [] });
551
592
  assert.deepEqual(JSON.parse(calls[0].init.body as string).chat_template_kwargs, { enable_thinking: false });
552
593
  });
553
594
 
554
595
  test("budget 0 suppresses effort and include_reasoning", async () => {
555
- const effort = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "effort" });
596
+ const effort = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "effort" });
556
597
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
557
598
  await effort.generate({ workerId: "r", messages: [] });
558
599
  assert.equal("reasoning_effort" in JSON.parse(calls[0].init.body as string), false);
559
600
 
560
601
  mock.restoreAll();
561
- const relay = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
602
+ const relay = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
562
603
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
563
604
  await relay.generate({ workerId: "r", messages: [] });
564
605
  assert.equal("include_reasoning" in JSON.parse(calls[0].init.body as string), false);
565
606
  });
566
607
 
567
608
  test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", async () => {
568
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
609
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
569
610
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
570
611
  await p.generate({ workerId: "r", messages: [] });
571
612
  assert.equal(JSON.parse(calls[0].init.body as string).include_reasoning, true);
@@ -574,7 +615,7 @@ test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", asy
574
615
  // — grammar-constrained sampling (SPEC §13, issues #8/#9) —
575
616
 
576
617
  test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor", async () => {
577
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
618
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
578
619
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
579
620
  await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
580
621
  const body = JSON.parse(calls[0].init.body as string);
@@ -584,7 +625,7 @@ test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor
584
625
  });
585
626
 
586
627
  test("grammar transport 'none' (default): the grammar is never sent — no silent unconstrained", async () => {
587
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
628
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
588
629
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
589
630
  await p.generate({ workerId: "r", messages: [], grammar: "root ::= statement" });
590
631
  const body = JSON.parse(calls[0].init.body as string);
@@ -595,7 +636,7 @@ test("grammar transport 'none' (default): the grammar is never sent — no silen
595
636
  // — grammar conformance OBSERVATION (SPEC §10.14, §13): a completed exchange always
596
637
  // returns; bytes flow; a non-accept verdict rides response.telemetry —
597
638
 
598
- const grammarProvider = () => new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", source: "provider:test" });
639
+ const grammarProvider = () => new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", source: "provider:test" });
599
640
  const streamingContent = (content: string) => installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
600
641
 
601
642
  test("enforcement: conforming output passes through unchanged", async () => {
@@ -644,7 +685,7 @@ test("observation: empty content under a non-empty grammar returns with the verd
644
685
  });
645
686
 
646
687
  test("enforcement: when no grammar is sent (grammarStyle 'none'), output is NOT validated — no wire fields, no error (SPEC )", async () => {
647
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // grammarStyle defaults to "none"
688
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // grammarStyle defaults to "none"
648
689
  streamingContent("anything goes");
649
690
  const { assistant } = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' }); // grammar passed but never transported
650
691
  assert.equal(assistant.content, "anything goes"); // no enforcement check
@@ -666,7 +707,7 @@ test("enforcement: a grammar our validator can't parse is a NON-FATAL verify gap
666
707
  // — PLURNK_PROVIDERS_GBNF_DEBUG: run unconstrained, then verify the free output against the grammar —
667
708
 
668
709
  test("gbnfDebug: the grammar is NOT transported; conforming free output passes through with NO telemetry", async () => {
669
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
710
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
670
711
  const calls = installFetch([{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }]);
671
712
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
672
713
  const body = JSON.parse(calls[0].init.body as string);
@@ -677,7 +718,7 @@ test("gbnfDebug: the grammar is NOT transported; conforming free output passes t
677
718
  });
678
719
 
679
720
  test("gbnfDebug: a conflict does NOT throw — it returns the bytes plus a grammar_unenforced telemetry event with the divergence position (#24)", async () => {
680
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
721
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
681
722
  const calls = installFetch([{ choices: [{ delta: { reasoning_content: "let me think about ok", content: "xon-conforming output" }, finish_reason: "stop" }] }]);
682
723
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
683
724
  // The model's bytes survive — not discarded by a throw (the empty-turn cascade root cause).
@@ -695,7 +736,7 @@ test("gbnfDebug: a conflict does NOT throw — it returns the bytes plus a gramm
695
736
  });
696
737
 
697
738
  test("gbnfDebug: an INVALID grammar throws before any wire call — it never reaches the model", async () => {
698
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true });
739
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true });
699
740
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
700
741
  await assert.rejects(
701
742
  () => p.generate({ workerId: "r", messages: [], grammar: 'foo ::= "a"' }), // no `root` rule → invalid GBNF
@@ -704,31 +745,15 @@ test("gbnfDebug: an INVALID grammar throws before any wire call — it never rea
704
745
  assert.equal(calls.length, 0); // fail-hard before the fetch — grammar never transported
705
746
  });
706
747
 
707
- // — meta bag: pass-through extras + validated known keys (#23) —
748
+ // — meta bag: verbatim provider metadata (#23) —
708
749
 
709
- test("meta: the spec's balance field is normalized to a validated meta.balancePico", async () => {
710
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, balanceMetaKey: "balance_pico" });
711
- installFetchJson({ ...jsonChoice, balance_pico: 4_200_000 });
750
+ test("meta: passes backend fields through without reinterpreting monetary values", async () => {
751
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
752
+ const balance = { amount: "0.0000042", currency: "XMR" };
753
+ installFetchJson({ ...jsonChoice, balance, system_fingerprint: "fp_abc" });
712
754
  const res = await p.generate({ workerId: "r", messages: [] });
713
- assert.equal(res.meta?.balancePico, 4_200_000);
714
- assert.equal("balance_pico" in (res.meta ?? {}), false); // raw key renamed to the canonical balancePico
715
- });
716
-
717
- test("meta: passes the backend's extra top-level fields through verbatim (every provider)", async () => {
718
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false }); // no balanceMetaKey
719
- installFetchJson({ ...jsonChoice, balance_pico: 4_200_000, system_fingerprint: "fp_abc" });
720
- const res = await p.generate({ workerId: "r", messages: [] });
721
- assert.equal(res.meta?.balance_pico, 4_200_000); // passed through raw — no balance contract on this provider
755
+ assert.deepEqual(res.meta?.balance, balance);
722
756
  assert.equal(res.meta?.system_fingerprint, "fp_abc");
723
- assert.equal("balancePico" in (res.meta ?? {}), false); // not normalized without the key
724
- });
725
-
726
- test("meta: a non-numeric balance is dropped, never surfaced as balancePico (null-honest)", async () => {
727
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, balanceMetaKey: "balance_pico" });
728
- installFetchJson({ ...jsonChoice, balance_pico: "lots" });
729
- const res = await p.generate({ workerId: "r", messages: [] });
730
- assert.equal("balancePico" in (res.meta ?? {}), false);
731
- assert.equal("balance_pico" in (res.meta ?? {}), false); // raw dropped too — the known key is validated away
732
757
  });
733
758
 
734
759
  // — first-party telemetry headers (attribution + client, SPEC §5) —
@@ -737,7 +762,7 @@ const headerVal = (init: RequestInit, name: string): string | undefined =>
737
762
  new Headers(init.headers).get(name) ?? undefined;
738
763
 
739
764
  test("firstPartyMetadata: attributions + client ride as Plurnk-* headers", async () => {
740
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
765
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
741
766
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
742
767
  await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0", "@foo/y@0.3.1"], client: "plurnk.nvim/1.4.0" });
743
768
  assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), '["@acme/x@1.2.0","@foo/y@0.3.1"]');
@@ -745,7 +770,7 @@ test("firstPartyMetadata: attributions + client ride as Plurnk-* headers", async
745
770
  });
746
771
 
747
772
  test("#522 Plurnk-Worker-Primary: the lineage root rides under the gate; emitted even when it equals workerId", async () => {
748
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
773
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
749
774
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
750
775
  await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
751
776
  assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), "w-root"); // a descendant: Primary != Worker-Id
@@ -764,14 +789,14 @@ test("#522 Plurnk-Worker-Primary: the lineage root rides under the gate; emitted
764
789
  });
765
790
 
766
791
  test("#522 Plurnk-Worker-Primary is structurally dropped when firstPartyMetadata is off", async () => {
767
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
792
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
768
793
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
769
794
  await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
770
795
  assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), undefined); // never reaches a third-party backend
771
796
  });
772
797
 
773
798
  test("firstPartyMetadata off (default): the headers are structurally dropped even when values are passed", async () => {
774
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
799
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
775
800
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
776
801
  await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0"], client: "plurnk-cli/2.0.0" });
777
802
  assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), undefined); // never leaks to a non-first-party backend
@@ -779,7 +804,7 @@ test("firstPartyMetadata off (default): the headers are structurally dropped eve
779
804
  });
780
805
 
781
806
  test("firstPartyMetadata on but empty values: no header emitted", async () => {
782
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
807
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
783
808
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
784
809
  await p.generate({ workerId: "r", messages: [], attributions: [], client: "" });
785
810
  assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), undefined);
@@ -787,7 +812,7 @@ test("firstPartyMetadata on but empty values: no header emitted", async () => {
787
812
  });
788
813
 
789
814
  test("grammar transport: no grammar passed sends no grammar field, but the penalty rides (#426)", async () => {
790
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
815
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
791
816
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
792
817
  await p.generate({ workerId: "r", messages: [] });
793
818
  const body = JSON.parse(calls[0].init.body as string);
@@ -796,7 +821,7 @@ test("grammar transport: no grammar passed sends no grammar field, but the penal
796
821
  });
797
822
 
798
823
  test("maxTokens transports as max_tokens; absent → no wire field (server default)", async () => {
799
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
824
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
800
825
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
801
826
  await p.generate({ workerId: "r", messages: [], maxTokens: 2048 });
802
827
  assert.equal(JSON.parse(calls[0].init.body as string).max_tokens, 2048);
@@ -808,7 +833,7 @@ test("maxTokens transports as max_tokens; absent → no wire field (server defau
808
833
  });
809
834
 
810
835
  test("slot affinity is internal: sticky per workerId, distinct runs spread across slots (#11)", async () => {
811
- const pinning = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
836
+ const pinning = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
812
837
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
813
838
  await pinning.generate({ workerId: "run-A", messages: [] });
814
839
  await pinning.generate({ workerId: "run-B", messages: [] });
@@ -819,20 +844,20 @@ test("slot affinity is internal: sticky per workerId, distinct runs spread acros
819
844
  });
820
845
 
821
846
  test("slot affinity: no pinning backend or unknown slotCount → no id_slot ever", async () => {
822
- const cloud = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // default: no pinning
847
+ const cloud = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // default: no pinning
823
848
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
824
849
  await cloud.generate({ workerId: "run-A", messages: [] });
825
850
  assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
826
851
 
827
852
  mock.restoreAll();
828
- const noCount = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true }); // slotCount null
853
+ const noCount = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true }); // slotCount null
829
854
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
830
855
  await noCount.generate({ workerId: "run-A", messages: [] });
831
856
  assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
832
857
  });
833
858
 
834
859
  test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; recent runs stay sticky (#11)", async () => {
835
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
860
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
836
861
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
837
862
  const slotOf = (i: number) => JSON.parse(calls[i].init.body as string).id_slot;
838
863
  for (let i = 0; i < 16; i++) await p.generate({ workerId: `r${i}`, messages: [] }); // fills the 16-entry window {r0..r15}
@@ -846,7 +871,7 @@ test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; r
846
871
 
847
872
  test("streaming:false: a non-ok response rejects as a classified ProviderError (covers the non-streamed transport)", async () => {
848
873
  const { ProviderError } = await import("./telemetry.ts");
849
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, source: "provider:test" });
874
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, source: "provider:test" });
850
875
  mock.method(globalThis, "fetch", async () => new Response("boom", { status: 500 }));
851
876
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
852
877
  assert.ok(err instanceof ProviderError);
@@ -857,14 +882,14 @@ test("streaming:false: a non-ok response rejects as a classified ProviderError (
857
882
  });
858
883
 
859
884
  test("generate fail-hards on a missing or empty workerId", async () => {
860
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
885
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
861
886
  installFetch([{ choices: [{ delta: { content: "x" } }] }]);
862
887
  await assert.rejects(() => p.generate({ workerId: "", messages: [] }), /workerId is required/);
863
888
  await assert.rejects(() => (p.generate as (a: object) => Promise<unknown>)({ messages: [] }), /workerId is required/);
864
889
  });
865
890
 
866
891
  test("messages pass through verbatim — the provider injects no turn (PLAN lives in the grammar, never a provider prefill)", async () => {
867
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
892
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
868
893
  const calls = installFetch([{ choices: [{ delta: { content: "out" } }] }]);
869
894
  const input = [{ role: "user" as const, content: "hi" }];
870
895
  const res = await p.generate({ workerId: "r", messages: input });
@@ -874,7 +899,7 @@ test("messages pass through verbatim — the provider injects no turn (PLAN live
874
899
 
875
900
  test("generate wraps an HTTP failure as a ProviderError carrying a TelemetryEvent", async () => {
876
901
  const { ProviderError } = await import("./telemetry.ts");
877
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, source: "provider:test" });
902
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, source: "provider:test" });
878
903
  mock.method(globalThis, "fetch", async () => new Response("rate limited", { status: 429 }));
879
904
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
880
905
  assert.ok(err instanceof ProviderError);
@@ -886,27 +911,28 @@ test("generate wraps an HTTP failure as a ProviderError carrying a TelemetryEven
886
911
  });
887
912
 
888
913
  test("generate rejects on a pre-aborted external signal", async () => {
889
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
914
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
890
915
  installFetch([{ choices: [{ delta: { content: "x" } }] }]);
891
916
  const signal = AbortSignal.abort(new Error("nope"));
892
917
  await assert.rejects(() => p.generate({ workerId: "r", messages: [], signal }));
893
918
  });
894
919
 
895
920
  test("configured headers and url are sent verbatim", async () => {
896
- const p = new OpenAICompatProvider({
897
- model: "m", url: "http://host/custom/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
921
+ const p = new AiSdkProvider({
922
+ model: "m", url: "http://host/custom/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
898
923
  headers: { Authorization: "Bearer secret", "X-Title": "plurnk" },
899
924
  });
900
925
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
901
926
  await p.generate({ workerId: "r", messages: [] });
902
927
  assert.equal(calls[0].url, "http://host/custom/chat/completions");
903
- assert.equal((calls[0].init.headers as Record<string, string>).Authorization, "Bearer secret");
904
- assert.equal((calls[0].init.headers as Record<string, string>)["X-Title"], "plurnk");
928
+ const headers = new Headers(calls[0].init.headers);
929
+ assert.equal(headers.get("authorization"), "Bearer secret");
930
+ assert.equal(headers.get("x-title"), "plurnk");
905
931
  });
906
932
 
907
933
  // — transient-failure retry (#18) —
908
934
 
909
- const retryCfg = { model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null } as const };
935
+ const retryCfg = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const };
910
936
 
911
937
  test("retry: a transient failure retries and a later success resolves", async () => {
912
938
  const calls = installFetchScript([
@@ -914,51 +940,52 @@ test("retry: a transient failure retries and a later success resolves", async ()
914
940
  { status: 503, retryAfter: 0 },
915
941
  { status: 200, chunks: [{ choices: [{ delta: { content: "ok" } }] }] },
916
942
  ]);
917
- const p = new OpenAICompatProvider({ ...retryCfg, retryAttempts: 3 });
943
+ const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 3 });
918
944
  const res = await p.generate({ workerId: "r", messages: [] });
919
945
  assert.equal(res.assistant.content, "ok");
920
946
  assert.equal(calls.length, 3); // 429 → 503 → 200
921
947
  });
922
948
 
923
- test("#559: streamed-body silence retries and records the recovery in durable response metadata", async () => {
949
+ test("#559: streamed-body silence fails the exchange without replaying partial output", async () => {
924
950
  let calls = 0;
925
951
  mock.method(globalThis, "fetch", async () => {
926
952
  calls++;
927
953
  if (calls === 1) {
928
954
  return new Response(new ReadableStream({
929
955
  start(controller) {
930
- controller.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"partial"}}]}\n\n'));
956
+ controller.enqueue(new TextEncoder().encode(
957
+ 'data: {"id":"first","object":"chat.completion.chunk","created":1,"model":"m","choices":[{"index":0,"delta":{"content":"partial"},"finish_reason":null}]}\n\n',
958
+ ));
959
+ setTimeout(() => controller.close(), 100);
931
960
  },
932
961
  }), { status: 200 });
933
962
  }
934
963
  return new Response(new ReadableStream({
935
964
  start(controller) {
936
- controller.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"recovered"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n'));
965
+ controller.enqueue(new TextEncoder().encode(
966
+ 'data: {"id":"second","object":"chat.completion.chunk","created":2,"model":"m","choices":[{"index":0,"delta":{"content":"recovered"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n',
967
+ ));
937
968
  controller.close();
938
969
  },
939
970
  }), { status: 200 });
940
971
  });
941
- const p = new OpenAICompatProvider({
972
+ const p = new AiSdkProvider({
942
973
  model: "m",
943
- url: "http://x",
974
+ url: "http://x/v1/chat/completions",
944
975
  fetchTimeoutMs: 1000,
945
976
  streamIdleTimeoutMs: 10,
946
977
  temperature: 0.2,
947
978
  repeatPenalty: 1.15,
948
- retryDelayMs: 1,
949
979
  reasoning: { mode: "off", budget: null },
950
980
  retryAttempts: 1,
951
981
  source: "provider:test",
952
982
  });
953
- const result = await p.generate({ workerId: "r", messages: [] });
954
- assert.equal(result.assistant.content, "recovered");
955
- assert.equal(calls, 2);
956
- const retries = result.meta?.transportRetries as Array<Record<string, unknown>>;
957
- assert.equal(retries.length, 1);
958
- assert.equal(retries[0].attempt, 1);
959
- assert.equal(retries[0].kind, "network_failure");
960
- assert.equal(typeof retries[0].elapsedMs, "number");
961
- assert.match(String(retries[0].message), /no body bytes for 10ms/);
983
+ await assert.rejects(
984
+ p.generate({ workerId: "r", messages: [] }),
985
+ (error: ProviderError) => error.kind === "network_failure"
986
+ && /chunk timeout/i.test(error.message),
987
+ );
988
+ assert.equal(calls, 1);
962
989
  mock.restoreAll();
963
990
  });
964
991
 
@@ -971,27 +998,25 @@ test("#559: a zero stream-idle timeout permits a slow inter-chunk pause", async
971
998
  controller.close();
972
999
  },
973
1000
  }), { status: 200 }));
974
- const p = new OpenAICompatProvider({
1001
+ const p = new AiSdkProvider({
975
1002
  model: "m",
976
- url: "http://x",
1003
+ url: "http://x/v1/chat/completions",
977
1004
  fetchTimeoutMs: 1000,
978
1005
  streamIdleTimeoutMs: 0,
979
1006
  temperature: 0.2,
980
1007
  repeatPenalty: 1.15,
981
- retryDelayMs: 1,
982
1008
  reasoning: { mode: "off", budget: null },
983
1009
  retryAttempts: 0,
984
1010
  });
985
1011
  const result = await p.generate({ workerId: "r", messages: [] });
986
1012
  assert.equal(result.assistant.content, "slow is valid");
987
- assert.equal(result.meta?.transportRetries, undefined);
988
1013
  mock.restoreAll();
989
1014
  });
990
1015
 
991
1016
  test("retry: exhausting the budget surfaces the classified ProviderError", async () => {
992
1017
  const { ProviderError } = await import("./telemetry.ts");
993
1018
  const calls = installFetchScript([{ status: 429, retryAfter: 0 }]); // always rate-limited
994
- const p = new OpenAICompatProvider({ ...retryCfg, retryAttempts: 2 });
1019
+ const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 2 });
995
1020
  await assert.rejects(
996
1021
  () => p.generate({ workerId: "r", messages: [] }),
997
1022
  (err: unknown) => { assert.ok(err instanceof ProviderError); assert.equal(err.kind, "rate_limit"); return true; },
@@ -1004,7 +1029,7 @@ test("retry: a Retry-After HTTP-date is honored — a past date parses to a 0ms
1004
1029
  { status: 503, retryAfter: "Wed, 21 Oct 2015 07:28:00 GMT" }, // date form, in the past → max(0, past−now) = 0
1005
1030
  { status: 200, chunks: [{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }] },
1006
1031
  ]);
1007
- const p = new OpenAICompatProvider({ ...retryCfg, retryAttempts: 1 });
1032
+ const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 1 });
1008
1033
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
1009
1034
  assert.equal(assistant.content, "ok");
1010
1035
  assert.equal(calls.length, 2); // initial 503 + one retry, no real wall-clock wait
@@ -1012,14 +1037,14 @@ test("retry: a Retry-After HTTP-date is honored — a past date parses to a 0ms
1012
1037
 
1013
1038
  test("retry: a terminal error (401 unauthorized) is never retried", async () => {
1014
1039
  const calls = installFetchScript([{ status: 401 }]);
1015
- const p = new OpenAICompatProvider({ ...retryCfg, retryAttempts: 5 });
1040
+ const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 5 });
1016
1041
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), /401/);
1017
1042
  assert.equal(calls.length, 1); // terminal — no retry despite budget
1018
1043
  });
1019
1044
 
1020
1045
  test("retry: retryAttempts 0 surfaces the first transient failure immediately", async () => {
1021
1046
  const calls = installFetchScript([{ status: 503, retryAfter: 0 }]);
1022
- const p = new OpenAICompatProvider({ ...retryCfg, retryAttempts: 0 });
1047
+ const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 0 });
1023
1048
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }));
1024
1049
  assert.equal(calls.length, 1); // no retry budget
1025
1050
  });
@@ -1027,7 +1052,7 @@ test("retry: retryAttempts 0 surfaces the first transient failure immediately",
1027
1052
  test("retry: a caller abort during backoff rejects promptly with no further attempt (mid-flight abort, SPEC )", async () => {
1028
1053
  const ac = new AbortController();
1029
1054
  const calls = installFetchScript([{ status: 503, retryAfter: 5 }]); // 5s backoff we never wait out
1030
- const p = new OpenAICompatProvider({ ...retryCfg, retryAttempts: 3 });
1055
+ const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 3 });
1031
1056
  const promise = p.generate({ workerId: "r", messages: [], signal: ac.signal });
1032
1057
  await flush(); // attempt 0 fails, enters the backoff sleep
1033
1058
  assert.equal(calls.length, 1);
@@ -1040,21 +1065,21 @@ test("retry: a caller abort during backoff rejects promptly with no further atte
1040
1065
 
1041
1066
  test("reasoningStyle 'anthropic' maps the budget to the thinking param", async () => {
1042
1067
  // N>0 → enabled with budget_tokens
1043
- const capped = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "on", budget: 4096 }, reasoningStyle: "anthropic" });
1068
+ const capped = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: 4096 }, reasoningStyle: "anthropic" });
1044
1069
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1045
1070
  await capped.generate({ workerId: "r", messages: [] });
1046
1071
  assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "enabled", budget_tokens: 4096 });
1047
1072
 
1048
1073
  mock.restoreAll();
1049
1074
  // 0 → explicit disabled
1050
- const off = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, reasoningStyle: "anthropic" });
1075
+ const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, reasoningStyle: "anthropic" });
1051
1076
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1052
1077
  await off.generate({ workerId: "r", messages: [] });
1053
1078
  assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "disabled" });
1054
1079
 
1055
1080
  mock.restoreAll();
1056
1081
  // -1 adaptive → omit (API default depth)
1057
- const adaptive = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, reasoningStyle: "anthropic" });
1082
+ const adaptive = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, reasoningStyle: "anthropic" });
1058
1083
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1059
1084
  await adaptive.generate({ workerId: "r", messages: [] });
1060
1085
  assert.equal("thinking" in JSON.parse(calls[0].init.body as string), false);
@@ -1072,7 +1097,7 @@ test("streaming:false posts without stream and parses the single JSON response",
1072
1097
  usage: { prompt_tokens: 3, completion_tokens: 1, total_tokens: 4 },
1073
1098
  }), { status: 200, headers: { "Content-Type": "application/json" } });
1074
1099
  });
1075
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
1100
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
1076
1101
  const res = await p.generate({ workerId: "r", messages: [] });
1077
1102
  const sent = JSON.parse(calls[0].body);
1078
1103
  assert.equal("stream" in sent, false); // no streaming flag
@@ -1084,11 +1109,11 @@ test("streaming:false posts without stream and parses the single JSON response",
1084
1109
  });
1085
1110
 
1086
1111
  // ── Data capture (#36): logprobs + verbatim rawBody, opt-in, off by default ──
1087
- const captureBase = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
1112
+ const captureBase = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
1088
1113
 
1089
1114
  test("#36 logprobs OFF by default: no wire request, no assistant.logprobs, no rawBody", async () => {
1090
1115
  const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
1091
- const p = new OpenAICompatProvider({ ...captureBase });
1116
+ const p = new AiSdkProvider({ ...captureBase });
1092
1117
  const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
1093
1118
  const body = JSON.parse((calls[0].init.body as string));
1094
1119
  assert.equal("logprobs" in body, false);
@@ -1105,7 +1130,7 @@ test("#36 logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw l
1105
1130
  { token: "no", logprob: -0.1, sampling_logprob: -0.1, top_logprobs: [{ token: "no", logprob: -0.1 }] },
1106
1131
  ] } }] };
1107
1132
  const calls = installFetch([chunk]);
1108
- const p = new OpenAICompatProvider({ ...captureBase, topLogprobs: 2 });
1133
+ const p = new AiSdkProvider({ ...captureBase, topLogprobs: 2 });
1109
1134
  const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
1110
1135
  const body = JSON.parse((calls[0].init.body as string));
1111
1136
  assert.equal(body.logprobs, true);
@@ -1119,7 +1144,7 @@ test("#36 logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw l
1119
1144
  test("#36 rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob preserved", async () => {
1120
1145
  const wire = { model: "m", extra_top_level: "kept", choices: [{ message: { content: "no" }, finish_reason: "stop", logprobs: { content: [{ token: "no", logprob: -0.1, sampling_logprob: -0.1, token_id: 42 }] } }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } };
1121
1146
  installFetchJson(wire);
1122
- const p = new OpenAICompatProvider({ ...captureBase, streaming: false, topLogprobs: 0, rawBody: true });
1147
+ const p = new AiSdkProvider({ ...captureBase, streaming: false, topLogprobs: 0, rawBody: true });
1123
1148
  const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
1124
1149
  assert.deepEqual(res.rawBody, wire); // verbatim
1125
1150
  assert.equal((res.rawBody as typeof wire).choices[0].logprobs.content[0].sampling_logprob, -0.1);
@@ -1130,7 +1155,7 @@ test("#36 rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob p
1130
1155
 
1131
1156
  test("#36 caller sampling cannot forge logprobs (reserved keys): the env flag is the only control", async () => {
1132
1157
  const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
1133
- const p = new OpenAICompatProvider({ ...captureBase }); // logprobs OFF
1158
+ const p = new AiSdkProvider({ ...captureBase }); // logprobs OFF
1134
1159
  await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }], sampling: { logprobs: true, top_logprobs: 5 } });
1135
1160
  const body = JSON.parse((calls[0].init.body as string));
1136
1161
  assert.equal("logprobs" in body, false); // sampling passthrough stripped it
@@ -1141,52 +1166,52 @@ test("#36 caller sampling cannot forge logprobs (reserved keys): the env flag is
1141
1166
  // — turn coordinate headers (#404, per #391): same gate as every first-party signal —
1142
1167
 
1143
1168
  test("#404: workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under the first-party gate", async () => {
1144
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1169
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1145
1170
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1146
1171
  await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
1147
- const h = (calls[0].init.headers ?? {}) as Record<string, string>;
1148
- assert.equal(h["Plurnk-Workspace-Id"], "s-9");
1149
- assert.equal(h["Plurnk-Loop"], "3");
1150
- assert.equal(h["Plurnk-Turn"], "41");
1172
+ const headers = new Headers(calls[0].init.headers);
1173
+ assert.equal(headers.get("plurnk-workspace-id"), "s-9");
1174
+ assert.equal(headers.get("plurnk-loop"), "3");
1175
+ assert.equal(headers.get("plurnk-turn"), "41");
1151
1176
  });
1152
1177
 
1153
1178
  test("#404: third-party providers structurally DROP the coordinate (gate off by default)", async () => {
1154
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1179
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1155
1180
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1156
1181
  await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
1157
- const h = (calls[0].init.headers ?? {}) as Record<string, string>;
1158
- assert.equal("Plurnk-Workspace-Id" in h, false);
1159
- assert.equal("Plurnk-Loop" in h, false);
1160
- assert.equal("Plurnk-Turn" in h, false);
1182
+ const headers = new Headers(calls[0].init.headers);
1183
+ assert.equal(headers.has("plurnk-workspace-id"), false);
1184
+ assert.equal(headers.has("plurnk-loop"), false);
1185
+ assert.equal(headers.has("plurnk-turn"), false);
1161
1186
  });
1162
1187
 
1163
1188
  test("#404: coordinates are 1-based — 0/absent/empty emit no header (no strikes-style zero exception)", async () => {
1164
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1189
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1165
1190
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1166
1191
  await p.generate({ workerId: "r", messages: [], workspaceId: "", loop: 0, turn: 0 });
1167
- const h = (calls[0].init.headers ?? {}) as Record<string, string>;
1168
- assert.equal("Plurnk-Workspace-Id" in h, false);
1169
- assert.equal("Plurnk-Loop" in h, false);
1170
- assert.equal("Plurnk-Turn" in h, false);
1171
- assert.equal(typeof h["Plurnk-Strikes"], "undefined"); // and absent strikes stays absent
1192
+ const headers = new Headers(calls[0].init.headers);
1193
+ assert.equal(headers.has("plurnk-workspace-id"), false);
1194
+ assert.equal(headers.has("plurnk-loop"), false);
1195
+ assert.equal(headers.has("plurnk-turn"), false);
1196
+ assert.equal(headers.has("plurnk-strikes"), false);
1172
1197
  });
1173
1198
 
1174
1199
  // -- #507: envelope surface + router-owned tuning --
1175
1200
 
1176
1201
  test("#507 reserves derive from the detected window; absolutes stand alone; null window + percent = no claim", () => {
1177
- const base = { model: "m", url: "http://x", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
1178
- const derived = new OpenAICompatProvider({ ...base, contextWindow: 49152, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 } });
1202
+ const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
1203
+ const derived = new AiSdkProvider({ ...base, contextWindow: 49152, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 } });
1179
1204
  assert.equal(derived.reasoningReserve, 4915); // jennifer/turboderp: 10% of 49152
1180
1205
  assert.equal(derived.completionReserve, 12288); // 25% of 49152
1181
- const pinned = new OpenAICompatProvider({ ...base, contextWindow: null, reasoningReserve: { tokens: 4096 }, completionReserve: { percent: 0.25 } });
1206
+ const pinned = new AiSdkProvider({ ...base, contextWindow: null, reasoningReserve: { tokens: 4096 }, completionReserve: { percent: 0.25 } });
1182
1207
  assert.equal(pinned.reasoningReserve, 4096); // absolute pin needs no window
1183
1208
  assert.equal(pinned.completionReserve, null); // percent without a window = underivable
1184
- const legacy = new OpenAICompatProvider({ ...base, contextWindow: 49152 });
1209
+ const legacy = new AiSdkProvider({ ...base, contextWindow: 49152 });
1185
1210
  assert.equal(legacy.reasoningReserve, null); // out-of-date sibling: no claim
1186
1211
  });
1187
1212
 
1188
1213
  test("#507 router-owned tuning: tuningFloors:false drops the temperature/penalty floors, caller sampling still rides", async () => {
1189
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, tuningFloors: false });
1214
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0, tuningFloors: false });
1190
1215
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1191
1216
  await p.generate({ workerId: "r", messages: [], sampling: { temperature: 0.9 } });
1192
1217
  const body = JSON.parse(calls[0].init.body as string);
@@ -1197,21 +1222,21 @@ test("#507 router-owned tuning: tuningFloors:false drops the temperature/penalty
1197
1222
  // -- #518: prompt-cache affinity (workerId -> prompt_cache_key) --
1198
1223
 
1199
1224
  test("#518 promptCacheKey on: body sends prompt_cache_key = workerId (serverless replica affinity)", async () => {
1200
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
1225
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
1201
1226
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1202
1227
  await p.generate({ workerId: "worker-abc", messages: [] });
1203
1228
  assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc");
1204
1229
  });
1205
1230
 
1206
1231
  test("#518 promptCacheKey off (default): no prompt_cache_key on the wire", async () => {
1207
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1232
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1208
1233
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1209
1234
  await p.generate({ workerId: "worker-abc", messages: [] });
1210
1235
  assert.equal("prompt_cache_key" in JSON.parse(calls[0].init.body as string), false);
1211
1236
  });
1212
1237
 
1213
1238
  test("#518 prompt_cache_key is managed: caller sampling cannot forge/override the affinity key", async () => {
1214
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
1239
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
1215
1240
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1216
1241
  await p.generate({ workerId: "worker-abc", messages: [], sampling: { prompt_cache_key: "hijack" } });
1217
1242
  assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc"); // managed wins