@plurnk/plurnk-providers 1.3.5 → 1.3.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/.env.defaults +35 -45
  2. package/README.md +44 -53
  3. package/SPEC.md +215 -371
  4. package/dist/AiSdkProvider.d.ts +78 -0
  5. package/dist/AiSdkProvider.d.ts.map +1 -0
  6. package/dist/AiSdkProvider.js +591 -0
  7. package/dist/AiSdkProvider.js.map +1 -0
  8. package/dist/OpenAICompat.d.ts +1 -2
  9. package/dist/OpenAICompat.d.ts.map +1 -1
  10. package/dist/OpenAICompat.js +39 -117
  11. package/dist/OpenAICompat.js.map +1 -1
  12. package/dist/ProviderRegistry.d.ts.map +1 -1
  13. package/dist/ProviderRegistry.js +37 -24
  14. package/dist/ProviderRegistry.js.map +1 -1
  15. package/dist/aiSdkTransport.d.ts +52 -0
  16. package/dist/aiSdkTransport.d.ts.map +1 -0
  17. package/dist/aiSdkTransport.js +294 -0
  18. package/dist/aiSdkTransport.js.map +1 -0
  19. package/dist/catalogProvider.d.ts +15 -0
  20. package/dist/catalogProvider.d.ts.map +1 -0
  21. package/dist/catalogProvider.js +103 -0
  22. package/dist/catalogProvider.js.map +1 -0
  23. package/dist/compatibleProvider.d.ts +3 -0
  24. package/dist/compatibleProvider.d.ts.map +1 -0
  25. package/dist/compatibleProvider.js +146 -0
  26. package/dist/compatibleProvider.js.map +1 -0
  27. package/dist/discover.d.ts.map +1 -1
  28. package/dist/discover.js.map +1 -1
  29. package/dist/env.d.ts +1 -0
  30. package/dist/env.d.ts.map +1 -1
  31. package/dist/env.js +13 -6
  32. package/dist/env.js.map +1 -1
  33. package/dist/index.d.ts +4 -6
  34. package/dist/index.d.ts.map +1 -1
  35. package/dist/index.js +3 -7
  36. package/dist/index.js.map +1 -1
  37. package/dist/ollama.d.ts +3 -0
  38. package/dist/ollama.d.ts.map +1 -0
  39. package/dist/ollama.js +39 -0
  40. package/dist/ollama.js.map +1 -0
  41. package/dist/openai.d.ts +2 -4
  42. package/dist/openai.d.ts.map +1 -1
  43. package/dist/openai.js +1 -2
  44. package/dist/openai.js.map +1 -1
  45. package/dist/sdkModels.d.ts +13 -0
  46. package/dist/sdkModels.d.ts.map +1 -0
  47. package/dist/sdkModels.js +153 -0
  48. package/dist/sdkModels.js.map +1 -0
  49. package/dist/standardProviders.d.ts.map +1 -1
  50. package/dist/standardProviders.js +0 -1
  51. package/dist/standardProviders.js.map +1 -1
  52. package/dist/telemetry.d.ts.map +1 -1
  53. package/dist/telemetry.js +20 -9
  54. package/dist/telemetry.js.map +1 -1
  55. package/dist/types.d.ts +3 -2
  56. package/dist/types.d.ts.map +1 -1
  57. package/package.json +18 -10
  58. package/src/{OpenAICompat.test.ts → AiSdkProvider.test.ts} +193 -152
  59. package/src/{OpenAICompat.ts → AiSdkProvider.ts} +83 -127
  60. package/src/Mock.test.ts +1 -1
  61. package/src/ProviderRegistry.test.ts +40 -27
  62. package/src/ProviderRegistry.ts +35 -24
  63. package/src/aiSdkTransport.test.ts +253 -0
  64. package/src/aiSdkTransport.ts +369 -0
  65. package/src/boundaries.test.ts +2 -2
  66. package/src/catalogProvider.test.ts +100 -0
  67. package/src/catalogProvider.ts +151 -0
  68. package/src/compatibleProvider.test.ts +44 -0
  69. package/src/compatibleProvider.ts +205 -0
  70. package/src/discover.test.ts +12 -12
  71. package/src/discover.ts +3 -6
  72. package/src/env.ts +14 -6
  73. package/src/index.ts +6 -10
  74. package/src/ollama.ts +63 -0
  75. package/src/openai.ts +2 -8
  76. package/src/sdkModels.test.ts +47 -0
  77. package/src/sdkModels.ts +194 -0
  78. package/src/telemetry.test.ts +17 -10
  79. package/src/telemetry.ts +22 -14
  80. package/src/types.ts +5 -8
  81. package/src/aiSdkAdapter.spike.test.ts +0 -242
  82. package/src/openaiStream.ts +0 -310
  83. package/src/standardProviders.test.ts +0 -939
  84. package/src/standardProviders.ts +0 -631
@@ -1,13 +1,37 @@
1
1
  import test, { mock } from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
- import OpenAICompatProvider, { effortFromBudget } from "./OpenAICompat.ts";
4
- import { OpenAiHttpError } from "./openaiStream.ts";
3
+ import AiSdkProvider, { effortFromBudget } from "./AiSdkProvider.ts";
5
4
  import { ProviderError } from "./telemetry.ts";
6
5
 
7
6
  // Build a fake fetch returning a one-chunk SSE stream, capturing the request
8
7
  // so tests can assert what the spine sent on the wire.
9
8
  const sseStream = (chunks: unknown[]) => {
10
- const lines = [...chunks.map((c) => `data: ${JSON.stringify(c)}`), "data: [DONE]"].join("\n\n");
9
+ const normalized = chunks.map((value, index) => {
10
+ const chunk = value as Record<string, any>;
11
+ const usage = chunk.usage !== undefined
12
+ ? {
13
+ ...chunk.usage,
14
+ ...(chunk.usage.cached_tokens !== undefined
15
+ ? {
16
+ prompt_tokens_details: {
17
+ cached_tokens: chunk.usage.cached_tokens,
18
+ },
19
+ }
20
+ : {}),
21
+ }
22
+ : undefined;
23
+ if (usage !== undefined) delete usage.cached_tokens;
24
+ return {
25
+ id: "test-completion",
26
+ object: "chat.completion.chunk",
27
+ created: index + 1,
28
+ model: "m",
29
+ ...chunk,
30
+ ...(chunk.choices === undefined && usage !== undefined ? { choices: [] } : {}),
31
+ ...(usage !== undefined ? { usage } : {}),
32
+ };
33
+ });
34
+ const lines = [...normalized.map((c) => `data: ${JSON.stringify(c)}`), "data: [DONE]"].join("\n\n");
11
35
  return new ReadableStream({
12
36
  start(controller) {
13
37
  controller.enqueue(new TextEncoder().encode(lines));
@@ -43,7 +67,6 @@ const injectedBase = {
43
67
  fetchTimeoutMs: 5000,
44
68
  temperature: 0.2,
45
69
  repeatPenalty: 1.15,
46
- retryDelayMs: 1,
47
70
  retryAttempts: 0,
48
71
  reasoning: { mode: "off" as const, budget: null },
49
72
  };
@@ -65,9 +88,9 @@ test("#608: per-instance fetch owns streaming and buffered requests", async () =
65
88
  }), { status: 200, headers: { "Content-Type": "application/json" } });
66
89
  };
67
90
 
68
- const streamed = await new OpenAICompatProvider({ ...injectedBase, fetch: streamingFetch, rawBody: true })
91
+ const streamed = await new AiSdkProvider({ ...injectedBase, fetch: streamingFetch, rawBody: true })
69
92
  .generate({ workerId: "stream", messages: [{ role: "user", content: "hello" }] });
70
- const buffered = await new OpenAICompatProvider({ ...injectedBase, fetch: bufferedFetch, streaming: false })
93
+ const buffered = await new AiSdkProvider({ ...injectedBase, fetch: bufferedFetch, streaming: false })
71
94
  .generate({ workerId: "buffer", messages: [{ role: "user", content: "hello" }] });
72
95
 
73
96
  assert.equal(streamed.assistant.content, "streamed");
@@ -83,18 +106,20 @@ test("#608: per-instance fetch owns streaming and buffered requests", async () =
83
106
  });
84
107
 
85
108
  test("#608: caller cancellation and provider timeout reach an injected fetch", async () => {
86
- const pendingFetch: typeof globalThis.fetch = async (_input, init) =>
87
- new Promise((_resolve, reject) => {
109
+ const pendingFetch: typeof globalThis.fetch = async (_input, init) => {
110
+ init?.signal?.throwIfAborted();
111
+ return new Promise((_resolve, reject) => {
88
112
  const signal = init?.signal;
89
113
  signal?.addEventListener("abort", () => reject(signal.reason), { once: true });
90
114
  });
115
+ };
91
116
  const caller = new AbortController();
92
- const callerProvider = new OpenAICompatProvider({ ...injectedBase, fetch: pendingFetch });
117
+ const callerProvider = new AiSdkProvider({ ...injectedBase, fetch: pendingFetch });
93
118
  const callerRequest = callerProvider.generate({ workerId: "cancel", messages: [], signal: caller.signal });
94
119
  caller.abort(new Error("operator cancelled"));
95
120
  await assert.rejects(callerRequest, /operator cancelled/);
96
121
 
97
- const timeoutProvider = new OpenAICompatProvider({ ...injectedBase, fetch: pendingFetch, fetchTimeoutMs: 1 });
122
+ const timeoutProvider = new AiSdkProvider({ ...injectedBase, fetch: pendingFetch, fetchTimeoutMs: 1 });
98
123
  await assert.rejects(
99
124
  timeoutProvider.generate({ workerId: "timeout", messages: [] }),
100
125
  (error: ProviderError) => error.kind === "network_failure",
@@ -116,7 +141,7 @@ test("#608: per-instance fetch owns tokenization and retry attempts", async () =
116
141
  { choices: [], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } },
117
142
  ]), { status: 200 });
118
143
  };
119
- const provider = new OpenAICompatProvider({
144
+ const provider = new AiSdkProvider({
120
145
  ...injectedBase,
121
146
  fetch: providerFetch,
122
147
  retryAttempts: 1,
@@ -135,7 +160,13 @@ test("#608: per-instance fetch owns tokenization and retry attempts", async () =
135
160
  // Sequenced fetch mock for retry tests: each entry is one HTTP response. A 200
136
161
  // streams its chunks; any other status returns that error (with an optional
137
162
  // retry-after header). The last entry repeats once the script runs out.
138
- type ScriptedResponse = { status: number; chunks?: unknown[]; retryAfter?: number | string; body?: string };
163
+ type ScriptedResponse = {
164
+ status: number;
165
+ chunks?: unknown[];
166
+ retryAfter?: number | string;
167
+ shouldRetry?: boolean;
168
+ body?: string;
169
+ };
139
170
  const installFetchScript = (responses: ScriptedResponse[]) => {
140
171
  const calls: { url: string; init: RequestInit }[] = [];
141
172
  let i = 0;
@@ -144,8 +175,15 @@ const installFetchScript = (responses: ScriptedResponse[]) => {
144
175
  const r = responses[Math.min(i, responses.length - 1)];
145
176
  i++;
146
177
  if (r.status === 200) return new Response(sseStream(r.chunks ?? []), { status: 200 });
147
- const headers = r.retryAfter !== undefined ? { "retry-after": String(r.retryAfter) } : {};
148
- return new Response(r.body ?? "err", { status: r.status, headers });
178
+ const headers = {
179
+ "content-type": "application/json",
180
+ ...(r.retryAfter !== undefined ? { "retry-after": String(r.retryAfter) } : {}),
181
+ ...(r.shouldRetry !== undefined ? { "x-should-retry": String(r.shouldRetry) } : {}),
182
+ };
183
+ return new Response(
184
+ r.body ?? JSON.stringify({ error: { message: `HTTP ${r.status}` } }),
185
+ { status: r.status, headers },
186
+ );
149
187
  });
150
188
  return calls;
151
189
  };
@@ -164,33 +202,25 @@ test("effortFromBudget: maps budget to tiers", () => {
164
202
  assert.equal(effortFromBudget(4001), "high");
165
203
  });
166
204
 
167
- test("#543: OpenAiHttpError distills a non-JSON (edge/CDN HTML) body and drops the OpenAI prefix", () => {
168
- const cf = new OpenAiHttpError(524, "<!DOCTYPE html><html><body>Error code 524</body></html>", null);
169
- assert.equal(cf.message, "524 origin timeout"); // distilled: no raw HTML, no "OpenAI" prefix
170
- assert.ok(cf.body.length > 20); // raw body retained on the field for forensics
171
- const api = new OpenAiHttpError(400, '{"error":{"message":"bad param"}}', null);
172
- assert.match(api.message, /^OpenAI 400 - \{/); // JSON API error passes through verbatim
173
- });
174
-
175
205
  test("#543: a 524 Cloudflare edge timeout fails fast - not retried despite retryAttempts", async () => {
176
206
  const calls = installFetchScript([{ status: 524, retryAfter: 120 }]);
177
- const p = new OpenAICompatProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
207
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
178
208
  await assert.rejects(p.generate({ workerId: "r", messages: [] }));
179
209
  await flush();
180
210
  assert.equal(calls.length, 1); // edge code: one attempt, no retry despite retryAttempts: 3
181
211
  mock.restoreAll();
182
212
  });
183
213
 
184
- test("#548: a 422 grammar_invalid is transient — retried on the budget, surfaces as grammar_invalid", async () => {
214
+ test("#548: a 422 grammar_invalid is a failed exchange, not transport replay policy", async () => {
185
215
  const body = JSON.stringify({ error: { message: "non-conforming emission rejected: ...", type: "grammar_invalid" } });
186
216
  const calls = installFetchScript([{ status: 422, body }]);
187
- const p = new OpenAICompatProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 2 });
217
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 2 });
188
218
  await assert.rejects(
189
219
  p.generate({ workerId: "r", messages: [] }),
190
220
  (e: unknown) => e instanceof ProviderError && e.kind === "grammar_invalid",
191
221
  );
192
222
  await flush();
193
- assert.equal(calls.length, 3); // initial + 2 retries: rode the bounded budget, unlike a terminal 422
223
+ assert.equal(calls.length, 1);
194
224
  mock.restoreAll();
195
225
  });
196
226
 
@@ -199,7 +229,7 @@ test("an SSE error frame is a failed exchange, not an empty completion", async (
199
229
  status: 422,
200
230
  error: { message: "non-conforming emission rejected", type: "grammar_invalid" },
201
231
  }]);
202
- const p = new OpenAICompatProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
232
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
203
233
  await assert.rejects(
204
234
  p.generate({ workerId: "r", messages: [] }),
205
235
  (e: unknown) => e instanceof ProviderError && e.kind === "grammar_invalid",
@@ -209,27 +239,27 @@ test("an SSE error frame is a failed exchange, not an empty completion", async (
209
239
 
210
240
  test("#539: a trailing eos_token (--special EOG leak) is stripped from content", async () => {
211
241
  installFetchJson({ model: "m", choices: [{ message: { content: "the answer<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
212
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
242
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
213
243
  const res = await p.generate({ workerId: "r", messages: [] });
214
244
  assert.equal(res.assistant.content, "the answer"); // trailing <eos> gone; packet + verdict see clean bytes
215
245
  });
216
246
 
217
247
  test("#539: without a probed eos_token the content passes through untouched", async () => {
218
248
  installFetchJson({ model: "m", choices: [{ message: { content: "keeps <eos> literally" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
219
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
249
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
220
250
  const res = await p.generate({ workerId: "r", messages: [] });
221
251
  assert.equal(res.assistant.content, "keeps <eos> literally"); // no eosText (a cloud backend) -> no strip
222
252
  });
223
253
 
224
254
  test("#539: only the TRAILING eos_token is stripped; a quoted one mid-body survives", async () => {
225
255
  installFetchJson({ model: "m", choices: [{ message: { content: "quotes <eos> in the body<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 } });
226
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
256
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
227
257
  const res = await p.generate({ workerId: "r", messages: [] });
228
258
  assert.equal(res.assistant.content, "quotes <eos> in the body"); // only the tail goes
229
259
  });
230
260
 
231
261
  test("identity getters and defaults", () => {
232
- const p = new OpenAICompatProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
262
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
233
263
  assert.equal(p.model, "m");
234
264
  assert.equal(p.contextWindow, null); // default
235
265
  assert.equal(p.countTokens(""), 0);
@@ -238,8 +268,8 @@ test("identity getters and defaults", () => {
238
268
  });
239
269
 
240
270
  test("injected countTokens and calculateCost are used", () => {
241
- const p = new OpenAICompatProvider({
242
- model: "m", url: "http://x", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
271
+ const p = new AiSdkProvider({
272
+ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
243
273
  countTokens: (t) => t.length,
244
274
  calculateCost: (u) => u.total * 2,
245
275
  });
@@ -248,7 +278,7 @@ test("injected countTokens and calculateCost are used", () => {
248
278
  });
249
279
 
250
280
  test("generate maps a streamed response into ProviderResponse", async () => {
251
- const p = new OpenAICompatProvider({ model: "req-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
281
+ const p = new AiSdkProvider({ model: "req-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
252
282
  installFetch([
253
283
  { model: "wire-model", choices: [{ delta: { content: "hel" } }] },
254
284
  { choices: [{ delta: { content: "lo" }, finish_reason: "stop" }] },
@@ -263,31 +293,42 @@ test("generate maps a streamed response into ProviderResponse", async () => {
263
293
  assert.notEqual(assistantRaw, undefined);
264
294
  });
265
295
 
266
- test("generate normalizes an out-of-set finish_reason to null", async () => {
267
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
296
+ test("generate surfaces and normalizes an out-of-set finish_reason", async () => {
297
+ const warnings: Array<{ message: string; code?: string }> = [];
298
+ mock.method(process, "emitWarning", (message: string | Error, options?: string | { code?: string }) => {
299
+ warnings.push({
300
+ message: String(message),
301
+ ...(typeof options === "object" && options.code !== undefined ? { code: options.code } : {}),
302
+ });
303
+ });
304
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
268
305
  installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "function_call" }] }]);
269
306
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
270
307
  assert.equal(assistant.finishReason, null);
308
+ assert.deepEqual(warnings, [{
309
+ message: 'unrecognized finish_reason "function_call"; treated as no-signal (finishReason=null). If it denotes a token-cap hit, core\'s length-cap detection will miss it.',
310
+ code: "PLURNK_FINISH_REASON_UNKNOWN",
311
+ }]);
271
312
  });
272
313
 
273
314
  test("generate translates a backend cap synonym to canonical length (#425)", async () => {
274
315
  // gemini shouts MAX_TOKENS, anthropic says max_tokens -- both must reach core as
275
316
  // "length" so its truncation check (=== "length") is a cross-backend invariant.
276
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
317
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
277
318
  installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "MAX_TOKENS" }] }]);
278
319
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
279
320
  assert.equal(assistant.finishReason, "length");
280
321
  });
281
322
 
282
323
  test("generate translates end_turn to canonical stop (#425)", async () => {
283
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
324
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
284
325
  installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "end_turn" }] }]);
285
326
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
286
327
  assert.equal(assistant.finishReason, "stop");
287
328
  });
288
329
 
289
330
  test("generate aggregates reasoning deltas under multiple field names", async () => {
290
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
331
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
291
332
  installFetch([{ choices: [{ delta: { reasoning_content: "be", thinking: "cause" } }] }]);
292
333
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
293
334
  assert.equal(assistant.reasoning, "because");
@@ -303,7 +344,7 @@ test("#482 sealed relay reasoning (non-streamed): encrypted reasoning_details su
303
344
  { type: "reasoning.text", text: "never surfaced here" },
304
345
  ],
305
346
  }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
306
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
347
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
307
348
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
308
349
  // item shape: wire `id` preserved, subtype from position (#482 widening)
309
350
  assert.deepEqual(assistant.reasoningEncrypted, [{ id: "rs_1", subtype: "message", encrypted: [{ data: "gAAAAABqBLOB", format: "openai-responses-v1" }] }]);
@@ -316,14 +357,14 @@ test("#482 widening: distinct wire ids stay distinct items (a single-object shap
316
357
  { type: "reasoning.encrypted", data: "AAA", format: "openai-responses-v1", id: "rs_1" },
317
358
  { type: "reasoning.encrypted", data: "BBB", format: "openai-responses-v1", id: "rs_2" },
318
359
  ] }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
319
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
360
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
320
361
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
321
362
  assert.equal(assistant.reasoningEncrypted?.length, 2);
322
363
  assert.deepEqual(assistant.reasoningEncrypted?.map((i) => i.id), ["rs_1", "rs_2"]);
323
364
  });
324
365
 
325
366
  test("#482 sealed relay reasoning (streamed): chunked blob concatenates per entry index", async () => {
326
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
367
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
327
368
  installFetch([
328
369
  { choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "gAAAA", format: "openai-responses-v1", id: "rs_1", index: 0 }] } }] },
329
370
  { choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "BqXYZ", id: "rs_1", index: 0 }] } }] },
@@ -335,20 +376,20 @@ test("#482 sealed relay reasoning (streamed): chunked blob concatenates per entr
335
376
  });
336
377
 
337
378
  test("reasoningStyle 'think' gates on budget != 0 (magnitude irrelevant for native)", async () => {
338
- const on = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
379
+ const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
339
380
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
340
381
  await on.generate({ workerId: "r", messages: [] });
341
382
  assert.equal(JSON.parse(calls[0].init.body as string).think, true);
342
383
 
343
384
  mock.restoreAll();
344
- const off = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
385
+ const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
345
386
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
346
387
  await off.generate({ workerId: "r", messages: [] });
347
388
  assert.equal("think" in JSON.parse(calls[0].init.body as string), false);
348
389
  });
349
390
 
350
391
  test("reasoningStyle 'effort' sends a reasoning_effort tier from the budget", async () => {
351
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "on", budget: 5000 }, retryAttempts: 0, reasoningStyle: "effort" });
392
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: 5000 }, retryAttempts: 0, reasoningStyle: "effort" });
352
393
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
353
394
  await p.generate({ workerId: "r", messages: [] });
354
395
  assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, "high");
@@ -359,7 +400,7 @@ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS (#403 —
359
400
  // 400s reasoning_effort='adaptive' for non-MiniMax models (wire-verified,
360
401
  // #403): adaptive = the backend's own default posture = omission.
361
402
  for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "on", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "on"; budget: number | null }, string | null]>) {
362
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
403
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
363
404
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
364
405
  await p.generate({ workerId: "r", messages: [] });
365
406
  const body = JSON.parse(calls[0].init.body as string);
@@ -370,7 +411,7 @@ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS (#403 —
370
411
  });
371
412
 
372
413
  test("the family temperature default rides every request; caller sampling overrides it (#30)", async () => {
373
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
414
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
374
415
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
375
416
  await p.generate({ workerId: "r", messages: [] });
376
417
  assert.equal(JSON.parse(calls[0].init.body as string).temperature, 0.2);
@@ -387,9 +428,9 @@ test("the family temperature default rides every request; caller sampling overri
387
428
  });
388
429
 
389
430
  test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box default; never on cloud", async () => {
390
- const base = { model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off" as const, budget: null }, retryAttempts: 0 };
431
+ const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off" as const, budget: null }, retryAttempts: 0 };
391
432
  // set + llamacpp -> the loop-breakers ride the wire
392
- const p = new OpenAICompatProvider({ ...base, grammarStyle: "llamacpp", dryMultiplier: 0.8, dryBase: 1.75, dryAllowedLength: 2, repeatLastN: 512 });
433
+ const p = new AiSdkProvider({ ...base, grammarStyle: "llamacpp", dryMultiplier: 0.8, dryBase: 1.75, dryAllowedLength: 2, repeatLastN: 512 });
393
434
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
394
435
  await p.generate({ workerId: "r", messages: [] });
395
436
  let body = JSON.parse(calls[0].init.body as string);
@@ -400,7 +441,7 @@ test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves th
400
441
  assert.equal(body.repeat_penalty, 1.15); // repeat_penalty always rides the llamacpp path
401
442
  mock.restoreAll();
402
443
  // unset -> no dry_*/repeat_last_n on the wire (box keeps its own defaults)
403
- const p2 = new OpenAICompatProvider({ ...base, grammarStyle: "llamacpp" });
444
+ const p2 = new AiSdkProvider({ ...base, grammarStyle: "llamacpp" });
404
445
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
405
446
  await p2.generate({ workerId: "r", messages: [] });
406
447
  body = JSON.parse(calls[0].init.body as string);
@@ -408,7 +449,7 @@ test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves th
408
449
  assert.equal("repeat_last_n" in body, false);
409
450
  mock.restoreAll();
410
451
  // DRY is a llama.cpp sampler: a cloud ("none") provider never emits it, even if configured
411
- const p3 = new OpenAICompatProvider({ ...base, grammarStyle: "none", dryMultiplier: 0.8, repeatLastN: 512 });
452
+ const p3 = new AiSdkProvider({ ...base, grammarStyle: "none", dryMultiplier: 0.8, repeatLastN: 512 });
412
453
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
413
454
  await p3.generate({ workerId: "r", messages: [] });
414
455
  body = JSON.parse(calls[0].init.body as string);
@@ -418,7 +459,7 @@ test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves th
418
459
  });
419
460
 
420
461
  test("llamacpp grammar path: temperature default + the managed repeat-penalty floor", async () => {
421
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
462
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
422
463
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
423
464
  await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
424
465
  const body = JSON.parse(calls[0].init.body as string);
@@ -428,13 +469,13 @@ test("llamacpp grammar path: temperature default + the managed repeat-penalty fl
428
469
 
429
470
  test("#426: the repeat penalty rides EVERY request rail-off, keyed per backend (cloud degeneration guard)", async () => {
430
471
  // llama.cpp with NO grammar carries its key too (unconstrained local is guarded)
431
- const llama = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
472
+ const llama = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
432
473
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
433
474
  await llama.generate({ workerId: "r", messages: [] });
434
475
  assert.equal(JSON.parse(calls[0].init.body as string).repeat_penalty, 1.15);
435
476
  mock.restoreAll();
436
477
  // a `none`-style cloud backend WITH a frequency penalty gets frequency_penalty (OpenAI-standard, #426)
437
- const cloud = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
478
+ const cloud = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
438
479
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
439
480
  await cloud.generate({ workerId: "r", messages: [] });
440
481
  const cloudBody = JSON.parse(calls[0].init.body as string);
@@ -443,14 +484,14 @@ test("#426: the repeat penalty rides EVERY request rail-off, keyed per backend (
443
484
  assert.equal("repeat_penalty" in cloudBody, false);
444
485
  mock.restoreAll();
445
486
  // frequencyPenalty unset (default 0) opts out cleanly - sends nothing (an out-of-date plugin runs unguarded, never breaks)
446
- const bare = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
487
+ const bare = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
447
488
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
448
489
  await bare.generate({ workerId: "r", messages: [] });
449
490
  assert.equal("frequency_penalty" in JSON.parse(calls[0].init.body as string), false);
450
491
  });
451
492
 
452
493
  test("sampling passthrough forwards caller params; managed + reserved keys win", async () => {
453
- const p = new OpenAICompatProvider({ model: "managed-model", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
494
+ const p = new AiSdkProvider({ model: "managed-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
454
495
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
455
496
  await p.generate({
456
497
  workerId: "r",
@@ -473,7 +514,7 @@ test("sampling passthrough forwards caller params; managed + reserved keys win",
473
514
  });
474
515
 
475
516
  test("#477 sampling passthrough guards contract invariants: n/tools/caps stripped, platform knobs pass", async () => {
476
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
517
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
477
518
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
478
519
  await p.generate({
479
520
  workerId: "r",
@@ -500,7 +541,7 @@ test("#477 sampling passthrough guards contract invariants: n/tools/caps strippe
500
541
  test("#488 postmortem: intent maps IDENTICALLY under a transported grammar — sanctioned channel coexists with rails", async () => {
501
542
  // The brief rails-win-the-channel clamp is REVERTED: closing the channel starved a
502
543
  // reasoning-tuned model into escaping mid-content (unconstrained, discarded, billed).
503
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
544
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
504
545
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
505
546
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
506
547
  const body = JSON.parse(calls[0].init.body as string);
@@ -514,7 +555,7 @@ test("#488 postmortem: intent maps IDENTICALLY under a transported grammar — s
514
555
  test("#488 channel-escape detector: billed completion tokens vastly beyond visible channels attach grammar_unenforced", async () => {
515
556
  // The run105 shape: tiny visible content, no reasoning, thousands billed — the decode
516
557
  // escaped into a discarded reasoning block, unconstrained.
517
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
558
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
518
559
  installFetch([
519
560
  { choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
520
561
  { usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
@@ -528,7 +569,7 @@ test("#488 channel-escape detector: billed completion tokens vastly beyond visib
528
569
  });
529
570
 
530
571
  test("#488 loud state absent on grammarless calls; no escape event without a transported grammar", async () => {
531
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
572
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
532
573
  installFetch([
533
574
  { choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
534
575
  { usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
@@ -539,33 +580,33 @@ test("#488 loud state absent on grammarless calls; no escape event without a tra
539
580
  });
540
581
 
541
582
  test("reasoningStyle 'template' always emits enable_thinking mirroring budget != 0 — explicit false, never omitted", async () => {
542
- const on = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
583
+ const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
543
584
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
544
585
  await on.generate({ workerId: "r", messages: [] });
545
586
  assert.deepEqual(JSON.parse(calls[0].init.body as string).chat_template_kwargs, { enable_thinking: true });
546
587
 
547
588
  mock.restoreAll();
548
- const off = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
589
+ const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
549
590
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
550
591
  await off.generate({ workerId: "r", messages: [] });
551
592
  assert.deepEqual(JSON.parse(calls[0].init.body as string).chat_template_kwargs, { enable_thinking: false });
552
593
  });
553
594
 
554
595
  test("budget 0 suppresses effort and include_reasoning", async () => {
555
- const effort = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "effort" });
596
+ const effort = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "effort" });
556
597
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
557
598
  await effort.generate({ workerId: "r", messages: [] });
558
599
  assert.equal("reasoning_effort" in JSON.parse(calls[0].init.body as string), false);
559
600
 
560
601
  mock.restoreAll();
561
- const relay = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
602
+ const relay = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
562
603
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
563
604
  await relay.generate({ workerId: "r", messages: [] });
564
605
  assert.equal("include_reasoning" in JSON.parse(calls[0].init.body as string), false);
565
606
  });
566
607
 
567
608
  test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", async () => {
568
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
609
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
569
610
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
570
611
  await p.generate({ workerId: "r", messages: [] });
571
612
  assert.equal(JSON.parse(calls[0].init.body as string).include_reasoning, true);
@@ -574,7 +615,7 @@ test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", asy
574
615
  // — grammar-constrained sampling (SPEC §13, issues #8/#9) —
575
616
 
576
617
  test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor", async () => {
577
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
618
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
578
619
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
579
620
  await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
580
621
  const body = JSON.parse(calls[0].init.body as string);
@@ -584,7 +625,7 @@ test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor
584
625
  });
585
626
 
586
627
  test("grammar transport 'none' (default): the grammar is never sent — no silent unconstrained", async () => {
587
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
628
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
588
629
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
589
630
  await p.generate({ workerId: "r", messages: [], grammar: "root ::= statement" });
590
631
  const body = JSON.parse(calls[0].init.body as string);
@@ -595,7 +636,7 @@ test("grammar transport 'none' (default): the grammar is never sent — no silen
595
636
  // — grammar conformance OBSERVATION (SPEC §10.14, §13): a completed exchange always
596
637
  // returns; bytes flow; a non-accept verdict rides response.telemetry —
597
638
 
598
- const grammarProvider = () => new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", source: "provider:test" });
639
+ const grammarProvider = () => new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", source: "provider:test" });
599
640
  const streamingContent = (content: string) => installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
600
641
 
601
642
  test("enforcement: conforming output passes through unchanged", async () => {
@@ -644,7 +685,7 @@ test("observation: empty content under a non-empty grammar returns with the verd
644
685
  });
645
686
 
646
687
  test("enforcement: when no grammar is sent (grammarStyle 'none'), output is NOT validated — no wire fields, no error (SPEC )", async () => {
647
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // grammarStyle defaults to "none"
688
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // grammarStyle defaults to "none"
648
689
  streamingContent("anything goes");
649
690
  const { assistant } = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' }); // grammar passed but never transported
650
691
  assert.equal(assistant.content, "anything goes"); // no enforcement check
@@ -666,7 +707,7 @@ test("enforcement: a grammar our validator can't parse is a NON-FATAL verify gap
666
707
  // — PLURNK_PROVIDERS_GBNF_DEBUG: run unconstrained, then verify the free output against the grammar —
667
708
 
668
709
  test("gbnfDebug: the grammar is NOT transported; conforming free output passes through with NO telemetry", async () => {
669
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
710
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
670
711
  const calls = installFetch([{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }]);
671
712
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
672
713
  const body = JSON.parse(calls[0].init.body as string);
@@ -677,7 +718,7 @@ test("gbnfDebug: the grammar is NOT transported; conforming free output passes t
677
718
  });
678
719
 
679
720
  test("gbnfDebug: a conflict does NOT throw — it returns the bytes plus a grammar_unenforced telemetry event with the divergence position (#24)", async () => {
680
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
721
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
681
722
  const calls = installFetch([{ choices: [{ delta: { reasoning_content: "let me think about ok", content: "xon-conforming output" }, finish_reason: "stop" }] }]);
682
723
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
683
724
  // The model's bytes survive — not discarded by a throw (the empty-turn cascade root cause).
@@ -695,7 +736,7 @@ test("gbnfDebug: a conflict does NOT throw — it returns the bytes plus a gramm
695
736
  });
696
737
 
697
738
  test("gbnfDebug: an INVALID grammar throws before any wire call — it never reaches the model", async () => {
698
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true });
739
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true });
699
740
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
700
741
  await assert.rejects(
701
742
  () => p.generate({ workerId: "r", messages: [], grammar: 'foo ::= "a"' }), // no `root` rule → invalid GBNF
@@ -707,7 +748,7 @@ test("gbnfDebug: an INVALID grammar throws before any wire call — it never rea
707
748
  // — meta bag: verbatim provider metadata (#23) —
708
749
 
709
750
  test("meta: passes backend fields through without reinterpreting monetary values", async () => {
710
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
751
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
711
752
  const balance = { amount: "0.0000042", currency: "XMR" };
712
753
  installFetchJson({ ...jsonChoice, balance, system_fingerprint: "fp_abc" });
713
754
  const res = await p.generate({ workerId: "r", messages: [] });
@@ -721,7 +762,7 @@ const headerVal = (init: RequestInit, name: string): string | undefined =>
721
762
  new Headers(init.headers).get(name) ?? undefined;
722
763
 
723
764
  test("firstPartyMetadata: attributions + client ride as Plurnk-* headers", async () => {
724
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
765
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
725
766
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
726
767
  await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0", "@foo/y@0.3.1"], client: "plurnk.nvim/1.4.0" });
727
768
  assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), '["@acme/x@1.2.0","@foo/y@0.3.1"]');
@@ -729,7 +770,7 @@ test("firstPartyMetadata: attributions + client ride as Plurnk-* headers", async
729
770
  });
730
771
 
731
772
  test("#522 Plurnk-Worker-Primary: the lineage root rides under the gate; emitted even when it equals workerId", async () => {
732
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
773
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
733
774
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
734
775
  await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
735
776
  assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), "w-root"); // a descendant: Primary != Worker-Id
@@ -748,14 +789,14 @@ test("#522 Plurnk-Worker-Primary: the lineage root rides under the gate; emitted
748
789
  });
749
790
 
750
791
  test("#522 Plurnk-Worker-Primary is structurally dropped when firstPartyMetadata is off", async () => {
751
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
792
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
752
793
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
753
794
  await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
754
795
  assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), undefined); // never reaches a third-party backend
755
796
  });
756
797
 
757
798
  test("firstPartyMetadata off (default): the headers are structurally dropped even when values are passed", async () => {
758
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
799
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
759
800
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
760
801
  await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0"], client: "plurnk-cli/2.0.0" });
761
802
  assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), undefined); // never leaks to a non-first-party backend
@@ -763,7 +804,7 @@ test("firstPartyMetadata off (default): the headers are structurally dropped eve
763
804
  });
764
805
 
765
806
  test("firstPartyMetadata on but empty values: no header emitted", async () => {
766
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
807
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
767
808
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
768
809
  await p.generate({ workerId: "r", messages: [], attributions: [], client: "" });
769
810
  assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), undefined);
@@ -771,7 +812,7 @@ test("firstPartyMetadata on but empty values: no header emitted", async () => {
771
812
  });
772
813
 
773
814
  test("grammar transport: no grammar passed sends no grammar field, but the penalty rides (#426)", async () => {
774
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
815
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
775
816
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
776
817
  await p.generate({ workerId: "r", messages: [] });
777
818
  const body = JSON.parse(calls[0].init.body as string);
@@ -780,7 +821,7 @@ test("grammar transport: no grammar passed sends no grammar field, but the penal
780
821
  });
781
822
 
782
823
  test("maxTokens transports as max_tokens; absent → no wire field (server default)", async () => {
783
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
824
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
784
825
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
785
826
  await p.generate({ workerId: "r", messages: [], maxTokens: 2048 });
786
827
  assert.equal(JSON.parse(calls[0].init.body as string).max_tokens, 2048);
@@ -792,7 +833,7 @@ test("maxTokens transports as max_tokens; absent → no wire field (server defau
792
833
  });
793
834
 
794
835
  test("slot affinity is internal: sticky per workerId, distinct runs spread across slots (#11)", async () => {
795
- const pinning = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
836
+ const pinning = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
796
837
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
797
838
  await pinning.generate({ workerId: "run-A", messages: [] });
798
839
  await pinning.generate({ workerId: "run-B", messages: [] });
@@ -803,20 +844,20 @@ test("slot affinity is internal: sticky per workerId, distinct runs spread acros
803
844
  });
804
845
 
805
846
  test("slot affinity: no pinning backend or unknown slotCount → no id_slot ever", async () => {
806
- const cloud = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // default: no pinning
847
+ const cloud = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // default: no pinning
807
848
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
808
849
  await cloud.generate({ workerId: "run-A", messages: [] });
809
850
  assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
810
851
 
811
852
  mock.restoreAll();
812
- const noCount = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true }); // slotCount null
853
+ const noCount = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true }); // slotCount null
813
854
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
814
855
  await noCount.generate({ workerId: "run-A", messages: [] });
815
856
  assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
816
857
  });
817
858
 
818
859
  test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; recent runs stay sticky (#11)", async () => {
819
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
860
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
820
861
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
821
862
  const slotOf = (i: number) => JSON.parse(calls[i].init.body as string).id_slot;
822
863
  for (let i = 0; i < 16; i++) await p.generate({ workerId: `r${i}`, messages: [] }); // fills the 16-entry window {r0..r15}
@@ -830,7 +871,7 @@ test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; r
830
871
 
831
872
  test("streaming:false: a non-ok response rejects as a classified ProviderError (covers the non-streamed transport)", async () => {
832
873
  const { ProviderError } = await import("./telemetry.ts");
833
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, source: "provider:test" });
874
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, source: "provider:test" });
834
875
  mock.method(globalThis, "fetch", async () => new Response("boom", { status: 500 }));
835
876
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
836
877
  assert.ok(err instanceof ProviderError);
@@ -841,14 +882,14 @@ test("streaming:false: a non-ok response rejects as a classified ProviderError (
841
882
  });
842
883
 
843
884
  test("generate fail-hards on a missing or empty workerId", async () => {
844
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
885
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
845
886
  installFetch([{ choices: [{ delta: { content: "x" } }] }]);
846
887
  await assert.rejects(() => p.generate({ workerId: "", messages: [] }), /workerId is required/);
847
888
  await assert.rejects(() => (p.generate as (a: object) => Promise<unknown>)({ messages: [] }), /workerId is required/);
848
889
  });
849
890
 
850
891
  test("messages pass through verbatim — the provider injects no turn (PLAN lives in the grammar, never a provider prefill)", async () => {
851
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
892
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
852
893
  const calls = installFetch([{ choices: [{ delta: { content: "out" } }] }]);
853
894
  const input = [{ role: "user" as const, content: "hi" }];
854
895
  const res = await p.generate({ workerId: "r", messages: input });
@@ -858,7 +899,7 @@ test("messages pass through verbatim — the provider injects no turn (PLAN live
858
899
 
859
900
  test("generate wraps an HTTP failure as a ProviderError carrying a TelemetryEvent", async () => {
860
901
  const { ProviderError } = await import("./telemetry.ts");
861
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, source: "provider:test" });
902
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, source: "provider:test" });
862
903
  mock.method(globalThis, "fetch", async () => new Response("rate limited", { status: 429 }));
863
904
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
864
905
  assert.ok(err instanceof ProviderError);
@@ -870,27 +911,28 @@ test("generate wraps an HTTP failure as a ProviderError carrying a TelemetryEven
870
911
  });
871
912
 
872
913
  test("generate rejects on a pre-aborted external signal", async () => {
873
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
914
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
874
915
  installFetch([{ choices: [{ delta: { content: "x" } }] }]);
875
916
  const signal = AbortSignal.abort(new Error("nope"));
876
917
  await assert.rejects(() => p.generate({ workerId: "r", messages: [], signal }));
877
918
  });
878
919
 
879
920
  test("configured headers and url are sent verbatim", async () => {
880
- const p = new OpenAICompatProvider({
881
- model: "m", url: "http://host/custom/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
921
+ const p = new AiSdkProvider({
922
+ model: "m", url: "http://host/custom/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
882
923
  headers: { Authorization: "Bearer secret", "X-Title": "plurnk" },
883
924
  });
884
925
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
885
926
  await p.generate({ workerId: "r", messages: [] });
886
927
  assert.equal(calls[0].url, "http://host/custom/chat/completions");
887
- assert.equal((calls[0].init.headers as Record<string, string>).Authorization, "Bearer secret");
888
- assert.equal((calls[0].init.headers as Record<string, string>)["X-Title"], "plurnk");
928
+ const headers = new Headers(calls[0].init.headers);
929
+ assert.equal(headers.get("authorization"), "Bearer secret");
930
+ assert.equal(headers.get("x-title"), "plurnk");
889
931
  });
890
932
 
891
933
  // — transient-failure retry (#18) —
892
934
 
893
- const retryCfg = { model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null } as const };
935
+ const retryCfg = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const };
894
936
 
895
937
  test("retry: a transient failure retries and a later success resolves", async () => {
896
938
  const calls = installFetchScript([
@@ -898,51 +940,52 @@ test("retry: a transient failure retries and a later success resolves", async ()
898
940
  { status: 503, retryAfter: 0 },
899
941
  { status: 200, chunks: [{ choices: [{ delta: { content: "ok" } }] }] },
900
942
  ]);
901
- const p = new OpenAICompatProvider({ ...retryCfg, retryAttempts: 3 });
943
+ const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 3 });
902
944
  const res = await p.generate({ workerId: "r", messages: [] });
903
945
  assert.equal(res.assistant.content, "ok");
904
946
  assert.equal(calls.length, 3); // 429 → 503 → 200
905
947
  });
906
948
 
907
- test("#559: streamed-body silence retries and records the recovery in durable response metadata", async () => {
949
+ test("#559: streamed-body silence fails the exchange without replaying partial output", async () => {
908
950
  let calls = 0;
909
951
  mock.method(globalThis, "fetch", async () => {
910
952
  calls++;
911
953
  if (calls === 1) {
912
954
  return new Response(new ReadableStream({
913
955
  start(controller) {
914
- controller.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"partial"}}]}\n\n'));
956
+ controller.enqueue(new TextEncoder().encode(
957
+ 'data: {"id":"first","object":"chat.completion.chunk","created":1,"model":"m","choices":[{"index":0,"delta":{"content":"partial"},"finish_reason":null}]}\n\n',
958
+ ));
959
+ setTimeout(() => controller.close(), 100);
915
960
  },
916
961
  }), { status: 200 });
917
962
  }
918
963
  return new Response(new ReadableStream({
919
964
  start(controller) {
920
- controller.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"recovered"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n'));
965
+ controller.enqueue(new TextEncoder().encode(
966
+ 'data: {"id":"second","object":"chat.completion.chunk","created":2,"model":"m","choices":[{"index":0,"delta":{"content":"recovered"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n',
967
+ ));
921
968
  controller.close();
922
969
  },
923
970
  }), { status: 200 });
924
971
  });
925
- const p = new OpenAICompatProvider({
972
+ const p = new AiSdkProvider({
926
973
  model: "m",
927
- url: "http://x",
974
+ url: "http://x/v1/chat/completions",
928
975
  fetchTimeoutMs: 1000,
929
976
  streamIdleTimeoutMs: 10,
930
977
  temperature: 0.2,
931
978
  repeatPenalty: 1.15,
932
- retryDelayMs: 1,
933
979
  reasoning: { mode: "off", budget: null },
934
980
  retryAttempts: 1,
935
981
  source: "provider:test",
936
982
  });
937
- const result = await p.generate({ workerId: "r", messages: [] });
938
- assert.equal(result.assistant.content, "recovered");
939
- assert.equal(calls, 2);
940
- const retries = result.meta?.transportRetries as Array<Record<string, unknown>>;
941
- assert.equal(retries.length, 1);
942
- assert.equal(retries[0].attempt, 1);
943
- assert.equal(retries[0].kind, "network_failure");
944
- assert.equal(typeof retries[0].elapsedMs, "number");
945
- assert.match(String(retries[0].message), /no body bytes for 10ms/);
983
+ await assert.rejects(
984
+ p.generate({ workerId: "r", messages: [] }),
985
+ (error: ProviderError) => error.kind === "network_failure"
986
+ && /chunk timeout/i.test(error.message),
987
+ );
988
+ assert.equal(calls, 1);
946
989
  mock.restoreAll();
947
990
  });
948
991
 
@@ -955,27 +998,25 @@ test("#559: a zero stream-idle timeout permits a slow inter-chunk pause", async
955
998
  controller.close();
956
999
  },
957
1000
  }), { status: 200 }));
958
- const p = new OpenAICompatProvider({
1001
+ const p = new AiSdkProvider({
959
1002
  model: "m",
960
- url: "http://x",
1003
+ url: "http://x/v1/chat/completions",
961
1004
  fetchTimeoutMs: 1000,
962
1005
  streamIdleTimeoutMs: 0,
963
1006
  temperature: 0.2,
964
1007
  repeatPenalty: 1.15,
965
- retryDelayMs: 1,
966
1008
  reasoning: { mode: "off", budget: null },
967
1009
  retryAttempts: 0,
968
1010
  });
969
1011
  const result = await p.generate({ workerId: "r", messages: [] });
970
1012
  assert.equal(result.assistant.content, "slow is valid");
971
- assert.equal(result.meta?.transportRetries, undefined);
972
1013
  mock.restoreAll();
973
1014
  });
974
1015
 
975
1016
  test("retry: exhausting the budget surfaces the classified ProviderError", async () => {
976
1017
  const { ProviderError } = await import("./telemetry.ts");
977
1018
  const calls = installFetchScript([{ status: 429, retryAfter: 0 }]); // always rate-limited
978
- const p = new OpenAICompatProvider({ ...retryCfg, retryAttempts: 2 });
1019
+ const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 2 });
979
1020
  await assert.rejects(
980
1021
  () => p.generate({ workerId: "r", messages: [] }),
981
1022
  (err: unknown) => { assert.ok(err instanceof ProviderError); assert.equal(err.kind, "rate_limit"); return true; },
@@ -988,7 +1029,7 @@ test("retry: a Retry-After HTTP-date is honored — a past date parses to a 0ms
988
1029
  { status: 503, retryAfter: "Wed, 21 Oct 2015 07:28:00 GMT" }, // date form, in the past → max(0, past−now) = 0
989
1030
  { status: 200, chunks: [{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }] },
990
1031
  ]);
991
- const p = new OpenAICompatProvider({ ...retryCfg, retryAttempts: 1 });
1032
+ const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 1 });
992
1033
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
993
1034
  assert.equal(assistant.content, "ok");
994
1035
  assert.equal(calls.length, 2); // initial 503 + one retry, no real wall-clock wait
@@ -996,14 +1037,14 @@ test("retry: a Retry-After HTTP-date is honored — a past date parses to a 0ms
996
1037
 
997
1038
  test("retry: a terminal error (401 unauthorized) is never retried", async () => {
998
1039
  const calls = installFetchScript([{ status: 401 }]);
999
- const p = new OpenAICompatProvider({ ...retryCfg, retryAttempts: 5 });
1040
+ const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 5 });
1000
1041
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), /401/);
1001
1042
  assert.equal(calls.length, 1); // terminal — no retry despite budget
1002
1043
  });
1003
1044
 
1004
1045
  test("retry: retryAttempts 0 surfaces the first transient failure immediately", async () => {
1005
1046
  const calls = installFetchScript([{ status: 503, retryAfter: 0 }]);
1006
- const p = new OpenAICompatProvider({ ...retryCfg, retryAttempts: 0 });
1047
+ const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 0 });
1007
1048
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }));
1008
1049
  assert.equal(calls.length, 1); // no retry budget
1009
1050
  });
@@ -1011,7 +1052,7 @@ test("retry: retryAttempts 0 surfaces the first transient failure immediately",
1011
1052
  test("retry: a caller abort during backoff rejects promptly with no further attempt (mid-flight abort, SPEC )", async () => {
1012
1053
  const ac = new AbortController();
1013
1054
  const calls = installFetchScript([{ status: 503, retryAfter: 5 }]); // 5s backoff we never wait out
1014
- const p = new OpenAICompatProvider({ ...retryCfg, retryAttempts: 3 });
1055
+ const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 3 });
1015
1056
  const promise = p.generate({ workerId: "r", messages: [], signal: ac.signal });
1016
1057
  await flush(); // attempt 0 fails, enters the backoff sleep
1017
1058
  assert.equal(calls.length, 1);
@@ -1024,21 +1065,21 @@ test("retry: a caller abort during backoff rejects promptly with no further atte
1024
1065
 
1025
1066
  test("reasoningStyle 'anthropic' maps the budget to the thinking param", async () => {
1026
1067
  // N>0 → enabled with budget_tokens
1027
- const capped = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "on", budget: 4096 }, reasoningStyle: "anthropic" });
1068
+ const capped = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: 4096 }, reasoningStyle: "anthropic" });
1028
1069
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1029
1070
  await capped.generate({ workerId: "r", messages: [] });
1030
1071
  assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "enabled", budget_tokens: 4096 });
1031
1072
 
1032
1073
  mock.restoreAll();
1033
1074
  // 0 → explicit disabled
1034
- const off = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, reasoningStyle: "anthropic" });
1075
+ const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, reasoningStyle: "anthropic" });
1035
1076
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1036
1077
  await off.generate({ workerId: "r", messages: [] });
1037
1078
  assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "disabled" });
1038
1079
 
1039
1080
  mock.restoreAll();
1040
1081
  // -1 adaptive → omit (API default depth)
1041
- const adaptive = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, reasoningStyle: "anthropic" });
1082
+ const adaptive = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, reasoningStyle: "anthropic" });
1042
1083
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1043
1084
  await adaptive.generate({ workerId: "r", messages: [] });
1044
1085
  assert.equal("thinking" in JSON.parse(calls[0].init.body as string), false);
@@ -1056,7 +1097,7 @@ test("streaming:false posts without stream and parses the single JSON response",
1056
1097
  usage: { prompt_tokens: 3, completion_tokens: 1, total_tokens: 4 },
1057
1098
  }), { status: 200, headers: { "Content-Type": "application/json" } });
1058
1099
  });
1059
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
1100
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
1060
1101
  const res = await p.generate({ workerId: "r", messages: [] });
1061
1102
  const sent = JSON.parse(calls[0].body);
1062
1103
  assert.equal("stream" in sent, false); // no streaming flag
@@ -1068,11 +1109,11 @@ test("streaming:false posts without stream and parses the single JSON response",
1068
1109
  });
1069
1110
 
1070
1111
  // ── Data capture (#36): logprobs + verbatim rawBody, opt-in, off by default ──
1071
- const captureBase = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
1112
+ const captureBase = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
1072
1113
 
1073
1114
  test("#36 logprobs OFF by default: no wire request, no assistant.logprobs, no rawBody", async () => {
1074
1115
  const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
1075
- const p = new OpenAICompatProvider({ ...captureBase });
1116
+ const p = new AiSdkProvider({ ...captureBase });
1076
1117
  const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
1077
1118
  const body = JSON.parse((calls[0].init.body as string));
1078
1119
  assert.equal("logprobs" in body, false);
@@ -1089,7 +1130,7 @@ test("#36 logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw l
1089
1130
  { token: "no", logprob: -0.1, sampling_logprob: -0.1, top_logprobs: [{ token: "no", logprob: -0.1 }] },
1090
1131
  ] } }] };
1091
1132
  const calls = installFetch([chunk]);
1092
- const p = new OpenAICompatProvider({ ...captureBase, topLogprobs: 2 });
1133
+ const p = new AiSdkProvider({ ...captureBase, topLogprobs: 2 });
1093
1134
  const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
1094
1135
  const body = JSON.parse((calls[0].init.body as string));
1095
1136
  assert.equal(body.logprobs, true);
@@ -1103,7 +1144,7 @@ test("#36 logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw l
1103
1144
  test("#36 rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob preserved", async () => {
1104
1145
  const wire = { model: "m", extra_top_level: "kept", choices: [{ message: { content: "no" }, finish_reason: "stop", logprobs: { content: [{ token: "no", logprob: -0.1, sampling_logprob: -0.1, token_id: 42 }] } }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } };
1105
1146
  installFetchJson(wire);
1106
- const p = new OpenAICompatProvider({ ...captureBase, streaming: false, topLogprobs: 0, rawBody: true });
1147
+ const p = new AiSdkProvider({ ...captureBase, streaming: false, topLogprobs: 0, rawBody: true });
1107
1148
  const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
1108
1149
  assert.deepEqual(res.rawBody, wire); // verbatim
1109
1150
  assert.equal((res.rawBody as typeof wire).choices[0].logprobs.content[0].sampling_logprob, -0.1);
@@ -1114,7 +1155,7 @@ test("#36 rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob p
1114
1155
 
1115
1156
  test("#36 caller sampling cannot forge logprobs (reserved keys): the env flag is the only control", async () => {
1116
1157
  const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
1117
- const p = new OpenAICompatProvider({ ...captureBase }); // logprobs OFF
1158
+ const p = new AiSdkProvider({ ...captureBase }); // logprobs OFF
1118
1159
  await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }], sampling: { logprobs: true, top_logprobs: 5 } });
1119
1160
  const body = JSON.parse((calls[0].init.body as string));
1120
1161
  assert.equal("logprobs" in body, false); // sampling passthrough stripped it
@@ -1125,52 +1166,52 @@ test("#36 caller sampling cannot forge logprobs (reserved keys): the env flag is
1125
1166
  // — turn coordinate headers (#404, per #391): same gate as every first-party signal —
1126
1167
 
1127
1168
  test("#404: workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under the first-party gate", async () => {
1128
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1169
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1129
1170
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1130
1171
  await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
1131
- const h = (calls[0].init.headers ?? {}) as Record<string, string>;
1132
- assert.equal(h["Plurnk-Workspace-Id"], "s-9");
1133
- assert.equal(h["Plurnk-Loop"], "3");
1134
- assert.equal(h["Plurnk-Turn"], "41");
1172
+ const headers = new Headers(calls[0].init.headers);
1173
+ assert.equal(headers.get("plurnk-workspace-id"), "s-9");
1174
+ assert.equal(headers.get("plurnk-loop"), "3");
1175
+ assert.equal(headers.get("plurnk-turn"), "41");
1135
1176
  });
1136
1177
 
1137
1178
  test("#404: third-party providers structurally DROP the coordinate (gate off by default)", async () => {
1138
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1179
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1139
1180
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1140
1181
  await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
1141
- const h = (calls[0].init.headers ?? {}) as Record<string, string>;
1142
- assert.equal("Plurnk-Workspace-Id" in h, false);
1143
- assert.equal("Plurnk-Loop" in h, false);
1144
- assert.equal("Plurnk-Turn" in h, false);
1182
+ const headers = new Headers(calls[0].init.headers);
1183
+ assert.equal(headers.has("plurnk-workspace-id"), false);
1184
+ assert.equal(headers.has("plurnk-loop"), false);
1185
+ assert.equal(headers.has("plurnk-turn"), false);
1145
1186
  });
1146
1187
 
1147
1188
  test("#404: coordinates are 1-based — 0/absent/empty emit no header (no strikes-style zero exception)", async () => {
1148
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1189
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1149
1190
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1150
1191
  await p.generate({ workerId: "r", messages: [], workspaceId: "", loop: 0, turn: 0 });
1151
- const h = (calls[0].init.headers ?? {}) as Record<string, string>;
1152
- assert.equal("Plurnk-Workspace-Id" in h, false);
1153
- assert.equal("Plurnk-Loop" in h, false);
1154
- assert.equal("Plurnk-Turn" in h, false);
1155
- assert.equal(typeof h["Plurnk-Strikes"], "undefined"); // and absent strikes stays absent
1192
+ const headers = new Headers(calls[0].init.headers);
1193
+ assert.equal(headers.has("plurnk-workspace-id"), false);
1194
+ assert.equal(headers.has("plurnk-loop"), false);
1195
+ assert.equal(headers.has("plurnk-turn"), false);
1196
+ assert.equal(headers.has("plurnk-strikes"), false);
1156
1197
  });
1157
1198
 
1158
1199
  // -- #507: envelope surface + router-owned tuning --
1159
1200
 
1160
1201
  test("#507 reserves derive from the detected window; absolutes stand alone; null window + percent = no claim", () => {
1161
- const base = { model: "m", url: "http://x", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
1162
- const derived = new OpenAICompatProvider({ ...base, contextWindow: 49152, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 } });
1202
+ const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
1203
+ const derived = new AiSdkProvider({ ...base, contextWindow: 49152, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 } });
1163
1204
  assert.equal(derived.reasoningReserve, 4915); // jennifer/turboderp: 10% of 49152
1164
1205
  assert.equal(derived.completionReserve, 12288); // 25% of 49152
1165
- const pinned = new OpenAICompatProvider({ ...base, contextWindow: null, reasoningReserve: { tokens: 4096 }, completionReserve: { percent: 0.25 } });
1206
+ const pinned = new AiSdkProvider({ ...base, contextWindow: null, reasoningReserve: { tokens: 4096 }, completionReserve: { percent: 0.25 } });
1166
1207
  assert.equal(pinned.reasoningReserve, 4096); // absolute pin needs no window
1167
1208
  assert.equal(pinned.completionReserve, null); // percent without a window = underivable
1168
- const legacy = new OpenAICompatProvider({ ...base, contextWindow: 49152 });
1209
+ const legacy = new AiSdkProvider({ ...base, contextWindow: 49152 });
1169
1210
  assert.equal(legacy.reasoningReserve, null); // out-of-date sibling: no claim
1170
1211
  });
1171
1212
 
1172
1213
  test("#507 router-owned tuning: tuningFloors:false drops the temperature/penalty floors, caller sampling still rides", async () => {
1173
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, tuningFloors: false });
1214
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0, tuningFloors: false });
1174
1215
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1175
1216
  await p.generate({ workerId: "r", messages: [], sampling: { temperature: 0.9 } });
1176
1217
  const body = JSON.parse(calls[0].init.body as string);
@@ -1181,21 +1222,21 @@ test("#507 router-owned tuning: tuningFloors:false drops the temperature/penalty
1181
1222
  // -- #518: prompt-cache affinity (workerId -> prompt_cache_key) --
1182
1223
 
1183
1224
  test("#518 promptCacheKey on: body sends prompt_cache_key = workerId (serverless replica affinity)", async () => {
1184
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
1225
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
1185
1226
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1186
1227
  await p.generate({ workerId: "worker-abc", messages: [] });
1187
1228
  assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc");
1188
1229
  });
1189
1230
 
1190
1231
  test("#518 promptCacheKey off (default): no prompt_cache_key on the wire", async () => {
1191
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1232
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1192
1233
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1193
1234
  await p.generate({ workerId: "worker-abc", messages: [] });
1194
1235
  assert.equal("prompt_cache_key" in JSON.parse(calls[0].init.body as string), false);
1195
1236
  });
1196
1237
 
1197
1238
  test("#518 prompt_cache_key is managed: caller sampling cannot forge/override the affinity key", async () => {
1198
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
1239
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
1199
1240
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1200
1241
  await p.generate({ workerId: "worker-abc", messages: [], sampling: { prompt_cache_key: "hijack" } });
1201
1242
  assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc"); // managed wins