@plurnk/plurnk-providers 1.5.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. package/.env.defaults +36 -22
  2. package/SPEC.md +133 -59
  3. package/dist/AiSdkProvider.d.ts +19 -26
  4. package/dist/AiSdkProvider.d.ts.map +1 -1
  5. package/dist/AiSdkProvider.js +318 -106
  6. package/dist/AiSdkProvider.js.map +1 -1
  7. package/dist/Mock.d.ts +4 -9
  8. package/dist/Mock.d.ts.map +1 -1
  9. package/dist/Mock.js +36 -9
  10. package/dist/Mock.js.map +1 -1
  11. package/dist/Pool.d.ts +2 -21
  12. package/dist/Pool.d.ts.map +1 -1
  13. package/dist/Pool.js +19 -14
  14. package/dist/Pool.js.map +1 -1
  15. package/dist/accounting.d.ts +5 -2
  16. package/dist/accounting.d.ts.map +1 -1
  17. package/dist/accounting.js +100 -16
  18. package/dist/accounting.js.map +1 -1
  19. package/dist/aiSdkTransport.d.ts +9 -2
  20. package/dist/aiSdkTransport.d.ts.map +1 -1
  21. package/dist/aiSdkTransport.js +160 -62
  22. package/dist/aiSdkTransport.js.map +1 -1
  23. package/dist/catalogProvider.d.ts +7 -3
  24. package/dist/catalogProvider.d.ts.map +1 -1
  25. package/dist/catalogProvider.js +30 -24
  26. package/dist/catalogProvider.js.map +1 -1
  27. package/dist/compatibleProvider.d.ts.map +1 -1
  28. package/dist/compatibleProvider.js +18 -7
  29. package/dist/compatibleProvider.js.map +1 -1
  30. package/dist/cost.d.ts +10 -10
  31. package/dist/cost.d.ts.map +1 -1
  32. package/dist/cost.js +90 -42
  33. package/dist/cost.js.map +1 -1
  34. package/dist/env.d.ts +5 -1
  35. package/dist/env.d.ts.map +1 -1
  36. package/dist/env.js +30 -10
  37. package/dist/env.js.map +1 -1
  38. package/dist/errors.d.ts +14 -2
  39. package/dist/errors.d.ts.map +1 -1
  40. package/dist/errors.js +58 -2
  41. package/dist/errors.js.map +1 -1
  42. package/dist/index.d.ts +4 -4
  43. package/dist/index.d.ts.map +1 -1
  44. package/dist/index.js +3 -2
  45. package/dist/index.js.map +1 -1
  46. package/dist/ollama.js +3 -3
  47. package/dist/ollama.js.map +1 -1
  48. package/dist/sdkModels.d.ts +6 -2
  49. package/dist/sdkModels.d.ts.map +1 -1
  50. package/dist/sdkModels.js +38 -5
  51. package/dist/sdkModels.js.map +1 -1
  52. package/dist/types.d.ts +33 -31
  53. package/dist/types.d.ts.map +1 -1
  54. package/dist/usage.d.ts +21 -5
  55. package/dist/usage.d.ts.map +1 -1
  56. package/dist/usage.js +164 -83
  57. package/dist/usage.js.map +1 -1
  58. package/package.json +7 -6
  59. package/src/AiSdkProvider.test.ts +788 -191
  60. package/src/AiSdkProvider.ts +381 -124
  61. package/src/Mock.test.ts +37 -12
  62. package/src/Mock.ts +45 -14
  63. package/src/Pool.test.ts +19 -6
  64. package/src/Pool.ts +20 -16
  65. package/src/ProviderRegistry.test.ts +16 -11
  66. package/src/accounting.test.ts +58 -22
  67. package/src/accounting.ts +120 -18
  68. package/src/aiSdkTransport.test.ts +42 -49
  69. package/src/aiSdkTransport.ts +174 -62
  70. package/src/boundaries.test.ts +1 -0
  71. package/src/catalogProvider.test.ts +258 -22
  72. package/src/catalogProvider.ts +42 -27
  73. package/src/compatibleProvider.test.ts +6 -3
  74. package/src/compatibleProvider.ts +20 -7
  75. package/src/cost.test.ts +55 -36
  76. package/src/cost.ts +111 -50
  77. package/src/defaults.test.ts +13 -3
  78. package/src/env.test.ts +54 -5
  79. package/src/env.ts +43 -18
  80. package/src/errors.test.ts +47 -2
  81. package/src/errors.ts +67 -3
  82. package/src/index.ts +21 -5
  83. package/src/ollama.test.ts +4 -1
  84. package/src/ollama.ts +3 -3
  85. package/src/sdkModels.test.ts +76 -4
  86. package/src/sdkModels.ts +45 -7
  87. package/src/types.ts +77 -38
  88. package/src/usage.test.ts +112 -116
  89. package/src/usage.ts +209 -93
@@ -1,10 +1,26 @@
1
1
  import test, { mock } from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
- import AiSdkProvider, { effortFromBudget } from "./AiSdkProvider.ts";
3
+ import AiSdkProvider, { effortFromBudget, type AiSdkProviderConfig } from "./AiSdkProvider.ts";
4
4
  import { ProviderError } from "./errors.ts";
5
- import { authoritativeChargeNormalizer } from "./accounting.ts";
5
+ import { providerCostNormalizer } from "./accounting.ts";
6
6
  import type { LanguageModel } from "ai";
7
7
 
8
+ type TestProviderConfig = Omit<AiSdkProviderConfig, "operationTimeoutMs" | "firstContentTimeoutMs">
9
+ & Partial<Pick<AiSdkProviderConfig, "operationTimeoutMs" | "firstContentTimeoutMs">>;
10
+
11
+ const testProvider = (config: TestProviderConfig): AiSdkProvider => {
12
+ const {
13
+ operationTimeoutMs = config.fetchTimeoutMs,
14
+ firstContentTimeoutMs = 0,
15
+ ...rest
16
+ } = config;
17
+ return new AiSdkProvider({
18
+ ...rest,
19
+ operationTimeoutMs,
20
+ firstContentTimeoutMs,
21
+ });
22
+ };
23
+
8
24
  // Build a fake fetch returning a one-chunk SSE stream, capturing the request
9
25
  // so tests can assert what the spine sent on the wire.
10
26
  const sseStream = (chunks: unknown[]) => {
@@ -61,6 +77,31 @@ const installFetchJson = (payload: unknown) => {
61
77
  return calls;
62
78
  };
63
79
 
80
+ const settledCharge = {
81
+ kind: "charged",
82
+ amount: { amount: "0.00000042", currency: "XMR" },
83
+ usdEquivalent: "0.000071",
84
+ source: "plurnk endpoint settlement",
85
+ } as const;
86
+
87
+ const billedErrorBody = {
88
+ status: 422,
89
+ error: {
90
+ message: "non-conforming emission rejected",
91
+ type: "grammar_invalid",
92
+ },
93
+ usage: {
94
+ prompt_tokens: 8,
95
+ completion_tokens: 3,
96
+ reasoning_tokens: 0,
97
+ prompt_tokens_details: { cached_tokens: 2 },
98
+ total_tokens: 11,
99
+ },
100
+ charge: settledCharge,
101
+ };
102
+
103
+ const directCost = ({ charge }: { charge?: unknown }) => charge as typeof settledCharge | undefined;
104
+
64
105
  const jsonChoice = { model: "m", choices: [{ message: { content: "x" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } };
65
106
 
66
107
  const injectedBase = {
@@ -90,9 +131,9 @@ test("per-instance fetch owns streaming and buffered requests", async () => {
90
131
  }), { status: 200, headers: { "Content-Type": "application/json" } });
91
132
  };
92
133
 
93
- const streamed = await new AiSdkProvider({ ...injectedBase, fetch: streamingFetch, rawBody: true })
134
+ const streamed = await testProvider({ ...injectedBase, fetch: streamingFetch, rawBody: true })
94
135
  .generate({ workerId: "stream", messages: [{ role: "user", content: "hello" }] });
95
- const buffered = await new AiSdkProvider({ ...injectedBase, fetch: bufferedFetch, streaming: false })
136
+ const buffered = await testProvider({ ...injectedBase, fetch: bufferedFetch, streaming: false })
96
137
  .generate({ workerId: "buffer", messages: [{ role: "user", content: "hello" }] });
97
138
 
98
139
  assert.equal(streamed.assistant.content, "streamed");
@@ -116,12 +157,17 @@ test("caller cancellation and provider timeout reach an injected fetch", async (
116
157
  });
117
158
  };
118
159
  const caller = new AbortController();
119
- const callerProvider = new AiSdkProvider({ ...injectedBase, fetch: pendingFetch });
160
+ const callerProvider = testProvider({ ...injectedBase, fetch: pendingFetch });
120
161
  const callerRequest = callerProvider.generate({ workerId: "cancel", messages: [], signal: caller.signal });
121
162
  caller.abort(new Error("operator cancelled"));
122
163
  await assert.rejects(callerRequest, /operator cancelled/);
123
164
 
124
- const timeoutProvider = new AiSdkProvider({ ...injectedBase, fetch: pendingFetch, fetchTimeoutMs: 1 });
165
+ const timeoutProvider = testProvider({
166
+ ...injectedBase,
167
+ fetch: pendingFetch,
168
+ fetchTimeoutMs: 1,
169
+ operationTimeoutMs: 100,
170
+ });
125
171
  await assert.rejects(
126
172
  timeoutProvider.generate({ workerId: "timeout", messages: [] }),
127
173
  (error: ProviderError) => error.kind === "network_failure",
@@ -143,7 +189,7 @@ test("per-instance fetch owns tokenization and retry attempts", async () => {
143
189
  { choices: [], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } },
144
190
  ]), { status: 200 });
145
191
  };
146
- const provider = new AiSdkProvider({
192
+ const provider = testProvider({
147
193
  ...injectedBase,
148
194
  fetch: providerFetch,
149
195
  retryAttempts: 1,
@@ -159,6 +205,60 @@ test("per-instance fetch owns tokenization and retry attempts", async () => {
159
205
  ]);
160
206
  });
161
207
 
208
+ test("request-observer open failures preserve the durability cause and issue no provider I/O", async () => {
209
+ const root = new Error("durable request open failed");
210
+ let calls = 0;
211
+ const provider = testProvider({
212
+ ...injectedBase,
213
+ retryAttempts: 3,
214
+ fetch: async () => {
215
+ calls++;
216
+ return new Response(JSON.stringify(jsonChoice), {
217
+ status: 200,
218
+ headers: { "Content-Type": "application/json" },
219
+ });
220
+ },
221
+ streaming: false,
222
+ });
223
+
224
+ await assert.rejects(
225
+ provider.generate({
226
+ workerId: "observer-open",
227
+ messages: [],
228
+ observeRequest: async () => { throw root; },
229
+ }),
230
+ (error: unknown) => error === root,
231
+ );
232
+ assert.equal(calls, 0);
233
+ });
234
+
235
+ test("request-observer settlement failures preserve the durability cause without retrying I/O", async () => {
236
+ const root = new Error("durable request settlement failed");
237
+ let calls = 0;
238
+ const provider = testProvider({
239
+ ...injectedBase,
240
+ retryAttempts: 3,
241
+ fetch: async () => {
242
+ calls++;
243
+ return new Response(JSON.stringify(jsonChoice), {
244
+ status: 200,
245
+ headers: { "Content-Type": "application/json" },
246
+ });
247
+ },
248
+ streaming: false,
249
+ });
250
+
251
+ await assert.rejects(
252
+ provider.generate({
253
+ workerId: "observer-settle",
254
+ messages: [],
255
+ observeRequest: async () => async () => { throw root; },
256
+ }),
257
+ (error: unknown) => error === root,
258
+ );
259
+ assert.equal(calls, 1);
260
+ });
261
+
162
262
  // Sequenced fetch mock for retry tests: each entry is one HTTP response. A 200
163
263
  // streams its chunks; any other status returns that error (with an optional
164
264
  // retry-after header). The last entry repeats once the script runs out.
@@ -206,8 +306,13 @@ test("effortFromBudget: maps budget to tiers", () => {
206
306
 
207
307
  test("a 524 Cloudflare edge timeout fails fast - not retried despite retryAttempts", async () => {
208
308
  const calls = installFetchScript([{ status: 524, retryAfter: 120 }]);
209
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
210
- await assert.rejects(p.generate({ workerId: "r", messages: [] }));
309
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
310
+ await assert.rejects(
311
+ p.generate({ workerId: "r", messages: [] }),
312
+ (error: ProviderError) => error.kind === "network_failure"
313
+ && error.status === 524
314
+ && error.problem.retryable === false,
315
+ );
211
316
  await flush();
212
317
  assert.equal(calls.length, 1); // edge code: one attempt, no retry despite retryAttempts: 3
213
318
  mock.restoreAll();
@@ -216,7 +321,7 @@ test("a 524 Cloudflare edge timeout fails fast - not retried despite retryAttemp
216
321
  test("a 422 grammar_invalid is a failed exchange, not transport replay policy", async () => {
217
322
  const body = JSON.stringify({ error: { message: "non-conforming emission rejected: ...", type: "grammar_invalid" } });
218
323
  const calls = installFetchScript([{ status: 422, body }]);
219
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 2 });
324
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 2 });
220
325
  await assert.rejects(
221
326
  p.generate({ workerId: "r", messages: [] }),
222
327
  (e: unknown) => e instanceof ProviderError && e.kind === "grammar_invalid",
@@ -231,7 +336,7 @@ test("an SSE error frame is a failed exchange, not an empty completion", async (
231
336
  status: 422,
232
337
  error: { message: "non-conforming emission rejected", type: "grammar_invalid" },
233
338
  }]);
234
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
339
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
235
340
  await assert.rejects(
236
341
  p.generate({ workerId: "r", messages: [] }),
237
342
  (e: unknown) => e instanceof ProviderError && e.kind === "grammar_invalid",
@@ -239,29 +344,138 @@ test("an SSE error frame is a failed exchange, not an empty completion", async (
239
344
  assert.equal(calls.length, 1);
240
345
  });
241
346
 
347
+ test("a buffered classified error retains normalized usage and settled charge", async () => {
348
+ const calls = installFetchScript([{ status: 422, body: JSON.stringify(billedErrorBody) }]);
349
+ const p = testProvider({
350
+ ...injectedBase,
351
+ streaming: false,
352
+ normalizeCost: directCost,
353
+ });
354
+ await assert.rejects(
355
+ p.generate({ workerId: "billed-json-error", messages: [] }),
356
+ (error: unknown) => {
357
+ assert.ok(error instanceof ProviderError);
358
+ assert.equal(error.kind, "grammar_invalid");
359
+ assert.deepEqual(error.accounting, [{
360
+ provider: "provider",
361
+ model: "m",
362
+ outcome: "error",
363
+ status: 422,
364
+ usage: {
365
+ inputTokens: 8,
366
+ outputTokens: 3,
367
+ totalTokens: 11,
368
+ inputTokenDetails: { cacheReadTokens: 2 },
369
+ outputTokenDetails: { textTokens: 3, reasoningTokens: 0 },
370
+ },
371
+ cost: settledCharge,
372
+ }]);
373
+ assert.equal(error.attempt, undefined, "accounting evidence does not fabricate an assistant response");
374
+ return true;
375
+ },
376
+ );
377
+ assert.equal(calls.length, 1);
378
+ });
379
+
380
+ test("an SSE classified error retains the same normalized usage and settled charge", async () => {
381
+ const calls = installFetch([billedErrorBody]);
382
+ const p = testProvider({
383
+ ...injectedBase,
384
+ normalizeCost: directCost,
385
+ });
386
+ await assert.rejects(
387
+ p.generate({ workerId: "billed-sse-error", messages: [] }),
388
+ (error: unknown) => {
389
+ assert.ok(error instanceof ProviderError);
390
+ assert.equal(error.kind, "grammar_invalid");
391
+ assert.deepEqual(error.accounting, [{
392
+ provider: "provider",
393
+ model: "m",
394
+ outcome: "error",
395
+ status: 422,
396
+ usage: {
397
+ inputTokens: 8,
398
+ outputTokens: 3,
399
+ totalTokens: 11,
400
+ inputTokenDetails: { cacheReadTokens: 2 },
401
+ outputTokenDetails: { textTokens: 3, reasoningTokens: 0 },
402
+ },
403
+ cost: settledCharge,
404
+ }]);
405
+ assert.equal(error.attempt, undefined);
406
+ return true;
407
+ },
408
+ );
409
+ assert.equal(calls.length, 1);
410
+ });
411
+
412
+ test("a successful response normalizes direct charge without duplicating it as metadata", async () => {
413
+ installFetchJson({ ...jsonChoice, charge: settledCharge });
414
+ const p = testProvider({
415
+ ...injectedBase,
416
+ streaming: false,
417
+ normalizeCost: directCost,
418
+ });
419
+ const response = await p.generate({ workerId: "billed-json-success", messages: [] });
420
+ assert.deepEqual(response.accounting[0]?.cost, settledCharge);
421
+ assert.equal(response.meta?.charge, undefined);
422
+ });
423
+
424
+ test("malformed monetary evidence closes the physical request before surfacing the normalization failure", async () => {
425
+ const root = new TypeError("direct charge is malformed");
426
+ const settled: unknown[] = [];
427
+ const calls = installFetchJson({ ...jsonChoice, charge: { malformed: true } });
428
+ const provider = testProvider({
429
+ ...injectedBase,
430
+ retryAttempts: 3,
431
+ streaming: false,
432
+ normalizeCost: () => { throw root; },
433
+ });
434
+
435
+ await assert.rejects(
436
+ provider.generate({
437
+ workerId: "malformed-charge",
438
+ messages: [],
439
+ observeRequest: async () => async (accounting) => { settled.push(accounting); },
440
+ }),
441
+ (error: unknown) => error === root,
442
+ );
443
+ assert.equal(calls.length, 1);
444
+ assert.deepEqual(settled, [{
445
+ provider: "provider",
446
+ model: "m",
447
+ outcome: "response",
448
+ usage: { inputTokens: 1, outputTokens: 1, totalTokens: 2 },
449
+ cost: {
450
+ kind: "unknown",
451
+ reason: "provider request accounting could not be normalized after physical I/O",
452
+ },
453
+ }]);
454
+ });
455
+
242
456
  test("a trailing eos_token (--special EOG leak) is stripped from content", async () => {
243
457
  installFetchJson({ model: "m", choices: [{ message: { content: "the answer<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
244
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
458
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
245
459
  const res = await p.generate({ workerId: "r", messages: [] });
246
460
  assert.equal(res.assistant.content, "the answer"); // trailing <eos> gone; packet + verdict see clean bytes
247
461
  });
248
462
 
249
463
  test("without a probed eos_token the content passes through untouched", async () => {
250
464
  installFetchJson({ model: "m", choices: [{ message: { content: "keeps <eos> literally" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
251
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
465
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
252
466
  const res = await p.generate({ workerId: "r", messages: [] });
253
467
  assert.equal(res.assistant.content, "keeps <eos> literally"); // no eosText (a cloud backend) -> no strip
254
468
  });
255
469
 
256
470
  test("only the trailing eos_token is stripped; a quoted one mid-body survives", async () => {
257
471
  installFetchJson({ model: "m", choices: [{ message: { content: "quotes <eos> in the body<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 } });
258
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
472
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
259
473
  const res = await p.generate({ workerId: "r", messages: [] });
260
474
  assert.equal(res.assistant.content, "quotes <eos> in the body"); // only the tail goes
261
475
  });
262
476
 
263
477
  test("identity getters and default prompt estimate", async () => {
264
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
478
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
265
479
  assert.equal(p.model, "m");
266
480
  assert.equal(p.contextWindow, null); // default
267
481
  assert.deepEqual(
@@ -274,39 +488,54 @@ test("identity getters and default prompt estimate", async () => {
274
488
  },
275
489
  "chars/2 is explicitly an estimate; high-token-density Unicode prevents an upper-bound claim",
276
490
  );
277
- assert.equal(p.calculateCost({ prompt: 9, completion: 9, reasoning: 0, cached: 0, total: 18 }), 0); // current unknown-rate sentinel
278
491
  });
279
492
 
280
- test("injected prompt measurement preserves provenance and calculateCost is used", async () => {
493
+ test("injected prompt measurement preserves provenance and request cost estimation stays internal", async () => {
281
494
  const seen: string[] = [];
282
- const p = new AiSdkProvider({
495
+ installFetchJson(jsonChoice);
496
+ const p = testProvider({
283
497
  model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
284
498
  countPromptTokens: (messages) => {
285
499
  seen.push(...messages.map(({ content }) => content));
286
500
  return { kind: "upper_bound", tokens: 7, source: "test:proven-bound" };
287
501
  },
288
- calculateCost: (u) => u.total * 2,
502
+ streaming: false,
503
+ estimateCost: (usage) => ({
504
+ kind: "estimated",
505
+ amount: { amount: String((usage?.totalTokens ?? 0) * 2), currency: "USD" },
506
+ source: "test estimator",
507
+ }),
289
508
  });
290
509
  assert.deepEqual(
291
510
  await p.countPromptTokens([{ role: "system", content: "system" }, { role: "user", content: "user" }]),
292
511
  { kind: "upper_bound", tokens: 7, source: "test:proven-bound" },
293
512
  );
294
513
  assert.deepEqual(seen, ["system", "user"]);
295
- assert.equal(p.calculateCost({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 5 }), 10);
514
+ const response = await p.generate({ workerId: "accounted", messages: [] });
515
+ assert.deepEqual(response.accounting[0]?.cost, {
516
+ kind: "estimated",
517
+ amount: { amount: "4", currency: "USD" },
518
+ source: "test estimator",
519
+ });
296
520
  });
297
521
 
298
522
  test("generate maps a streamed response into ProviderResponse", async () => {
299
- const p = new AiSdkProvider({ model: "req-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
523
+ const p = testProvider({ model: "req-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
300
524
  installFetch([
301
525
  { model: "wire-model", choices: [{ delta: { content: "hel" } }] },
302
526
  { choices: [{ delta: { content: "lo" }, finish_reason: "stop" }] },
303
527
  { usage: { prompt_tokens: 3, completion_tokens: 2, total_tokens: 5, cached_tokens: 1 } },
304
528
  ]);
305
- const { assistant, assistantRaw } = await p.generate({ workerId: "r", messages: [{ role: "user", content: "hi" }] });
529
+ const { assistant, assistantRaw, accounting } = await p.generate({ workerId: "r", messages: [{ role: "user", content: "hi" }] });
306
530
  assert.equal(assistant.content, "hello");
307
531
  assert.equal(assistant.model, "wire-model"); // wire-reported wins
308
532
  assert.equal(assistant.finishReason, "stop");
309
- assert.deepEqual(assistant.usage, { prompt: 3, completion: 2, reasoning: 0, cached: 1, total: 5 });
533
+ assert.deepEqual(accounting[0]?.usage, {
534
+ inputTokens: 3,
535
+ outputTokens: 2,
536
+ totalTokens: 5,
537
+ inputTokenDetails: { cacheReadTokens: 1 },
538
+ });
310
539
  assert.equal(assistant.reasoning, null); // none emitted
311
540
  assert.notEqual(assistantRaw, undefined);
312
541
  });
@@ -318,9 +547,8 @@ test("native SDK accounting metadata becomes a normalized charge in buffered and
318
547
  };
319
548
  const providerMetadata = { openrouter: { usage: { cost: 0.00154935 } } };
320
549
  const charge = {
321
- kind: "authoritative",
550
+ kind: "charged",
322
551
  amount: { amount: "0.00154935", currency: "USD" },
323
- usdEquivalent: "0.00154935",
324
552
  source: "OpenRouter response usage.cost",
325
553
  };
326
554
  const languageModel = {
@@ -364,23 +592,91 @@ test("native SDK accounting metadata becomes a normalized charge in buffered and
364
592
  repeatPenalty: 1.15,
365
593
  reasoning: { mode: "off" as const, budget: null },
366
594
  retryAttempts: 0,
367
- normalizeCharge: authoritativeChargeNormalizer("@openrouter/ai-sdk-provider"),
595
+ normalizeCost: providerCostNormalizer("@openrouter/ai-sdk-provider"),
368
596
  };
369
597
 
370
598
  await t.test("buffered", async () => {
371
- const response = await new AiSdkProvider({ ...config, streaming: false })
599
+ const response = await testProvider({ ...config, streaming: false })
372
600
  .generate({ workerId: "buffered", messages: [] });
373
- assert.deepEqual(response.charge, charge);
601
+ assert.deepEqual(response.accounting[0]?.cost, charge);
374
602
  });
375
603
  await t.test("streamed", async () => {
376
- const response = await new AiSdkProvider(config)
604
+ const response = await testProvider(config)
377
605
  .generate({ workerId: "streamed", messages: [] });
378
- assert.deepEqual(response.charge, charge);
606
+ assert.deepEqual(response.accounting[0]?.cost, charge);
607
+ });
608
+ });
609
+
610
+ test("native SDK providers share the first-content retry contract", async () => {
611
+ let calls = 0;
612
+ const usage = {
613
+ inputTokens: { total: 1, noCache: 1, cacheRead: 0, cacheWrite: 0 },
614
+ outputTokens: { total: 1, text: 1, reasoning: 0 },
615
+ };
616
+ const languageModel = {
617
+ specificationVersion: "v4",
618
+ provider: "native.test",
619
+ modelId: "native-timeout",
620
+ supportedUrls: {},
621
+ doGenerate: async () => { throw new Error("buffered generation is not under test"); },
622
+ doStream: async ({ abortSignal }: { abortSignal?: AbortSignal }) => {
623
+ calls++;
624
+ if (calls > 1) {
625
+ return {
626
+ stream: new ReadableStream({
627
+ start(controller) {
628
+ controller.enqueue({ type: "stream-start", warnings: [] });
629
+ controller.enqueue({ type: "response-metadata", id: "native-retry", modelId: "native-timeout" });
630
+ controller.enqueue({ type: "text-start", id: "text-1" });
631
+ controller.enqueue({ type: "text-delta", id: "text-1", delta: "recovered" });
632
+ controller.enqueue({ type: "text-end", id: "text-1" });
633
+ controller.enqueue({
634
+ type: "finish",
635
+ finishReason: { unified: "stop", raw: "completed" },
636
+ usage,
637
+ });
638
+ controller.close();
639
+ },
640
+ }),
641
+ response: {},
642
+ };
643
+ }
644
+ return {
645
+ stream: new ReadableStream({
646
+ start(controller) {
647
+ controller.enqueue({ type: "stream-start", warnings: [] });
648
+ const timer = setTimeout(() => controller.close(), 100);
649
+ abortSignal?.addEventListener("abort", () => {
650
+ clearTimeout(timer);
651
+ controller.error(abortSignal.reason);
652
+ }, { once: true });
653
+ },
654
+ }),
655
+ response: {},
656
+ };
657
+ },
658
+ } as unknown as LanguageModel;
659
+ const provider = testProvider({
660
+ model: "native-timeout",
661
+ languageModel,
662
+ fetchTimeoutMs: 5_000,
663
+ operationTimeoutMs: 5_000,
664
+ firstContentTimeoutMs: 10,
665
+ temperature: 0.2,
666
+ repeatPenalty: 1.15,
667
+ reasoning: { mode: "off", budget: null },
668
+ retryAttempts: 1,
669
+ source: "provider:test-native",
379
670
  });
671
+
672
+ const result = await provider.generate({ workerId: "native-retry", messages: [] });
673
+ assert.equal(result.assistant.content, "recovered");
674
+ assert.equal(calls, 2);
675
+ assert.deepEqual(result.accounting.map(({ outcome }) => outcome), ["error", "response"]);
380
676
  });
381
677
 
382
678
  test("compatible xAI wire usage becomes an exact tick charge without raw-body capture", async () => {
383
- const p = new AiSdkProvider({
679
+ const p = testProvider({
384
680
  model: "grok-test",
385
681
  url: "http://x/v1/chat/completions",
386
682
  fetchTimeoutMs: 5_000,
@@ -389,7 +685,7 @@ test("compatible xAI wire usage becomes an exact tick charge without raw-body ca
389
685
  reasoning: { mode: "off", budget: null },
390
686
  retryAttempts: 0,
391
687
  streaming: false,
392
- normalizeCharge: authoritativeChargeNormalizer("@ai-sdk/xai"),
688
+ normalizeCost: providerCostNormalizer("@ai-sdk/xai"),
393
689
  });
394
690
  installFetchJson({
395
691
  id: "response-1",
@@ -403,8 +699,8 @@ test("compatible xAI wire usage becomes an exact tick charge without raw-body ca
403
699
  },
404
700
  });
405
701
  const response = await p.generate({ workerId: "xai", messages: [] });
406
- assert.deepEqual(response.charge, {
407
- kind: "authoritative",
702
+ assert.deepEqual(response.accounting[0]?.cost, {
703
+ kind: "charged",
408
704
  amount: { amount: "15493500", currency: "USDTICK" },
409
705
  usdEquivalent: "0.00154935",
410
706
  source: "xAI response usage.cost_in_usd_ticks",
@@ -413,7 +709,7 @@ test("compatible xAI wire usage becomes an exact tick charge without raw-body ca
413
709
  });
414
710
 
415
711
  test("streamed xAI final usage retains its exact tick charge", async () => {
416
- const p = new AiSdkProvider({
712
+ const p = testProvider({
417
713
  model: "grok-test",
418
714
  url: "http://x/v1/chat/completions",
419
715
  fetchTimeoutMs: 5_000,
@@ -421,7 +717,7 @@ test("streamed xAI final usage retains its exact tick charge", async () => {
421
717
  repeatPenalty: 1.15,
422
718
  reasoning: { mode: "off", budget: null },
423
719
  retryAttempts: 0,
424
- normalizeCharge: authoritativeChargeNormalizer("@ai-sdk/xai"),
720
+ normalizeCost: providerCostNormalizer("@ai-sdk/xai"),
425
721
  });
426
722
  installFetch([
427
723
  { choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] },
@@ -436,8 +732,8 @@ test("streamed xAI final usage retains its exact tick charge", async () => {
436
732
  },
437
733
  ]);
438
734
  const response = await p.generate({ workerId: "xai", messages: [] });
439
- assert.deepEqual(response.charge, {
440
- kind: "authoritative",
735
+ assert.deepEqual(response.accounting[0]?.cost, {
736
+ kind: "charged",
441
737
  amount: { amount: "15493500", currency: "USDTICK" },
442
738
  usdEquivalent: "0.00154935",
443
739
  source: "xAI response usage.cost_in_usd_ticks",
@@ -452,7 +748,7 @@ test("generate surfaces and normalizes an out-of-set finish_reason", async () =>
452
748
  ...(typeof options === "object" && options.code !== undefined ? { code: options.code } : {}),
453
749
  });
454
750
  });
455
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
751
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
456
752
  installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "function_call" }] }]);
457
753
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
458
754
  assert.equal(assistant.finishReason, null);
@@ -470,7 +766,7 @@ test("#161: a streamed resource interruption is a failed exchange with complete
470
766
  usage: { prompt_tokens: 7, completion_tokens: 5, total_tokens: 12 },
471
767
  },
472
768
  ]);
473
- const provider = new AiSdkProvider({
769
+ const provider = testProvider({
474
770
  ...injectedBase,
475
771
  retryAttempts: 2,
476
772
  rawBody: true,
@@ -489,12 +785,10 @@ test("#161: a streamed resource interruption is a failed exchange with complete
489
785
  assert.equal(error.attempt?.assistant.content, "partial answer");
490
786
  assert.equal(error.attempt?.assistant.reasoning, "partial thought");
491
787
  assert.equal(error.attempt?.assistant.finishReason, "resource_interrupted");
492
- assert.deepEqual(error.attempt?.assistant.usage, {
493
- prompt: 7,
494
- completion: 2,
495
- reasoning: 3,
496
- cached: 0,
497
- total: 12,
788
+ assert.deepEqual(error.accounting[0]?.usage, {
789
+ inputTokens: 7,
790
+ outputTokens: 5,
791
+ totalTokens: 12,
498
792
  });
499
793
  assert.equal(
500
794
  (error.attempt?.assistantRaw as { rawFinishReason?: string }).rawFinishReason,
@@ -517,7 +811,7 @@ test("#161: a buffered resource interruption preserves the successful wire respo
517
811
  usage: { prompt_tokens: 7, completion_tokens: 5, total_tokens: 12 },
518
812
  };
519
813
  const calls = installFetchJson(wire);
520
- const provider = new AiSdkProvider({
814
+ const provider = testProvider({
521
815
  ...injectedBase,
522
816
  streaming: false,
523
817
  retryAttempts: 2,
@@ -546,37 +840,37 @@ test("#161: a buffered resource interruption preserves the successful wire respo
546
840
  test("generate translates a backend cap synonym to canonical length", async () => {
547
841
  // gemini shouts MAX_TOKENS, anthropic says max_tokens -- both must reach core as
548
842
  // "length" so its truncation check (=== "length") is a cross-backend invariant.
549
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
843
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
550
844
  installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "MAX_TOKENS" }] }]);
551
845
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
552
846
  assert.equal(assistant.finishReason, "length");
553
847
  });
554
848
 
555
849
  test("generate translates end_turn to canonical stop", async () => {
556
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
850
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
557
851
  installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "end_turn" }] }]);
558
852
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
559
853
  assert.equal(assistant.finishReason, "stop");
560
854
  });
561
855
 
562
856
  test("generate translates xAI completed to canonical stop", async () => {
563
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
857
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
564
858
  installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "completed" }] }]);
565
859
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
566
860
  assert.equal(assistant.finishReason, "stop");
567
861
  });
568
862
 
569
863
  test("generate aggregates reasoning deltas under multiple field names", async () => {
570
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
864
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
571
865
  installFetch([{ choices: [{ delta: { reasoning_content: "be", thinking: "cause" } }] }]);
572
866
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
573
867
  assert.equal(assistant.reasoning, "because");
574
868
  assert.equal("reasoningEncrypted" in assistant, false); // open reasoning only -> field absent
575
869
  });
576
870
 
577
- test("{§provider-tagged-reasoning} explicit think-tags projects one streamed leading envelope and reclassifies usage", async () => {
871
+ test("{§provider-tagged-reasoning} explicit think-tags project content without estimating token attribution", async () => {
578
872
  const config = { ...injectedBase, reasoningResponseStyle: "think-tags" as const, rawBody: true };
579
- const p = new AiSdkProvider(config);
873
+ const p = testProvider(config);
580
874
  installFetch([
581
875
  { choices: [{ delta: { content: "<thi" } }] },
582
876
  { choices: [{ delta: { content: "nk>12345</th" } }] },
@@ -588,12 +882,10 @@ test("{§provider-tagged-reasoning} explicit think-tags projects one streamed le
588
882
 
589
883
  assert.equal(response.assistant.reasoning, "12345");
590
884
  assert.equal(response.assistant.content, "abcde");
591
- assert.deepEqual(response.assistant.usage, {
592
- prompt: 3,
593
- completion: 5,
594
- reasoning: 5,
595
- cached: 0,
596
- total: 13,
885
+ assert.deepEqual(response.accounting[0]?.usage, {
886
+ inputTokens: 3,
887
+ outputTokens: 10,
888
+ totalTokens: 13,
597
889
  });
598
890
  assert.deepEqual(
599
891
  ((response.assistantRaw as { content: string; reasoning: string }).content),
@@ -611,12 +903,12 @@ test("{§provider-tagged-reasoning} explicit think-tags projects one buffered le
611
903
  usage: { prompt_tokens: 3, completion_tokens: 10, total_tokens: 13 },
612
904
  });
613
905
  const config = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
614
- const response = await new AiSdkProvider(config).generate({ workerId: "tagged-buffer", messages: [] });
906
+ const response = await testProvider(config).generate({ workerId: "tagged-buffer", messages: [] });
615
907
 
616
908
  assert.equal(response.assistant.reasoning, "12345");
617
909
  assert.equal(response.assistant.content, "abcde");
618
- assert.equal(response.assistant.usage.completion, 5);
619
- assert.equal(response.assistant.usage.reasoning, 5);
910
+ assert.equal(response.accounting[0]?.usage?.outputTokens, 10);
911
+ assert.equal(response.accounting[0]?.usage?.outputTokenDetails, undefined);
620
912
  });
621
913
 
622
914
  test("{§provider-tagged-reasoning} tagged text does not overwrite itemized reasoning usage", async () => {
@@ -631,12 +923,14 @@ test("{§provider-tagged-reasoning} tagged text does not overwrite itemized reas
631
923
  },
632
924
  });
633
925
  const config = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
634
- const response = await new AiSdkProvider(config).generate({ workerId: "tagged-itemized", messages: [] });
926
+ const response = await testProvider(config).generate({ workerId: "tagged-itemized", messages: [] });
635
927
 
636
928
  assert.equal(response.assistant.reasoning, "12345");
637
929
  assert.equal(response.assistant.content, "abcde");
638
- assert.equal(response.assistant.usage.completion, 7);
639
- assert.equal(response.assistant.usage.reasoning, 3);
930
+ assert.deepEqual(response.accounting[0]?.usage?.outputTokenDetails, {
931
+ textTokens: 7,
932
+ reasoningTokens: 3,
933
+ });
640
934
  });
641
935
 
642
936
  test("{§provider-tagged-reasoning} an unclosed capped envelope is wholly reasoning in streamed and buffered responses", async () => {
@@ -645,15 +939,13 @@ test("{§provider-tagged-reasoning} an unclosed capped envelope is wholly reason
645
939
  { choices: [{ delta: { content: "<think>unfinished" }, finish_reason: "length" }] },
646
940
  { usage: { prompt_tokens: 3, completion_tokens: 8, total_tokens: 11 } },
647
941
  ]);
648
- const streamed = await new AiSdkProvider(config).generate({ workerId: "tagged-capped-stream", messages: [] });
942
+ const streamed = await testProvider(config).generate({ workerId: "tagged-capped-stream", messages: [] });
649
943
  assert.equal(streamed.assistant.reasoning, "unfinished");
650
944
  assert.equal(streamed.assistant.content, "");
651
- assert.deepEqual(streamed.assistant.usage, {
652
- prompt: 3,
653
- completion: 0,
654
- reasoning: 8,
655
- cached: 0,
656
- total: 11,
945
+ assert.deepEqual(streamed.accounting[0]?.usage, {
946
+ inputTokens: 3,
947
+ outputTokens: 8,
948
+ totalTokens: 11,
657
949
  });
658
950
 
659
951
  mock.restoreAll();
@@ -663,11 +955,11 @@ test("{§provider-tagged-reasoning} an unclosed capped envelope is wholly reason
663
955
  usage: { prompt_tokens: 3, completion_tokens: 8, total_tokens: 11 },
664
956
  });
665
957
  const bufferedConfig = { ...config, streaming: false };
666
- const buffered = await new AiSdkProvider(bufferedConfig).generate({ workerId: "tagged-capped-buffer", messages: [] });
958
+ const buffered = await testProvider(bufferedConfig).generate({ workerId: "tagged-capped-buffer", messages: [] });
667
959
  assert.equal(buffered.assistant.reasoning, "unfinished");
668
960
  assert.equal(buffered.assistant.content, "");
669
- assert.equal(buffered.assistant.usage.completion, 0);
670
- assert.equal(buffered.assistant.usage.reasoning, 8);
961
+ assert.equal(buffered.accounting[0]?.usage?.outputTokens, 8);
962
+ assert.equal(buffered.accounting[0]?.usage?.outputTokenDetails, undefined);
671
963
  });
672
964
 
673
965
  test("{§provider-tagged-reasoning} verbatim, non-leading, and structured-reasoning controls preserve literal tags", async () => {
@@ -676,11 +968,11 @@ test("{§provider-tagged-reasoning} verbatim, non-leading, and structured-reason
676
968
  choices: [{ message: { content: "<think>literal</think>answer" }, finish_reason: "stop" }],
677
969
  usage: { prompt_tokens: 1, completion_tokens: 4, total_tokens: 5 },
678
970
  });
679
- const verbatim = await new AiSdkProvider({ ...injectedBase, streaming: false })
971
+ const verbatim = await testProvider({ ...injectedBase, streaming: false })
680
972
  .generate({ workerId: "verbatim", messages: [] });
681
973
  assert.equal(verbatim.assistant.content, "<think>literal</think>answer");
682
974
  assert.equal(verbatim.assistant.reasoning, null);
683
- assert.equal(verbatim.assistant.usage.completion, 4);
975
+ assert.equal(verbatim.accounting[0]?.usage?.outputTokens, 4);
684
976
 
685
977
  mock.restoreAll();
686
978
  installFetchJson({
@@ -689,7 +981,7 @@ test("{§provider-tagged-reasoning} verbatim, non-leading, and structured-reason
689
981
  usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 },
690
982
  });
691
983
  const taggedConfig = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
692
- const nonLeading = await new AiSdkProvider(taggedConfig)
984
+ const nonLeading = await testProvider(taggedConfig)
693
985
  .generate({ workerId: "non-leading", messages: [] });
694
986
  assert.equal(nonLeading.assistant.content, "show <think>literal</think> exactly");
695
987
  assert.equal(nonLeading.assistant.reasoning, null);
@@ -703,14 +995,14 @@ test("{§provider-tagged-reasoning} verbatim, non-leading, and structured-reason
703
995
  }, finish_reason: "stop" }],
704
996
  usage: { prompt_tokens: 1, completion_tokens: 7, total_tokens: 8 },
705
997
  });
706
- const structured = await new AiSdkProvider(taggedConfig)
998
+ const structured = await testProvider(taggedConfig)
707
999
  .generate({ workerId: "structured", messages: [] });
708
1000
  assert.equal(structured.assistant.content, "<think>literal visible bytes</think>");
709
1001
  assert.equal(structured.assistant.reasoning, "structured reasoning");
710
1002
  });
711
1003
 
712
1004
  test("{§provider-tagged-reasoning} grammar evidence retains the exact pre-projection tagged sentence", async () => {
713
- const content = "<think>🧠reason</think><<PLAN::PLAN\n<<SEND[200]:done:SEND";
1005
+ const content = "<think>🧠reason</think># PLAN0\n\n## SEND0 [200]\ndone";
714
1006
  const config = {
715
1007
  ...injectedBase,
716
1008
  contextWindow: 640,
@@ -721,14 +1013,14 @@ test("{§provider-tagged-reasoning} grammar evidence retains the exact pre-proje
721
1013
  };
722
1014
  installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
723
1015
 
724
- const response = await new AiSdkProvider(config).generate({
1016
+ const response = await testProvider(config).generate({
725
1017
  workerId: "tagged-grammar",
726
1018
  messages: [],
727
1019
  grammar: `root ::= ${JSON.stringify(content)}`,
728
1020
  });
729
1021
 
730
1022
  assert.equal(response.assistant.reasoning, "🧠reason");
731
- assert.equal(response.assistant.content, "<<PLAN::PLAN\n<<SEND[200]:done:SEND");
1023
+ assert.equal(response.assistant.content, "# PLAN0\n\n## SEND0 [200]\ndone");
732
1024
  assert.deepEqual(response.grammarEvidence, {
733
1025
  input: content,
734
1026
  contentStart: [..."<think>🧠reason</think>"].length,
@@ -745,7 +1037,7 @@ test("encrypted reasoning (non-streamed): encrypted entries normalize and text e
745
1037
  { type: "reasoning.text", text: "never surfaced here" },
746
1038
  ],
747
1039
  }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
748
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
1040
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
749
1041
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
750
1042
  // Wire detail ID is preserved; the assistant-message location supports the
751
1043
  // derived classification but supplies no downstream client entity ID.
@@ -759,7 +1051,7 @@ test("distinct encrypted-reasoning wire ids stay distinct items", async () => {
759
1051
  { type: "reasoning.encrypted", data: "AAA", format: "openai-responses-v1", id: "rs_1" },
760
1052
  { type: "reasoning.encrypted", data: "BBB", format: "openai-responses-v1", id: "rs_2" },
761
1053
  ] }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
762
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
1054
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
763
1055
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
764
1056
  assert.equal(assistant.reasoningEncrypted?.length, 2);
765
1057
  assert.deepEqual(assistant.reasoningEncrypted?.map((i) => i.id), ["rs_1", "rs_2"]);
@@ -769,7 +1061,7 @@ test("assistant-message location classifies encrypted reasoning without inventin
769
1061
  installFetchJson({ model: "m", choices: [{ message: { content: "ok", reasoning_details: [
770
1062
  { type: "reasoning.encrypted", data: "OPAQUE", format: "openai-responses-v1", id: null, index: 0 },
771
1063
  ] }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
772
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
1064
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
773
1065
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
774
1066
  assert.deepEqual(assistant.reasoningEncrypted, [{
775
1067
  id: null,
@@ -779,7 +1071,7 @@ test("assistant-message location classifies encrypted reasoning without inventin
779
1071
  });
780
1072
 
781
1073
  test("encrypted reasoning (streamed): chunked blob concatenates per entry index", async () => {
782
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1074
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
783
1075
  installFetch([
784
1076
  { choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "gAAAA", format: "openai-responses-v1", id: "rs_1", index: 0 }] } }] },
785
1077
  { choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "BqXYZ", id: "rs_1", index: 0 }] } }] },
@@ -790,32 +1082,35 @@ test("encrypted reasoning (streamed): chunked blob concatenates per entry index"
790
1082
  assert.equal(assistant.content, "4");
791
1083
  });
792
1084
 
793
- test("reasoningStyle 'think' gates on budget != 0 (magnitude irrelevant for native)", async () => {
794
- const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
1085
+ test("reasoningStyle 'think' follows activation (magnitude is irrelevant to the boolean wire control)", async () => {
1086
+ const on = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
795
1087
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
796
1088
  await on.generate({ workerId: "r", messages: [] });
797
1089
  assert.equal(JSON.parse(calls[0].init.body as string).think, true);
798
1090
 
799
1091
  mock.restoreAll();
800
- const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
1092
+ const off = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
801
1093
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
802
1094
  await off.generate({ workerId: "r", messages: [] });
803
1095
  assert.equal("think" in JSON.parse(calls[0].init.body as string), false);
804
1096
  });
805
1097
 
806
- test("reasoningStyle 'effort' sends a reasoning_effort tier from the budget", async () => {
807
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: 5000 }, retryAttempts: 0, reasoningStyle: "effort" });
808
- const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
809
- await p.generate({ workerId: "r", messages: [] });
810
- assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, "high");
1098
+ test("reasoningStyle 'effort' enables at the portable default without inventing a budget", async () => {
1099
+ for (const [budget, expected] of [[null, "medium"], [5000, "high"]] as const) {
1100
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget }, retryAttempts: 0, reasoningStyle: "effort" });
1101
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1102
+ await p.generate({ workerId: "r", messages: [] });
1103
+ assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, expected);
1104
+ mock.restoreAll();
1105
+ }
811
1106
  });
812
1107
 
813
1108
  test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS, on sends the tier", async () => {
814
1109
  // expected === null → the field must be ABSENT from the wire body. Fireworks
815
1110
  // 400s reasoning_effort='adaptive' for non-MiniMax models (wire-verified,
816
1111
  // Adaptive = the backend's own default posture = omission.
817
- for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "on", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "on"; budget: number | null }, string | null]>) {
818
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
1112
+ for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "on", budget: null }, "medium"], [{ mode: "on", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "on"; budget: number | null }, string | null]>) {
1113
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
819
1114
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
820
1115
  await p.generate({ workerId: "r", messages: [] });
821
1116
  const body = JSON.parse(calls[0].init.body as string);
@@ -829,10 +1124,11 @@ test("{§deepseek-reasoning-request} #157: thinking_effort maps the complete Dee
829
1124
  const cases = [
830
1125
  [{ mode: "off", budget: null }, { thinking: { type: "disabled" } }],
831
1126
  [{ mode: "adaptive", budget: null }, {}],
1127
+ [{ mode: "on", budget: null }, { thinking: { type: "enabled" } }],
832
1128
  [{ mode: "on", budget: 5000 }, { thinking: { type: "enabled" }, reasoning_effort: "high" }],
833
1129
  ] as const;
834
1130
  for (const [reasoning, expected] of cases) {
835
- const p = new AiSdkProvider({
1131
+ const p = testProvider({
836
1132
  model: "m",
837
1133
  url: "http://x/v1/chat/completions",
838
1134
  fetchTimeoutMs: 5000,
@@ -858,7 +1154,7 @@ test("{§deepseek-reasoning-request} #157: thinking_effort maps the complete Dee
858
1154
  });
859
1155
 
860
1156
  test("the family temperature default rides every request; caller sampling overrides it", async () => {
861
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1157
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
862
1158
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
863
1159
  await p.generate({ workerId: "r", messages: [] });
864
1160
  assert.equal(JSON.parse(calls[0].init.body as string).temperature, 0.2);
@@ -877,7 +1173,7 @@ test("the family temperature default rides every request; caller sampling overri
877
1173
  test("DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box default; never on cloud", async () => {
878
1174
  const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off" as const, budget: null }, retryAttempts: 0 };
879
1175
  // set + llamacpp -> the loop-breakers ride the wire
880
- const p = new AiSdkProvider({ ...base, grammarStyle: "llamacpp", dryMultiplier: 0.8, dryBase: 1.75, dryAllowedLength: 2, repeatLastN: 512 });
1176
+ const p = testProvider({ ...base, grammarStyle: "llamacpp", dryMultiplier: 0.8, dryBase: 1.75, dryAllowedLength: 2, repeatLastN: 512 });
881
1177
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
882
1178
  await p.generate({ workerId: "r", messages: [] });
883
1179
  let body = JSON.parse(calls[0].init.body as string);
@@ -888,7 +1184,7 @@ test("DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box
888
1184
  assert.equal(body.repeat_penalty, 1.15); // repeat_penalty always rides the llamacpp path
889
1185
  mock.restoreAll();
890
1186
  // unset -> no dry_*/repeat_last_n on the wire (box keeps its own defaults)
891
- const p2 = new AiSdkProvider({ ...base, grammarStyle: "llamacpp" });
1187
+ const p2 = testProvider({ ...base, grammarStyle: "llamacpp" });
892
1188
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
893
1189
  await p2.generate({ workerId: "r", messages: [] });
894
1190
  body = JSON.parse(calls[0].init.body as string);
@@ -896,7 +1192,7 @@ test("DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box
896
1192
  assert.equal("repeat_last_n" in body, false);
897
1193
  mock.restoreAll();
898
1194
  // DRY is a llama.cpp sampler: a cloud ("none") provider never emits it, even if configured
899
- const p3 = new AiSdkProvider({ ...base, grammarStyle: "none", dryMultiplier: 0.8, repeatLastN: 512 });
1195
+ const p3 = testProvider({ ...base, grammarStyle: "none", dryMultiplier: 0.8, repeatLastN: 512 });
900
1196
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
901
1197
  await p3.generate({ workerId: "r", messages: [] });
902
1198
  body = JSON.parse(calls[0].init.body as string);
@@ -906,7 +1202,7 @@ test("DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box
906
1202
  });
907
1203
 
908
1204
  test("llamacpp grammar path: temperature default + the managed repeat-penalty floor", async () => {
909
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
1205
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
910
1206
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
911
1207
  await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
912
1208
  const body = JSON.parse(calls[0].init.body as string);
@@ -916,13 +1212,13 @@ test("llamacpp grammar path: temperature default + the managed repeat-penalty fl
916
1212
 
917
1213
  test("the repeat penalty rides every request rail-off, keyed per backend", async () => {
918
1214
  // llama.cpp with NO grammar carries its key too (unconstrained local is guarded)
919
- const llama = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
1215
+ const llama = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
920
1216
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
921
1217
  await llama.generate({ workerId: "r", messages: [] });
922
1218
  assert.equal(JSON.parse(calls[0].init.body as string).repeat_penalty, 1.15);
923
1219
  mock.restoreAll();
924
1220
  // A `none`-style cloud backend with a frequency penalty gets frequency_penalty.
925
- const cloud = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1221
+ const cloud = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
926
1222
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
927
1223
  await cloud.generate({ workerId: "r", messages: [] });
928
1224
  const cloudBody = JSON.parse(calls[0].init.body as string);
@@ -931,14 +1227,14 @@ test("the repeat penalty rides every request rail-off, keyed per backend", async
931
1227
  assert.equal("repeat_penalty" in cloudBody, false);
932
1228
  mock.restoreAll();
933
1229
  // frequencyPenalty unset (default 0) opts out cleanly - sends nothing (an out-of-date plugin runs unguarded, never breaks)
934
- const bare = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1230
+ const bare = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
935
1231
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
936
1232
  await bare.generate({ workerId: "r", messages: [] });
937
1233
  assert.equal("frequency_penalty" in JSON.parse(calls[0].init.body as string), false);
938
1234
  });
939
1235
 
940
1236
  test("sampling passthrough forwards caller params; managed + reserved keys win", async () => {
941
- const p = new AiSdkProvider({ model: "managed-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1237
+ const p = testProvider({ model: "managed-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
942
1238
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
943
1239
  await p.generate({
944
1240
  workerId: "r",
@@ -961,7 +1257,7 @@ test("sampling passthrough forwards caller params; managed + reserved keys win",
961
1257
  });
962
1258
 
963
1259
  test("sampling passthrough guards contract invariants: n/tools/caps stripped, platform knobs pass", async () => {
964
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1260
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
965
1261
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
966
1262
  await p.generate({
967
1263
  workerId: "r",
@@ -986,7 +1282,7 @@ test("sampling passthrough guards contract invariants: n/tools/caps stripped, pl
986
1282
  });
987
1283
 
988
1284
  test("template reasoning returns the exact pre-projection grammar sentence ({§gbnf-response-observation})", async () => {
989
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1285
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
990
1286
  const grammarInput = "<|channel>thought\ncon🙂sider<channel|>x";
991
1287
  const calls = installFetch([{ choices: [{ delta: { content: grammarInput } }] }]);
992
1288
  const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(grammarInput)}` });
@@ -1005,8 +1301,24 @@ test("template reasoning returns the exact pre-projection grammar sentence ({§g
1005
1301
  assert.equal(res.meta?.railsVerdict, undefined, "the provider represents evidence but does not grade itself");
1006
1302
  });
1007
1303
 
1304
+ test("template reasoning projects a leading think envelope without losing grammar evidence", async () => {
1305
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1306
+ const input = "<think>\ncon🙂sider</think>x";
1307
+ const calls = installFetch([{ choices: [{ delta: { content: input } }] }]);
1308
+ const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(input)}` });
1309
+ const body = JSON.parse(calls[0].init.body as string);
1310
+ assert.equal(body.reasoning_format, "none");
1311
+ assert.equal(res.assistant.reasoning, "con🙂sider");
1312
+ assert.equal(res.assistant.content, "x");
1313
+ assert.deepEqual(res.grammarEvidence, {
1314
+ input,
1315
+ contentStart: [..."<think>\ncon🙂sider</think>"].length,
1316
+ transported: true,
1317
+ });
1318
+ });
1319
+
1008
1320
  test("a verbatim template response remains exact evidence when it has no channel envelope", async () => {
1009
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1321
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1010
1322
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1011
1323
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
1012
1324
  const body = JSON.parse(calls[0].init.body as string);
@@ -1015,7 +1327,7 @@ test("a verbatim template response remains exact evidence when it has no channel
1015
1327
  });
1016
1328
 
1017
1329
  test("a template grammar preserves exact evidence when reasoning is disabled", async () => {
1018
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1330
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1019
1331
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1020
1332
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
1021
1333
  const body = JSON.parse(calls[0].init.body as string);
@@ -1025,14 +1337,14 @@ test("a template grammar preserves exact evidence when reasoning is disabled", a
1025
1337
  });
1026
1338
 
1027
1339
  test("an unexpectedly projected template response cannot claim pre-projection evidence", async () => {
1028
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1340
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1029
1341
  installFetch([{ choices: [{ delta: { reasoning_content: "reason", content: "x" } }] }]);
1030
1342
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
1031
1343
  assert.equal(res.grammarEvidence, undefined);
1032
1344
  });
1033
1345
 
1034
1346
  test("template reasoning preserves an empty grammar-required channel as exact evidence", async () => {
1035
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1347
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1036
1348
  const input = "<|channel>thought\n<channel|>x";
1037
1349
  const calls = installFetch([{ choices: [{ delta: { content: input } }] }]);
1038
1350
  const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(input)}` });
@@ -1063,16 +1375,16 @@ test("channel-escape detector: billed completion tokens vastly beyond visible ch
1063
1375
  }
1064
1376
  return new Response(sseStream(chunks), { status: 200 });
1065
1377
  };
1066
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetch, tokenizeUrl: "http://x/tokenize", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1378
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetch, tokenizeUrl: "http://x/tokenize", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1067
1379
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
1068
1380
  const escape = res.notices?.find((e) => e.message.includes("escaped the grammar"));
1069
1381
  assert.ok(escape, "escape notice attached");
1070
1382
  assert.equal(escape!.kind, "grammar_unenforced");
1071
- assert.match(escape!.message ?? "", /5000 completion tokens billed/);
1383
+ assert.match(escape!.message ?? "", /5000 output tokens billed/);
1072
1384
  });
1073
1385
 
1074
1386
  test("channel-escape state is absent without a transported grammar", async () => {
1075
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1387
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1076
1388
  installFetch([
1077
1389
  { choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
1078
1390
  { usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
@@ -1083,7 +1395,7 @@ test("channel-escape state is absent without a transported grammar", async () =>
1083
1395
  });
1084
1396
 
1085
1397
  test("reasoningStyle 'template' sends llama-server activation, parser, and response-wide allowance", async () => {
1086
- const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
1398
+ const on = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
1087
1399
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1088
1400
  await on.generate({ workerId: "r", messages: [] });
1089
1401
  let body = JSON.parse(calls[0].init.body as string);
@@ -1092,7 +1404,7 @@ test("reasoningStyle 'template' sends llama-server activation, parser, and respo
1092
1404
  assert.equal(body.thinking_budget_tokens, 64);
1093
1405
 
1094
1406
  mock.restoreAll();
1095
- const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
1407
+ const off = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
1096
1408
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1097
1409
  await off.generate({ workerId: "r", messages: [] });
1098
1410
  body = JSON.parse(calls[0].init.body as string);
@@ -1103,33 +1415,54 @@ test("reasoningStyle 'template' sends llama-server activation, parser, and respo
1103
1415
 
1104
1416
  test("reasoningStyle 'template' explicit budget tightens the reserve and cannot exceed it", async () => {
1105
1417
  const base = { model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 } as const, completionReserve: { tokens: 160 } as const, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryAttempts: 0, reasoningStyle: "template" as const };
1106
- const p = new AiSdkProvider({ ...base, reasoning: { mode: "on", budget: 32 } });
1418
+ const p = testProvider({ ...base, reasoning: { mode: "on", budget: 32 } });
1107
1419
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1108
1420
  await p.generate({ workerId: "r", messages: [], sampling: { thinking_budget_tokens: 999, reasoning_format: "none" } });
1109
1421
  const body = JSON.parse(calls[0].init.body as string);
1110
1422
  assert.equal(body.thinking_budget_tokens, 32);
1111
1423
  assert.equal(body.reasoning_format, "auto");
1112
1424
  assert.throws(
1113
- () => new AiSdkProvider({ ...base, reasoning: { mode: "on", budget: 65 } }),
1425
+ () => testProvider({ ...base, reasoning: { mode: "on", budget: 65 } }),
1114
1426
  /REASONING_BUDGET \(65\) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE \(64\)/,
1115
1427
  );
1116
1428
  });
1117
1429
 
1118
- test("budget 0 suppresses effort and include_reasoning", async () => {
1119
- const effort = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "effort" });
1430
+ test("reasoningStyle 'template' explicit activation without a budget uses the resolved reserve", async () => {
1431
+ const p = testProvider({
1432
+ model: "m",
1433
+ url: "http://x/v1/chat/completions",
1434
+ contextWindow: 640,
1435
+ reasoningReserve: { tokens: 64 },
1436
+ completionReserve: { tokens: 160 },
1437
+ fetchTimeoutMs: 5000,
1438
+ temperature: 0.2,
1439
+ repeatPenalty: 1.15,
1440
+ reasoning: { mode: "on", budget: null },
1441
+ retryAttempts: 0,
1442
+ reasoningStyle: "template",
1443
+ });
1444
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1445
+ await p.generate({ workerId: "r", messages: [] });
1446
+ const body = JSON.parse(calls[0].init.body as string);
1447
+ assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true });
1448
+ assert.equal(body.thinking_budget_tokens, 64);
1449
+ });
1450
+
1451
+ test("reasoning off suppresses effort and include_reasoning controls", async () => {
1452
+ const effort = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "effort" });
1120
1453
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1121
1454
  await effort.generate({ workerId: "r", messages: [] });
1122
1455
  assert.equal("reasoning_effort" in JSON.parse(calls[0].init.body as string), false);
1123
1456
 
1124
1457
  mock.restoreAll();
1125
- const relay = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
1458
+ const relay = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
1126
1459
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1127
1460
  await relay.generate({ workerId: "r", messages: [] });
1128
1461
  assert.equal("include_reasoning" in JSON.parse(calls[0].init.body as string), false);
1129
1462
  });
1130
1463
 
1131
1464
  test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", async () => {
1132
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
1465
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
1133
1466
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1134
1467
  await p.generate({ workerId: "r", messages: [] });
1135
1468
  assert.equal(JSON.parse(calls[0].init.body as string).include_reasoning, true);
@@ -1138,7 +1471,7 @@ test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", asy
1138
1471
  // — grammar-constrained sampling —
1139
1472
 
1140
1473
  test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor", async () => {
1141
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
1474
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
1142
1475
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1143
1476
  await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
1144
1477
  const body = JSON.parse(calls[0].init.body as string);
@@ -1148,7 +1481,7 @@ test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor
1148
1481
  });
1149
1482
 
1150
1483
  test("grammar transport 'none' (default): the grammar is never sent — no silent unconstrained", async () => {
1151
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1484
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1152
1485
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1153
1486
  await p.generate({ workerId: "r", messages: [], grammar: "root ::= statement" });
1154
1487
  const body = JSON.parse(calls[0].init.body as string);
@@ -1158,7 +1491,7 @@ test("grammar transport 'none' (default): the grammar is never sent — no silen
1158
1491
 
1159
1492
  // — exact pre-projection grammar evidence ({§gbnf-response-observation}) —
1160
1493
 
1161
- const grammarProvider = () => new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", source: "provider:test" });
1494
+ const grammarProvider = () => testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", source: "provider:test" });
1162
1495
  const streamingContent = (content: string) => installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
1163
1496
 
1164
1497
  test("an unsplit grammar response carries the exact observed sentence", async () => {
@@ -1192,7 +1525,7 @@ test("empty unsplit content remains exact grammar evidence", async () => {
1192
1525
  });
1193
1526
 
1194
1527
  test("grammarStyle 'none' produces no grammar observation", async () => {
1195
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // grammarStyle defaults to "none"
1528
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // grammarStyle defaults to "none"
1196
1529
  streamingContent("anything goes");
1197
1530
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
1198
1531
  assert.equal(res.assistant.content, "anything goes");
@@ -1211,7 +1544,7 @@ test("provider evidence does not depend on the local validator understanding the
1211
1544
  // — PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar, withhold it, and preserve the observation —
1212
1545
 
1213
1546
  test("gbnfDebug marks an unconstrained observation as not transported", async () => {
1214
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
1547
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
1215
1548
  const calls = installFetch([{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }]);
1216
1549
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
1217
1550
  const body = JSON.parse(calls[0].init.body as string);
@@ -1223,7 +1556,7 @@ test("gbnfDebug marks an unconstrained observation as not transported", async ()
1223
1556
  });
1224
1557
 
1225
1558
  test("gbnfDebug preserves conflicting bytes without a provider verdict", async () => {
1226
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
1559
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
1227
1560
  const calls = installFetch([{ choices: [{ delta: { content: "xon-conforming output" }, finish_reason: "stop" }] }]);
1228
1561
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
1229
1562
  assert.equal(res.assistant.content, "xon-conforming output");
@@ -1234,7 +1567,7 @@ test("gbnfDebug preserves conflicting bytes without a provider verdict", async (
1234
1567
  });
1235
1568
 
1236
1569
  test("gbnfDebug: an INVALID grammar throws before any wire call — it never reaches the model", async () => {
1237
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true });
1570
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true });
1238
1571
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1239
1572
  await assert.rejects(
1240
1573
  () => p.generate({ workerId: "r", messages: [], grammar: 'foo ::= "a"' }), // no `root` rule → invalid GBNF
@@ -1246,7 +1579,7 @@ test("gbnfDebug: an INVALID grammar throws before any wire call — it never rea
1246
1579
  // — meta bag: verbatim provider metadata —
1247
1580
 
1248
1581
  test("meta: passes backend fields through without reinterpreting monetary values", async () => {
1249
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
1582
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
1250
1583
  const balance = { amount: "0.0000042", currency: "XMR" };
1251
1584
  installFetchJson({ ...jsonChoice, balance, system_fingerprint: "fp_abc" });
1252
1585
  const res = await p.generate({ workerId: "r", messages: [] });
@@ -1260,15 +1593,34 @@ const headerVal = (init: RequestInit, name: string): string | undefined =>
1260
1593
  new Headers(init.headers).get(name) ?? undefined;
1261
1594
 
1262
1595
  test("firstPartyMetadata: attributions + client ride as Plurnk-* headers", async () => {
1263
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1596
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1264
1597
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1265
1598
  await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0", "@foo/y@0.3.1"], client: "plurnk.nvim/1.4.0" });
1266
1599
  assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), '["@acme/x@1.2.0","@foo/y@0.3.1"]');
1267
1600
  assert.equal(headerVal(calls[0].init, "Plurnk-Client"), "plurnk.nvim/1.4.0");
1268
1601
  });
1269
1602
 
1603
+ test("Plurnk-Call-Kind carries the caller's emission or bare output contract under the first-party gate", async () => {
1604
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1605
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1606
+ await p.generate({ workerId: "emission", messages: [], callKind: "emission" });
1607
+ await p.generate({ workerId: "bare", messages: [], callKind: "bare" });
1608
+ assert.equal(headerVal(calls[0].init, "Plurnk-Call-Kind"), "emission");
1609
+ assert.equal(headerVal(calls[1].init, "Plurnk-Call-Kind"), "bare");
1610
+ });
1611
+
1612
+ test("generate rejects an unknown call kind before provider I/O", async () => {
1613
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1614
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1615
+ await assert.rejects(
1616
+ p.generate({ workerId: "invalid", messages: [], callKind: "unknown" as never }),
1617
+ /unsupported callKind "unknown"/,
1618
+ );
1619
+ assert.equal(calls.length, 0);
1620
+ });
1621
+
1270
1622
  test("Plurnk-Worker-Primary: the lineage root rides under the gate; emitted even when it equals workerId", async () => {
1271
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1623
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1272
1624
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1273
1625
  await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
1274
1626
  assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), "w-root"); // a descendant: Primary != Worker-Id
@@ -1287,22 +1639,23 @@ test("Plurnk-Worker-Primary: the lineage root rides under the gate; emitted even
1287
1639
  });
1288
1640
 
1289
1641
  test("Plurnk-Worker-Primary is structurally dropped when firstPartyMetadata is off", async () => {
1290
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1642
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1291
1643
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1292
1644
  await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
1293
1645
  assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), undefined); // never reaches a third-party backend
1294
1646
  });
1295
1647
 
1296
1648
  test("firstPartyMetadata off (default): the headers are structurally dropped even when values are passed", async () => {
1297
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1649
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1298
1650
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1299
- await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0"], client: "plurnk-cli/2.0.0" });
1651
+ await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0"], client: "plurnk-cli/2.0.0", callKind: "bare" });
1300
1652
  assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), undefined); // never leaks to a non-first-party backend
1301
1653
  assert.equal(headerVal(calls[0].init, "Plurnk-Client"), undefined);
1654
+ assert.equal(headerVal(calls[0].init, "Plurnk-Call-Kind"), undefined);
1302
1655
  });
1303
1656
 
1304
1657
  test("firstPartyMetadata on but empty values: no header emitted", async () => {
1305
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1658
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1306
1659
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1307
1660
  await p.generate({ workerId: "r", messages: [], attributions: [], client: "" });
1308
1661
  assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), undefined);
@@ -1310,7 +1663,7 @@ test("firstPartyMetadata on but empty values: no header emitted", async () => {
1310
1663
  });
1311
1664
 
1312
1665
  test("grammar transport: no grammar passed sends no grammar field, but the penalty rides", async () => {
1313
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
1666
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
1314
1667
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1315
1668
  await p.generate({ workerId: "r", messages: [] });
1316
1669
  const body = JSON.parse(calls[0].init.body as string);
@@ -1319,7 +1672,7 @@ test("grammar transport: no grammar passed sends no grammar field, but the penal
1319
1672
  });
1320
1673
 
1321
1674
  test("maxTokens transports as max_tokens; absent → no wire field (server default)", async () => {
1322
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1675
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1323
1676
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1324
1677
  await p.generate({ workerId: "r", messages: [], maxTokens: 2048 });
1325
1678
  assert.equal(JSON.parse(calls[0].init.body as string).max_tokens, 2048);
@@ -1331,7 +1684,7 @@ test("maxTokens transports as max_tokens; absent → no wire field (server defau
1331
1684
  });
1332
1685
 
1333
1686
  test("slot affinity is internal: sticky per workerId, distinct workers spread across slots", async () => {
1334
- const pinning = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
1687
+ const pinning = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
1335
1688
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1336
1689
  await pinning.generate({ workerId: "run-A", messages: [] });
1337
1690
  await pinning.generate({ workerId: "run-B", messages: [] });
@@ -1342,20 +1695,20 @@ test("slot affinity is internal: sticky per workerId, distinct workers spread ac
1342
1695
  });
1343
1696
 
1344
1697
  test("slot affinity: no pinning backend or unknown slotCount → no id_slot ever", async () => {
1345
- const cloud = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // default: no pinning
1698
+ const cloud = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // default: no pinning
1346
1699
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1347
1700
  await cloud.generate({ workerId: "run-A", messages: [] });
1348
1701
  assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
1349
1702
 
1350
1703
  mock.restoreAll();
1351
- const noCount = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true }); // slotCount null
1704
+ const noCount = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true }); // slotCount null
1352
1705
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1353
1706
  await noCount.generate({ workerId: "run-A", messages: [] });
1354
1707
  assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
1355
1708
  });
1356
1709
 
1357
1710
  test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; recent workers stay sticky", async () => {
1358
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
1711
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
1359
1712
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1360
1713
  const slotOf = (i: number) => JSON.parse(calls[i].init.body as string).id_slot;
1361
1714
  for (let i = 0; i < 16; i++) await p.generate({ workerId: `r${i}`, messages: [] }); // fills the 16-entry window {r0..r15}
@@ -1369,7 +1722,7 @@ test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; r
1369
1722
 
1370
1723
  test("streaming:false: a non-ok response rejects as a classified ProviderError (covers the non-streamed transport)", async () => {
1371
1724
  const { ProviderError } = await import("./errors.ts");
1372
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, source: "provider:test" });
1725
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, source: "provider:test" });
1373
1726
  mock.method(globalThis, "fetch", async () => new Response("boom", { status: 500 }));
1374
1727
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
1375
1728
  assert.ok(err instanceof ProviderError, `expected ProviderError, got ${String(err)}`);
@@ -1380,14 +1733,14 @@ test("streaming:false: a non-ok response rejects as a classified ProviderError (
1380
1733
  });
1381
1734
 
1382
1735
  test("generate fail-hards on a missing or empty workerId", async () => {
1383
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1736
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1384
1737
  installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1385
1738
  await assert.rejects(() => p.generate({ workerId: "", messages: [] }), /workerId is required/);
1386
1739
  await assert.rejects(() => (p.generate as (a: object) => Promise<unknown>)({ messages: [] }), /workerId is required/);
1387
1740
  });
1388
1741
 
1389
1742
  test("messages pass through verbatim — the provider injects no turn (PLAN lives in the grammar, never a provider prefill)", async () => {
1390
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1743
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1391
1744
  const calls = installFetch([{ choices: [{ delta: { content: "out" } }] }]);
1392
1745
  const input = [{ role: "user" as const, content: "hi" }];
1393
1746
  const res = await p.generate({ workerId: "r", messages: input });
@@ -1397,7 +1750,7 @@ test("messages pass through verbatim — the provider injects no turn (PLAN live
1397
1750
 
1398
1751
  test("generate wraps an HTTP failure as a ProviderError carrying Problem Details", async () => {
1399
1752
  const { ProviderError } = await import("./errors.ts");
1400
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, source: "provider:test" });
1753
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, source: "provider:test" });
1401
1754
  mock.method(globalThis, "fetch", async () => new Response("rate limited", { status: 429 }));
1402
1755
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
1403
1756
  assert.ok(err instanceof ProviderError, `expected ProviderError, got ${String(err)}`);
@@ -1411,14 +1764,14 @@ test("generate wraps an HTTP failure as a ProviderError carrying Problem Details
1411
1764
  });
1412
1765
 
1413
1766
  test("generate rejects on a pre-aborted external signal", async () => {
1414
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1767
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1415
1768
  installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1416
1769
  const signal = AbortSignal.abort(new Error("nope"));
1417
1770
  await assert.rejects(() => p.generate({ workerId: "r", messages: [], signal }));
1418
1771
  });
1419
1772
 
1420
1773
  test("configured headers and url are sent verbatim", async () => {
1421
- const p = new AiSdkProvider({
1774
+ const p = testProvider({
1422
1775
  model: "m", url: "http://host/custom/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
1423
1776
  headers: { Authorization: "Bearer secret", "X-Title": "plurnk" },
1424
1777
  });
@@ -1445,14 +1798,16 @@ const stalledStreamResponse = (): Response => new Response(new ReadableStream({
1445
1798
 
1446
1799
  test("retry: a transient failure retries and a later success resolves", async () => {
1447
1800
  const calls = installFetchScript([
1801
+ { status: 408, retryAfter: 0 },
1802
+ { status: 409, retryAfter: 0 },
1448
1803
  { status: 429, retryAfter: 0 },
1449
1804
  { status: 503, retryAfter: 0 },
1450
1805
  { status: 200, chunks: [{ choices: [{ delta: { content: "ok" } }] }] },
1451
1806
  ]);
1452
- const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 3 });
1807
+ const p = testProvider({ ...retryCfg, retryAttempts: 4 });
1453
1808
  const res = await p.generate({ workerId: "r", messages: [] });
1454
1809
  assert.equal(res.assistant.content, "ok");
1455
- assert.equal(calls.length, 3); // 429 → 503 → 200
1810
+ assert.equal(calls.length, 5); // 408 → 409 → 429 → 503 → 200
1456
1811
  });
1457
1812
 
1458
1813
  test("streamed-body silence retries and returns the retry's complete output", async () => {
@@ -1469,7 +1824,7 @@ test("streamed-body silence retries and returns the retry's complete output", as
1469
1824
  },
1470
1825
  }), { status: 200 });
1471
1826
  });
1472
- const p = new AiSdkProvider({
1827
+ const p = testProvider({
1473
1828
  model: "m",
1474
1829
  url: "http://x/v1/chat/completions",
1475
1830
  fetchTimeoutMs: 5000,
@@ -1492,7 +1847,7 @@ test("streamed-body silence does not replay when retries are disabled", async ()
1492
1847
  calls++;
1493
1848
  return stalledStreamResponse();
1494
1849
  });
1495
- const p = new AiSdkProvider({
1850
+ const p = testProvider({
1496
1851
  model: "m",
1497
1852
  url: "http://x/v1/chat/completions",
1498
1853
  fetchTimeoutMs: 1000,
@@ -1506,7 +1861,8 @@ test("streamed-body silence does not replay when retries are disabled", async ()
1506
1861
  await assert.rejects(
1507
1862
  p.generate({ workerId: "r", messages: [] }),
1508
1863
  (error: ProviderError) => error.kind === "network_failure"
1509
- && /chunk timeout/i.test(error.message),
1864
+ && error.problem.timeoutPhase === "stream_idle"
1865
+ && error.problem.timeoutMs === 10,
1510
1866
  );
1511
1867
  assert.equal(calls, 1, "zero retries permits exactly one provider request");
1512
1868
  mock.restoreAll();
@@ -1518,7 +1874,7 @@ test("streamed-body silence exhausts the configured retry budget once", async ()
1518
1874
  calls++;
1519
1875
  return stalledStreamResponse();
1520
1876
  });
1521
- const p = new AiSdkProvider({
1877
+ const p = testProvider({
1522
1878
  model: "m",
1523
1879
  url: "http://x/v1/chat/completions",
1524
1880
  fetchTimeoutMs: 5000,
@@ -1540,6 +1896,116 @@ test("streamed-body silence exhausts the configured retry budget once", async ()
1540
1896
  mock.restoreAll();
1541
1897
  });
1542
1898
 
1899
+ test("an attempt timeout retries within the larger operation deadline and settles every physical request", async () => {
1900
+ let calls = 0;
1901
+ mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
1902
+ calls++;
1903
+ if (calls > 1) {
1904
+ return new Response(sseStream([
1905
+ { choices: [{ delta: { content: "recovered" }, finish_reason: "stop" }] },
1906
+ ]), { status: 200 });
1907
+ }
1908
+ return await new Promise<Response>((_resolve, reject) => {
1909
+ const signal = init?.signal;
1910
+ signal?.addEventListener("abort", () => reject(signal.reason), { once: true });
1911
+ });
1912
+ });
1913
+ const connectivity = { operationTimeoutMs: 5_000 };
1914
+ const settled: Array<{ outcome: string }> = [];
1915
+ const p = testProvider({
1916
+ model: "m",
1917
+ url: "http://x/v1/chat/completions",
1918
+ fetchTimeoutMs: 10,
1919
+ streamIdleTimeoutMs: 0,
1920
+ temperature: 0.2,
1921
+ repeatPenalty: 1.15,
1922
+ reasoning: { mode: "off", budget: null },
1923
+ retryAttempts: 1,
1924
+ source: "provider:test",
1925
+ ...connectivity,
1926
+ });
1927
+ const result = await p.generate({
1928
+ workerId: "r",
1929
+ messages: [],
1930
+ observeRequest: async () => async (accounting) => { settled.push(accounting); },
1931
+ });
1932
+ assert.equal(result.assistant.content, "recovered");
1933
+ assert.equal(calls, 2);
1934
+ assert.deepEqual(settled.map(({ outcome }) => outcome), ["error", "response"]);
1935
+ assert.deepEqual(result.accounting.map(({ outcome }) => outcome), ["error", "response"]);
1936
+ mock.restoreAll();
1937
+ });
1938
+
1939
+ test("first-content silence retries independently of the stream-idle deadline", async () => {
1940
+ let calls = 0;
1941
+ mock.method(globalThis, "fetch", async () => {
1942
+ calls++;
1943
+ if (calls > 1) {
1944
+ return new Response(sseStream([
1945
+ { choices: [{ delta: { content: "recovered" }, finish_reason: "stop" }] },
1946
+ ]), { status: 200 });
1947
+ }
1948
+ return new Response(new ReadableStream({
1949
+ start(controller) {
1950
+ setTimeout(() => controller.close(), 100);
1951
+ },
1952
+ }), { status: 200 });
1953
+ });
1954
+ const connectivity = { operationTimeoutMs: 5_000, firstContentTimeoutMs: 10 };
1955
+ const p = testProvider({
1956
+ model: "m",
1957
+ url: "http://x/v1/chat/completions",
1958
+ fetchTimeoutMs: 5_000,
1959
+ streamIdleTimeoutMs: 0,
1960
+ temperature: 0.2,
1961
+ repeatPenalty: 1.15,
1962
+ reasoning: { mode: "off", budget: null },
1963
+ retryAttempts: 1,
1964
+ source: "provider:test",
1965
+ ...connectivity,
1966
+ });
1967
+ const result = await p.generate({ workerId: "r", messages: [] });
1968
+ assert.equal(result.assistant.content, "recovered");
1969
+ assert.equal(calls, 2);
1970
+ mock.restoreAll();
1971
+ });
1972
+
1973
+ test("operation-deadline exhaustion is a distinct non-retryable failure", async () => {
1974
+ let calls = 0;
1975
+ mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
1976
+ calls++;
1977
+ return await new Promise<Response>((_resolve, reject) => {
1978
+ const signal = init?.signal;
1979
+ signal?.addEventListener("abort", () => reject(signal.reason), { once: true });
1980
+ });
1981
+ });
1982
+ const connectivity = { operationTimeoutMs: 10 };
1983
+ const p = testProvider({
1984
+ model: "m",
1985
+ url: "http://x/v1/chat/completions",
1986
+ fetchTimeoutMs: 50,
1987
+ streamIdleTimeoutMs: 0,
1988
+ temperature: 0.2,
1989
+ repeatPenalty: 1.15,
1990
+ reasoning: { mode: "off", budget: null },
1991
+ retryAttempts: 3,
1992
+ source: "provider:test",
1993
+ ...connectivity,
1994
+ });
1995
+ await assert.rejects(
1996
+ p.generate({ workerId: "r", messages: [] }),
1997
+ (error: ProviderError) => error.kind === "deadline_exceeded"
1998
+ && error.status === 504
1999
+ && error.problem.retryable === false
2000
+ && error.problem.timeoutPhase === "operation"
2001
+ && error.problem.timeoutMs === 10
2002
+ && error.accounting.length === 1
2003
+ && error.accounting[0]?.outcome === "error",
2004
+ );
2005
+ assert.equal(calls, 1);
2006
+ mock.restoreAll();
2007
+ });
2008
+
1543
2009
  test("the total generation deadline spans stalled-stream retry scheduling", async () => {
1544
2010
  let calls = 0;
1545
2011
  mock.method(globalThis, "fetch", async () => {
@@ -1556,10 +2022,11 @@ test("the total generation deadline spans stalled-stream retry scheduling", asyn
1556
2022
  }
1557
2023
  return stalledStreamResponse();
1558
2024
  });
1559
- const p = new AiSdkProvider({
2025
+ const p = testProvider({
1560
2026
  model: "m",
1561
2027
  url: "http://x/v1/chat/completions",
1562
- fetchTimeoutMs: 50,
2028
+ fetchTimeoutMs: 5000,
2029
+ operationTimeoutMs: 50,
1563
2030
  streamIdleTimeoutMs: 10,
1564
2031
  temperature: 0.2,
1565
2032
  repeatPenalty: 1.15,
@@ -1570,7 +2037,8 @@ test("the total generation deadline spans stalled-stream retry scheduling", asyn
1570
2037
  const started = Date.now();
1571
2038
  await assert.rejects(
1572
2039
  p.generate({ workerId: "r", messages: [] }),
1573
- (error: ProviderError) => error.kind === "network_failure",
2040
+ (error: ProviderError) => error.kind === "deadline_exceeded"
2041
+ && error.problem.timeoutPhase === "operation",
1574
2042
  );
1575
2043
  assert.ok(Date.now() - started < 500, "the configured total deadline ends retry scheduling");
1576
2044
  assert.equal(calls, 1, "the total deadline expires before another request begins");
@@ -1586,7 +2054,7 @@ test("a zero stream-idle timeout permits a slow inter-chunk pause", async () =>
1586
2054
  controller.close();
1587
2055
  },
1588
2056
  }), { status: 200 }));
1589
- const p = new AiSdkProvider({
2057
+ const p = testProvider({
1590
2058
  model: "m",
1591
2059
  url: "http://x/v1/chat/completions",
1592
2060
  fetchTimeoutMs: 1000,
@@ -1604,7 +2072,7 @@ test("a zero stream-idle timeout permits a slow inter-chunk pause", async () =>
1604
2072
  test("retry: exhausting the budget surfaces the classified ProviderError", async () => {
1605
2073
  const { ProviderError } = await import("./errors.ts");
1606
2074
  const calls = installFetchScript([{ status: 429, retryAfter: 0 }]); // always rate-limited
1607
- const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 2 });
2075
+ const p = testProvider({ ...retryCfg, retryAttempts: 2 });
1608
2076
  await assert.rejects(
1609
2077
  () => p.generate({ workerId: "r", messages: [] }),
1610
2078
  (err: unknown) => { assert.ok(err instanceof ProviderError); assert.equal(err.kind, "rate_limit"); return true; },
@@ -1617,7 +2085,7 @@ test("retry: a Retry-After HTTP-date is honored — a past date parses to a 0ms
1617
2085
  { status: 503, retryAfter: "Wed, 21 Oct 2015 07:28:00 GMT" }, // date form, in the past → max(0, past−now) = 0
1618
2086
  { status: 200, chunks: [{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }] },
1619
2087
  ]);
1620
- const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 1 });
2088
+ const p = testProvider({ ...retryCfg, retryAttempts: 1 });
1621
2089
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
1622
2090
  assert.equal(assistant.content, "ok");
1623
2091
  assert.equal(calls.length, 2); // initial 503 + one retry, no real wall-clock wait
@@ -1625,14 +2093,14 @@ test("retry: a Retry-After HTTP-date is honored — a past date parses to a 0ms
1625
2093
 
1626
2094
  test("retry: a terminal error (401 unauthorized) is never retried", async () => {
1627
2095
  const calls = installFetchScript([{ status: 401 }]);
1628
- const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 5 });
2096
+ const p = testProvider({ ...retryCfg, retryAttempts: 5 });
1629
2097
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), /401/);
1630
2098
  assert.equal(calls.length, 1); // terminal — no retry despite budget
1631
2099
  });
1632
2100
 
1633
2101
  test("retry: retryAttempts 0 surfaces the first transient failure immediately", async () => {
1634
2102
  const calls = installFetchScript([{ status: 503, retryAfter: 0 }]);
1635
- const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 0 });
2103
+ const p = testProvider({ ...retryCfg, retryAttempts: 0 });
1636
2104
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }));
1637
2105
  assert.equal(calls.length, 1); // no retry budget
1638
2106
  });
@@ -1640,7 +2108,7 @@ test("retry: retryAttempts 0 surfaces the first transient failure immediately",
1640
2108
  test("retry: a caller abort during backoff rejects promptly with no further attempt", async () => {
1641
2109
  const ac = new AbortController();
1642
2110
  const calls = installFetchScript([{ status: 503, retryAfter: 5 }]); // 5s backoff we never wait out
1643
- const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 3 });
2111
+ const p = testProvider({ ...retryCfg, retryAttempts: 3 });
1644
2112
  const promise = p.generate({ workerId: "r", messages: [], signal: ac.signal });
1645
2113
  await flush(); // attempt 0 fails, enters the backoff sleep
1646
2114
  assert.equal(calls.length, 1);
@@ -1651,23 +2119,29 @@ test("retry: a caller abort during backoff rejects promptly with no further atte
1651
2119
 
1652
2120
  // — Anthropic reasoning style (wire `thinking` parameter) —
1653
2121
 
1654
- test("reasoningStyle 'anthropic' maps the budget to the thinking param", async () => {
2122
+ test("reasoningStyle 'anthropic' maps an optional budget or the resolved reserve to the thinking param", async () => {
1655
2123
  // N>0 → enabled with budget_tokens
1656
- const capped = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: 4096 }, reasoningStyle: "anthropic" });
2124
+ const capped = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: 4096 }, reasoningStyle: "anthropic" });
1657
2125
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1658
2126
  await capped.generate({ workerId: "r", messages: [] });
1659
2127
  assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "enabled", budget_tokens: 4096 });
1660
2128
 
2129
+ mock.restoreAll();
2130
+ const unbudgeted = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 8192, reasoningReserve: { tokens: 2048 }, fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: null }, reasoningStyle: "anthropic" });
2131
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
2132
+ await unbudgeted.generate({ workerId: "r", messages: [] });
2133
+ assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "enabled", budget_tokens: 2048 });
2134
+
1661
2135
  mock.restoreAll();
1662
2136
  // 0 → explicit disabled
1663
- const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, reasoningStyle: "anthropic" });
2137
+ const off = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, reasoningStyle: "anthropic" });
1664
2138
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1665
2139
  await off.generate({ workerId: "r", messages: [] });
1666
2140
  assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "disabled" });
1667
2141
 
1668
2142
  mock.restoreAll();
1669
2143
  // -1 adaptive → omit (API default depth)
1670
- const adaptive = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, reasoningStyle: "anthropic" });
2144
+ const adaptive = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, reasoningStyle: "anthropic" });
1671
2145
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1672
2146
  await adaptive.generate({ workerId: "r", messages: [] });
1673
2147
  assert.equal("thinking" in JSON.parse(calls[0].init.body as string), false);
@@ -1685,14 +2159,14 @@ test("streaming:false posts without stream and parses the single JSON response",
1685
2159
  usage: { prompt_tokens: 3, completion_tokens: 1, total_tokens: 4 },
1686
2160
  }), { status: 200, headers: { "Content-Type": "application/json" } });
1687
2161
  });
1688
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
2162
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
1689
2163
  const res = await p.generate({ workerId: "r", messages: [] });
1690
2164
  const sent = JSON.parse(calls[0].body);
1691
2165
  assert.equal("stream" in sent, false); // no streaming flag
1692
2166
  assert.equal(res.assistant.content, "hello"); // content from message.content
1693
2167
  assert.equal(res.assistant.reasoning, "because"); // reasoning_content mapped
1694
2168
  assert.equal(res.assistant.finishReason, "stop");
1695
- assert.equal(res.assistant.usage.total, 4);
2169
+ assert.equal(res.accounting[0]?.usage?.totalTokens, 4);
1696
2170
  mock.restoreAll();
1697
2171
  });
1698
2172
 
@@ -1701,7 +2175,7 @@ const captureBase = { model: "m", url: "http://x/v1/chat/completions", fetchTime
1701
2175
 
1702
2176
  test("logprobs OFF by default: no wire request, no assistant.logprobs, no rawBody", async () => {
1703
2177
  const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
1704
- const p = new AiSdkProvider({ ...captureBase });
2178
+ const p = testProvider({ ...captureBase });
1705
2179
  const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
1706
2180
  const body = JSON.parse((calls[0].init.body as string));
1707
2181
  assert.equal("logprobs" in body, false);
@@ -1718,7 +2192,7 @@ test("logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw logpr
1718
2192
  { token: "no", logprob: -0.1, sampling_logprob: -0.1, top_logprobs: [{ token: "no", logprob: -0.1 }] },
1719
2193
  ] } }] };
1720
2194
  const calls = installFetch([chunk]);
1721
- const p = new AiSdkProvider({ ...captureBase, topLogprobs: 2 });
2195
+ const p = testProvider({ ...captureBase, topLogprobs: 2 });
1722
2196
  const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
1723
2197
  const body = JSON.parse((calls[0].init.body as string));
1724
2198
  assert.equal(body.logprobs, true);
@@ -1732,7 +2206,7 @@ test("logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw logpr
1732
2206
  test("rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob preserved", async () => {
1733
2207
  const wire = { model: "m", extra_top_level: "kept", choices: [{ message: { content: "no" }, finish_reason: "stop", logprobs: { content: [{ token: "no", logprob: -0.1, sampling_logprob: -0.1, token_id: 42 }] } }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } };
1734
2208
  installFetchJson(wire);
1735
- const p = new AiSdkProvider({ ...captureBase, streaming: false, topLogprobs: 0, rawBody: true });
2209
+ const p = testProvider({ ...captureBase, streaming: false, topLogprobs: 0, rawBody: true });
1736
2210
  const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
1737
2211
  assert.deepEqual(res.rawBody, wire); // verbatim
1738
2212
  assert.equal((res.rawBody as typeof wire).choices[0].logprobs.content[0].sampling_logprob, -0.1);
@@ -1743,7 +2217,7 @@ test("rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob prese
1743
2217
 
1744
2218
  test("caller sampling cannot forge logprobs (reserved keys): the env flag is the only control", async () => {
1745
2219
  const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
1746
- const p = new AiSdkProvider({ ...captureBase }); // logprobs OFF
2220
+ const p = testProvider({ ...captureBase }); // logprobs OFF
1747
2221
  await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }], sampling: { logprobs: true, top_logprobs: 5 } });
1748
2222
  const body = JSON.parse((calls[0].init.body as string));
1749
2223
  assert.equal("logprobs" in body, false); // sampling passthrough stripped it
@@ -1754,7 +2228,7 @@ test("caller sampling cannot forge logprobs (reserved keys): the env flag is the
1754
2228
  // — turn coordinate headers ({§lifecycle-terms}): same gate as every first-party signal —
1755
2229
 
1756
2230
  test("workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under the first-party gate", async () => {
1757
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
2231
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1758
2232
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1759
2233
  await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
1760
2234
  const headers = new Headers(calls[0].init.headers);
@@ -1764,7 +2238,7 @@ test("workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under the firs
1764
2238
  });
1765
2239
 
1766
2240
  test("third-party providers structurally DROP the coordinate (gate off by default)", async () => {
1767
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
2241
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1768
2242
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1769
2243
  await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
1770
2244
  const headers = new Headers(calls[0].init.headers);
@@ -1774,7 +2248,7 @@ test("third-party providers structurally DROP the coordinate (gate off by defaul
1774
2248
  });
1775
2249
 
1776
2250
  test("coordinates are 1-based — 0/absent/empty emit no header", async () => {
1777
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
2251
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1778
2252
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1779
2253
  await p.generate({ workerId: "r", messages: [], workspaceId: "", loop: 0, turn: 0 });
1780
2254
  const headers = new Headers(calls[0].init.headers);
@@ -1788,18 +2262,18 @@ test("coordinates are 1-based — 0/absent/empty emit no header", async () => {
1788
2262
 
1789
2263
  test("reserves derive from the detected window; absolutes stand alone; null window + percent = no claim", () => {
1790
2264
  const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
1791
- const derived = new AiSdkProvider({ ...base, contextWindow: 49152, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 } });
2265
+ const derived = testProvider({ ...base, contextWindow: 49152, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 } });
1792
2266
  assert.equal(derived.reasoningReserve, 4915); // jennifer/turboderp: 10% of 49152
1793
2267
  assert.equal(derived.completionReserve, 12288); // 25% of 49152
1794
- const pinned = new AiSdkProvider({ ...base, contextWindow: null, reasoningReserve: { tokens: 4096 }, completionReserve: { percent: 0.25 } });
2268
+ const pinned = testProvider({ ...base, contextWindow: null, reasoningReserve: { tokens: 4096 }, completionReserve: { percent: 0.25 } });
1795
2269
  assert.equal(pinned.reasoningReserve, 4096); // absolute pin needs no window
1796
2270
  assert.equal(pinned.completionReserve, null); // percent without a window = underivable
1797
- const legacy = new AiSdkProvider({ ...base, contextWindow: 49152 });
2271
+ const legacy = testProvider({ ...base, contextWindow: 49152 });
1798
2272
  assert.equal(legacy.reasoningReserve, null); // out-of-date sibling: no claim
1799
2273
  });
1800
2274
 
1801
2275
  test("router-owned tuning: tuningFloors:false drops the temperature/penalty floors, caller sampling still rides", async () => {
1802
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0, tuningFloors: false });
2276
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0, tuningFloors: false });
1803
2277
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1804
2278
  await p.generate({ workerId: "r", messages: [], sampling: { temperature: 0.9 } });
1805
2279
  const body = JSON.parse(calls[0].init.body as string);
@@ -1807,25 +2281,148 @@ test("router-owned tuning: tuningFloors:false drops the temperature/penalty floo
1807
2281
  assert.equal("frequency_penalty" in body, false); // the floor is suppressed; the router owns tuning
1808
2282
  });
1809
2283
 
1810
- // -- prompt-cache affinity (workerId -> prompt_cache_key) --
2284
+ // -- {§provider-cache-affinity} / {§provider-cache-write-policy} --
1811
2285
 
1812
- test("promptCacheKey on: body sends prompt_cache_key = workerId (serverless replica affinity)", async () => {
1813
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
2286
+ test("a compatible route's declared body affinity is managed by workerId", async () => {
2287
+ const p = testProvider({
2288
+ model: "m",
2289
+ url: "http://x/v1/chat/completions",
2290
+ fetchTimeoutMs: 5000,
2291
+ temperature: 0.2,
2292
+ repeatPenalty: 1.15,
2293
+ reasoning: { mode: "off", budget: null },
2294
+ retryAttempts: 0,
2295
+ cacheAffinity: { target: "body", name: "prompt_cache_key" },
2296
+ });
1814
2297
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1815
- await p.generate({ workerId: "worker-abc", messages: [] });
2298
+ await p.generate({ workerId: "worker-abc", messages: [], sampling: { prompt_cache_key: "hijack" } });
1816
2299
  assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc");
1817
2300
  });
1818
2301
 
1819
- test("promptCacheKey off (default): no prompt_cache_key on the wire", async () => {
1820
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
2302
+ test("an undeclared compatible route receives no guessed cache field", async () => {
2303
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1821
2304
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1822
2305
  await p.generate({ workerId: "worker-abc", messages: [] });
1823
2306
  assert.equal("prompt_cache_key" in JSON.parse(calls[0].init.body as string), false);
1824
2307
  });
1825
2308
 
1826
- test("prompt_cache_key is managed: caller sampling cannot forge/override the affinity key", async () => {
1827
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
2309
+ test("a compatible route's declared header affinity composes with static headers", async () => {
2310
+ const p = testProvider({
2311
+ model: "m",
2312
+ url: "http://x/v1/chat/completions",
2313
+ headers: { Authorization: "Bearer key" },
2314
+ fetchTimeoutMs: 5000,
2315
+ temperature: 0.2,
2316
+ repeatPenalty: 1.15,
2317
+ reasoning: { mode: "off", budget: null },
2318
+ retryAttempts: 0,
2319
+ cacheAffinity: { target: "header", name: "x-grok-conv-id" },
2320
+ });
1828
2321
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1829
- await p.generate({ workerId: "worker-abc", messages: [], sampling: { prompt_cache_key: "hijack" } });
1830
- assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc"); // managed wins
2322
+ await p.generate({ workerId: "worker-abc", messages: [] });
2323
+ const headers = new Headers(calls[0].init.headers);
2324
+ assert.equal(headers.get("authorization"), "Bearer key");
2325
+ assert.equal(headers.get("x-grok-conv-id"), "worker-abc");
2326
+ });
2327
+
2328
+ test("native request projections compose reasoning visibility, affinity, and system cache control", async () => {
2329
+ let request: Record<string, unknown> | undefined;
2330
+ const usage = {
2331
+ inputTokens: { total: 2, noCache: 2, cacheRead: 0, cacheWrite: 0 },
2332
+ outputTokens: { total: 1, text: 1, reasoning: 0 },
2333
+ };
2334
+ const languageModel = {
2335
+ specificationVersion: "v4",
2336
+ provider: "native.test",
2337
+ modelId: "native-cache",
2338
+ supportedUrls: {},
2339
+ doGenerate: async (options: Record<string, unknown>) => {
2340
+ request = options;
2341
+ return {
2342
+ content: [{ type: "text", text: "ok" }],
2343
+ finishReason: { unified: "stop", raw: "stop" },
2344
+ usage,
2345
+ response: { id: "response", modelId: "native-cache" },
2346
+ warnings: [],
2347
+ };
2348
+ },
2349
+ doStream: async () => { throw new Error("streaming is not under test"); },
2350
+ } as unknown as LanguageModel;
2351
+ const p = testProvider({
2352
+ model: "native-cache",
2353
+ languageModel,
2354
+ fetchTimeoutMs: 5000,
2355
+ temperature: 0.2,
2356
+ repeatPenalty: 1.15,
2357
+ reasoning: { mode: "adaptive", budget: null },
2358
+ retryAttempts: 0,
2359
+ streaming: false,
2360
+ cacheAffinity: { target: "provider-option", provider: "openai", name: "promptCacheKey" },
2361
+ reasoningResponseProviderOptions: {
2362
+ google: { thinkingConfig: { includeThoughts: true } },
2363
+ },
2364
+ systemCacheProviderOptions: {
2365
+ anthropic: { cacheControl: { type: "ephemeral" } },
2366
+ },
2367
+ });
2368
+ await p.generate({
2369
+ workerId: "worker-native",
2370
+ messages: [
2371
+ { role: "system", content: "stable definition" },
2372
+ { role: "system", content: "stable policy" },
2373
+ { role: "user", content: "changing packet" },
2374
+ ],
2375
+ });
2376
+
2377
+ assert.deepEqual(request?.providerOptions, {
2378
+ google: { thinkingConfig: { includeThoughts: true } },
2379
+ openai: { promptCacheKey: "worker-native" },
2380
+ });
2381
+ assert.deepEqual(request?.prompt, [
2382
+ { role: "system", content: "stable definition", providerOptions: undefined },
2383
+ {
2384
+ role: "system",
2385
+ content: "stable policy",
2386
+ providerOptions: { anthropic: { cacheControl: { type: "ephemeral" } } },
2387
+ },
2388
+ { role: "user", content: [{ type: "text", text: "changing packet" }], providerOptions: undefined },
2389
+ ]);
2390
+ });
2391
+
2392
+ test("native AI SDK reasoning turns on without an operator token budget", async () => {
2393
+ let request: Record<string, unknown> | undefined;
2394
+ const languageModel = {
2395
+ specificationVersion: "v4",
2396
+ provider: "native.test",
2397
+ modelId: "native-reasoning",
2398
+ supportedUrls: {},
2399
+ doGenerate: async (options: Record<string, unknown>) => {
2400
+ request = options;
2401
+ return {
2402
+ content: [{ type: "reasoning", text: "consider" }, { type: "text", text: "ok" }],
2403
+ finishReason: { unified: "stop", raw: "stop" },
2404
+ usage: {
2405
+ inputTokens: { total: 2, noCache: 2, cacheRead: 0, cacheWrite: 0 },
2406
+ outputTokens: { total: 2, text: 1, reasoning: 1 },
2407
+ },
2408
+ response: { id: "response", modelId: "native-reasoning" },
2409
+ warnings: [],
2410
+ };
2411
+ },
2412
+ doStream: async () => { throw new Error("streaming is not under test"); },
2413
+ } as unknown as LanguageModel;
2414
+ const p = testProvider({
2415
+ model: "native-reasoning",
2416
+ languageModel,
2417
+ fetchTimeoutMs: 5000,
2418
+ temperature: 0.2,
2419
+ repeatPenalty: 1.15,
2420
+ reasoning: { mode: "on", budget: null },
2421
+ retryAttempts: 0,
2422
+ streaming: false,
2423
+ });
2424
+ const response = await p.generate({ workerId: "worker-native", messages: [{ role: "user", content: "hello" }] });
2425
+
2426
+ assert.equal(request?.reasoning, "medium");
2427
+ assert.equal(response.assistant.reasoning, "consider");
1831
2428
  });