@plurnk/plurnk-providers 1.4.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. package/.env.defaults +40 -34
  2. package/README.md +3 -0
  3. package/SPEC.md +153 -62
  4. package/dist/AiSdkProvider.d.ts +19 -25
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +353 -120
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +7 -13
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +36 -8
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +2 -21
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +19 -14
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/accounting.d.ts +6 -0
  17. package/dist/accounting.d.ts.map +1 -0
  18. package/dist/accounting.js +168 -0
  19. package/dist/accounting.js.map +1 -0
  20. package/dist/aiSdkTransport.d.ts +11 -3
  21. package/dist/aiSdkTransport.d.ts.map +1 -1
  22. package/dist/aiSdkTransport.js +198 -29
  23. package/dist/aiSdkTransport.js.map +1 -1
  24. package/dist/catalogProvider.d.ts +7 -2
  25. package/dist/catalogProvider.d.ts.map +1 -1
  26. package/dist/catalogProvider.js +32 -26
  27. package/dist/catalogProvider.js.map +1 -1
  28. package/dist/compatibleProvider.d.ts.map +1 -1
  29. package/dist/compatibleProvider.js +18 -7
  30. package/dist/compatibleProvider.js.map +1 -1
  31. package/dist/cost.d.ts +10 -10
  32. package/dist/cost.d.ts.map +1 -1
  33. package/dist/cost.js +88 -43
  34. package/dist/cost.js.map +1 -1
  35. package/dist/env.d.ts +5 -7
  36. package/dist/env.d.ts.map +1 -1
  37. package/dist/env.js +30 -32
  38. package/dist/env.js.map +1 -1
  39. package/dist/errors.d.ts +14 -2
  40. package/dist/errors.d.ts.map +1 -1
  41. package/dist/errors.js +60 -2
  42. package/dist/errors.js.map +1 -1
  43. package/dist/index.d.ts +4 -4
  44. package/dist/index.d.ts.map +1 -1
  45. package/dist/index.js +3 -2
  46. package/dist/index.js.map +1 -1
  47. package/dist/ollama.js +3 -3
  48. package/dist/ollama.js.map +1 -1
  49. package/dist/sdkModels.d.ts +6 -0
  50. package/dist/sdkModels.d.ts.map +1 -1
  51. package/dist/sdkModels.js +46 -3
  52. package/dist/sdkModels.js.map +1 -1
  53. package/dist/types.d.ts +40 -29
  54. package/dist/types.d.ts.map +1 -1
  55. package/dist/usage.d.ts +21 -4
  56. package/dist/usage.d.ts.map +1 -1
  57. package/dist/usage.js +188 -74
  58. package/dist/usage.js.map +1 -1
  59. package/package.json +9 -7
  60. package/src/AiSdkProvider.test.ts +1039 -182
  61. package/src/AiSdkProvider.ts +428 -141
  62. package/src/Mock.test.ts +37 -12
  63. package/src/Mock.ts +46 -12
  64. package/src/Pool.test.ts +19 -6
  65. package/src/Pool.ts +20 -16
  66. package/src/ProviderRegistry.test.ts +16 -11
  67. package/src/accounting.test.ts +94 -0
  68. package/src/accounting.ts +190 -0
  69. package/src/aiSdkTransport.test.ts +42 -49
  70. package/src/aiSdkTransport.ts +218 -32
  71. package/src/boundaries.test.ts +2 -0
  72. package/src/catalogProvider.test.ts +271 -24
  73. package/src/catalogProvider.ts +44 -28
  74. package/src/compatibleProvider.test.ts +6 -3
  75. package/src/compatibleProvider.ts +20 -7
  76. package/src/cost.test.ts +55 -35
  77. package/src/cost.ts +110 -54
  78. package/src/defaults.test.ts +13 -3
  79. package/src/env.test.ts +50 -26
  80. package/src/env.ts +43 -42
  81. package/src/errors.test.ts +47 -2
  82. package/src/errors.ts +68 -3
  83. package/src/index.ts +21 -5
  84. package/src/ollama.test.ts +4 -1
  85. package/src/ollama.ts +3 -3
  86. package/src/sdkModels.test.ts +94 -3
  87. package/src/sdkModels.ts +53 -3
  88. package/src/types.ts +91 -33
  89. package/src/usage.test.ts +112 -108
  90. package/src/usage.ts +233 -84
@@ -1,7 +1,25 @@
1
1
  import test, { mock } from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
- import AiSdkProvider, { effortFromBudget } from "./AiSdkProvider.ts";
3
+ import AiSdkProvider, { effortFromBudget, type AiSdkProviderConfig } from "./AiSdkProvider.ts";
4
4
  import { ProviderError } from "./errors.ts";
5
+ import { providerCostNormalizer } from "./accounting.ts";
6
+ import type { LanguageModel } from "ai";
7
+
8
+ type TestProviderConfig = Omit<AiSdkProviderConfig, "operationTimeoutMs" | "firstContentTimeoutMs">
9
+ & Partial<Pick<AiSdkProviderConfig, "operationTimeoutMs" | "firstContentTimeoutMs">>;
10
+
11
+ const testProvider = (config: TestProviderConfig): AiSdkProvider => {
12
+ const {
13
+ operationTimeoutMs = config.fetchTimeoutMs,
14
+ firstContentTimeoutMs = 0,
15
+ ...rest
16
+ } = config;
17
+ return new AiSdkProvider({
18
+ ...rest,
19
+ operationTimeoutMs,
20
+ firstContentTimeoutMs,
21
+ });
22
+ };
5
23
 
6
24
  // Build a fake fetch returning a one-chunk SSE stream, capturing the request
7
25
  // so tests can assert what the spine sent on the wire.
@@ -59,6 +77,31 @@ const installFetchJson = (payload: unknown) => {
59
77
  return calls;
60
78
  };
61
79
 
80
+ const settledCharge = {
81
+ kind: "charged",
82
+ amount: { amount: "0.00000042", currency: "XMR" },
83
+ usdEquivalent: "0.000071",
84
+ source: "plurnk endpoint settlement",
85
+ } as const;
86
+
87
+ const billedErrorBody = {
88
+ status: 422,
89
+ error: {
90
+ message: "non-conforming emission rejected",
91
+ type: "grammar_invalid",
92
+ },
93
+ usage: {
94
+ prompt_tokens: 8,
95
+ completion_tokens: 3,
96
+ reasoning_tokens: 0,
97
+ prompt_tokens_details: { cached_tokens: 2 },
98
+ total_tokens: 11,
99
+ },
100
+ charge: settledCharge,
101
+ };
102
+
103
+ const directCost = ({ charge }: { charge?: unknown }) => charge as typeof settledCharge | undefined;
104
+
62
105
  const jsonChoice = { model: "m", choices: [{ message: { content: "x" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } };
63
106
 
64
107
  const injectedBase = {
@@ -88,9 +131,9 @@ test("per-instance fetch owns streaming and buffered requests", async () => {
88
131
  }), { status: 200, headers: { "Content-Type": "application/json" } });
89
132
  };
90
133
 
91
- const streamed = await new AiSdkProvider({ ...injectedBase, fetch: streamingFetch, rawBody: true })
134
+ const streamed = await testProvider({ ...injectedBase, fetch: streamingFetch, rawBody: true })
92
135
  .generate({ workerId: "stream", messages: [{ role: "user", content: "hello" }] });
93
- const buffered = await new AiSdkProvider({ ...injectedBase, fetch: bufferedFetch, streaming: false })
136
+ const buffered = await testProvider({ ...injectedBase, fetch: bufferedFetch, streaming: false })
94
137
  .generate({ workerId: "buffer", messages: [{ role: "user", content: "hello" }] });
95
138
 
96
139
  assert.equal(streamed.assistant.content, "streamed");
@@ -114,12 +157,17 @@ test("caller cancellation and provider timeout reach an injected fetch", async (
114
157
  });
115
158
  };
116
159
  const caller = new AbortController();
117
- const callerProvider = new AiSdkProvider({ ...injectedBase, fetch: pendingFetch });
160
+ const callerProvider = testProvider({ ...injectedBase, fetch: pendingFetch });
118
161
  const callerRequest = callerProvider.generate({ workerId: "cancel", messages: [], signal: caller.signal });
119
162
  caller.abort(new Error("operator cancelled"));
120
163
  await assert.rejects(callerRequest, /operator cancelled/);
121
164
 
122
- const timeoutProvider = new AiSdkProvider({ ...injectedBase, fetch: pendingFetch, fetchTimeoutMs: 1 });
165
+ const timeoutProvider = testProvider({
166
+ ...injectedBase,
167
+ fetch: pendingFetch,
168
+ fetchTimeoutMs: 1,
169
+ operationTimeoutMs: 100,
170
+ });
123
171
  await assert.rejects(
124
172
  timeoutProvider.generate({ workerId: "timeout", messages: [] }),
125
173
  (error: ProviderError) => error.kind === "network_failure",
@@ -141,7 +189,7 @@ test("per-instance fetch owns tokenization and retry attempts", async () => {
141
189
  { choices: [], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } },
142
190
  ]), { status: 200 });
143
191
  };
144
- const provider = new AiSdkProvider({
192
+ const provider = testProvider({
145
193
  ...injectedBase,
146
194
  fetch: providerFetch,
147
195
  retryAttempts: 1,
@@ -157,6 +205,60 @@ test("per-instance fetch owns tokenization and retry attempts", async () => {
157
205
  ]);
158
206
  });
159
207
 
208
+ test("request-observer open failures preserve the durability cause and issue no provider I/O", async () => {
209
+ const root = new Error("durable request open failed");
210
+ let calls = 0;
211
+ const provider = testProvider({
212
+ ...injectedBase,
213
+ retryAttempts: 3,
214
+ fetch: async () => {
215
+ calls++;
216
+ return new Response(JSON.stringify(jsonChoice), {
217
+ status: 200,
218
+ headers: { "Content-Type": "application/json" },
219
+ });
220
+ },
221
+ streaming: false,
222
+ });
223
+
224
+ await assert.rejects(
225
+ provider.generate({
226
+ workerId: "observer-open",
227
+ messages: [],
228
+ observeRequest: async () => { throw root; },
229
+ }),
230
+ (error: unknown) => error === root,
231
+ );
232
+ assert.equal(calls, 0);
233
+ });
234
+
235
+ test("request-observer settlement failures preserve the durability cause without retrying I/O", async () => {
236
+ const root = new Error("durable request settlement failed");
237
+ let calls = 0;
238
+ const provider = testProvider({
239
+ ...injectedBase,
240
+ retryAttempts: 3,
241
+ fetch: async () => {
242
+ calls++;
243
+ return new Response(JSON.stringify(jsonChoice), {
244
+ status: 200,
245
+ headers: { "Content-Type": "application/json" },
246
+ });
247
+ },
248
+ streaming: false,
249
+ });
250
+
251
+ await assert.rejects(
252
+ provider.generate({
253
+ workerId: "observer-settle",
254
+ messages: [],
255
+ observeRequest: async () => async () => { throw root; },
256
+ }),
257
+ (error: unknown) => error === root,
258
+ );
259
+ assert.equal(calls, 1);
260
+ });
261
+
160
262
  // Sequenced fetch mock for retry tests: each entry is one HTTP response. A 200
161
263
  // streams its chunks; any other status returns that error (with an optional
162
264
  // retry-after header). The last entry repeats once the script runs out.
@@ -204,8 +306,13 @@ test("effortFromBudget: maps budget to tiers", () => {
204
306
 
205
307
  test("a 524 Cloudflare edge timeout fails fast - not retried despite retryAttempts", async () => {
206
308
  const calls = installFetchScript([{ status: 524, retryAfter: 120 }]);
207
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
208
- await assert.rejects(p.generate({ workerId: "r", messages: [] }));
309
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
310
+ await assert.rejects(
311
+ p.generate({ workerId: "r", messages: [] }),
312
+ (error: ProviderError) => error.kind === "network_failure"
313
+ && error.status === 524
314
+ && error.problem.retryable === false,
315
+ );
209
316
  await flush();
210
317
  assert.equal(calls.length, 1); // edge code: one attempt, no retry despite retryAttempts: 3
211
318
  mock.restoreAll();
@@ -214,7 +321,7 @@ test("a 524 Cloudflare edge timeout fails fast - not retried despite retryAttemp
214
321
  test("a 422 grammar_invalid is a failed exchange, not transport replay policy", async () => {
215
322
  const body = JSON.stringify({ error: { message: "non-conforming emission rejected: ...", type: "grammar_invalid" } });
216
323
  const calls = installFetchScript([{ status: 422, body }]);
217
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 2 });
324
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 2 });
218
325
  await assert.rejects(
219
326
  p.generate({ workerId: "r", messages: [] }),
220
327
  (e: unknown) => e instanceof ProviderError && e.kind === "grammar_invalid",
@@ -229,7 +336,7 @@ test("an SSE error frame is a failed exchange, not an empty completion", async (
229
336
  status: 422,
230
337
  error: { message: "non-conforming emission rejected", type: "grammar_invalid" },
231
338
  }]);
232
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
339
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
233
340
  await assert.rejects(
234
341
  p.generate({ workerId: "r", messages: [] }),
235
342
  (e: unknown) => e instanceof ProviderError && e.kind === "grammar_invalid",
@@ -237,29 +344,138 @@ test("an SSE error frame is a failed exchange, not an empty completion", async (
237
344
  assert.equal(calls.length, 1);
238
345
  });
239
346
 
347
+ test("a buffered classified error retains normalized usage and settled charge", async () => {
348
+ const calls = installFetchScript([{ status: 422, body: JSON.stringify(billedErrorBody) }]);
349
+ const p = testProvider({
350
+ ...injectedBase,
351
+ streaming: false,
352
+ normalizeCost: directCost,
353
+ });
354
+ await assert.rejects(
355
+ p.generate({ workerId: "billed-json-error", messages: [] }),
356
+ (error: unknown) => {
357
+ assert.ok(error instanceof ProviderError);
358
+ assert.equal(error.kind, "grammar_invalid");
359
+ assert.deepEqual(error.accounting, [{
360
+ provider: "provider",
361
+ model: "m",
362
+ outcome: "error",
363
+ status: 422,
364
+ usage: {
365
+ inputTokens: 8,
366
+ outputTokens: 3,
367
+ totalTokens: 11,
368
+ inputTokenDetails: { cacheReadTokens: 2 },
369
+ outputTokenDetails: { textTokens: 3, reasoningTokens: 0 },
370
+ },
371
+ cost: settledCharge,
372
+ }]);
373
+ assert.equal(error.attempt, undefined, "accounting evidence does not fabricate an assistant response");
374
+ return true;
375
+ },
376
+ );
377
+ assert.equal(calls.length, 1);
378
+ });
379
+
380
+ test("an SSE classified error retains the same normalized usage and settled charge", async () => {
381
+ const calls = installFetch([billedErrorBody]);
382
+ const p = testProvider({
383
+ ...injectedBase,
384
+ normalizeCost: directCost,
385
+ });
386
+ await assert.rejects(
387
+ p.generate({ workerId: "billed-sse-error", messages: [] }),
388
+ (error: unknown) => {
389
+ assert.ok(error instanceof ProviderError);
390
+ assert.equal(error.kind, "grammar_invalid");
391
+ assert.deepEqual(error.accounting, [{
392
+ provider: "provider",
393
+ model: "m",
394
+ outcome: "error",
395
+ status: 422,
396
+ usage: {
397
+ inputTokens: 8,
398
+ outputTokens: 3,
399
+ totalTokens: 11,
400
+ inputTokenDetails: { cacheReadTokens: 2 },
401
+ outputTokenDetails: { textTokens: 3, reasoningTokens: 0 },
402
+ },
403
+ cost: settledCharge,
404
+ }]);
405
+ assert.equal(error.attempt, undefined);
406
+ return true;
407
+ },
408
+ );
409
+ assert.equal(calls.length, 1);
410
+ });
411
+
412
+ test("a successful response normalizes direct charge without duplicating it as metadata", async () => {
413
+ installFetchJson({ ...jsonChoice, charge: settledCharge });
414
+ const p = testProvider({
415
+ ...injectedBase,
416
+ streaming: false,
417
+ normalizeCost: directCost,
418
+ });
419
+ const response = await p.generate({ workerId: "billed-json-success", messages: [] });
420
+ assert.deepEqual(response.accounting[0]?.cost, settledCharge);
421
+ assert.equal(response.meta?.charge, undefined);
422
+ });
423
+
424
+ test("malformed monetary evidence closes the physical request before surfacing the normalization failure", async () => {
425
+ const root = new TypeError("direct charge is malformed");
426
+ const settled: unknown[] = [];
427
+ const calls = installFetchJson({ ...jsonChoice, charge: { malformed: true } });
428
+ const provider = testProvider({
429
+ ...injectedBase,
430
+ retryAttempts: 3,
431
+ streaming: false,
432
+ normalizeCost: () => { throw root; },
433
+ });
434
+
435
+ await assert.rejects(
436
+ provider.generate({
437
+ workerId: "malformed-charge",
438
+ messages: [],
439
+ observeRequest: async () => async (accounting) => { settled.push(accounting); },
440
+ }),
441
+ (error: unknown) => error === root,
442
+ );
443
+ assert.equal(calls.length, 1);
444
+ assert.deepEqual(settled, [{
445
+ provider: "provider",
446
+ model: "m",
447
+ outcome: "response",
448
+ usage: { inputTokens: 1, outputTokens: 1, totalTokens: 2 },
449
+ cost: {
450
+ kind: "unknown",
451
+ reason: "provider request accounting could not be normalized after physical I/O",
452
+ },
453
+ }]);
454
+ });
455
+
240
456
  test("a trailing eos_token (--special EOG leak) is stripped from content", async () => {
241
457
  installFetchJson({ model: "m", choices: [{ message: { content: "the answer<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
242
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
458
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
243
459
  const res = await p.generate({ workerId: "r", messages: [] });
244
460
  assert.equal(res.assistant.content, "the answer"); // trailing <eos> gone; packet + verdict see clean bytes
245
461
  });
246
462
 
247
463
  test("without a probed eos_token the content passes through untouched", async () => {
248
464
  installFetchJson({ model: "m", choices: [{ message: { content: "keeps <eos> literally" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
249
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
465
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
250
466
  const res = await p.generate({ workerId: "r", messages: [] });
251
467
  assert.equal(res.assistant.content, "keeps <eos> literally"); // no eosText (a cloud backend) -> no strip
252
468
  });
253
469
 
254
470
  test("only the trailing eos_token is stripped; a quoted one mid-body survives", async () => {
255
471
  installFetchJson({ model: "m", choices: [{ message: { content: "quotes <eos> in the body<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 } });
256
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
472
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
257
473
  const res = await p.generate({ workerId: "r", messages: [] });
258
474
  assert.equal(res.assistant.content, "quotes <eos> in the body"); // only the tail goes
259
475
  });
260
476
 
261
477
  test("identity getters and default prompt estimate", async () => {
262
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
478
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
263
479
  assert.equal(p.model, "m");
264
480
  assert.equal(p.contextWindow, null); // default
265
481
  assert.deepEqual(
@@ -272,43 +488,258 @@ test("identity getters and default prompt estimate", async () => {
272
488
  },
273
489
  "chars/2 is explicitly an estimate; high-token-density Unicode prevents an upper-bound claim",
274
490
  );
275
- assert.equal(p.calculateCost({ prompt: 9, completion: 9, reasoning: 0, cached: 0, total: 18 }), 0); // current unknown-rate sentinel
276
491
  });
277
492
 
278
- test("injected prompt measurement preserves provenance and calculateCost is used", async () => {
493
+ test("injected prompt measurement preserves provenance and request cost estimation stays internal", async () => {
279
494
  const seen: string[] = [];
280
- const p = new AiSdkProvider({
495
+ installFetchJson(jsonChoice);
496
+ const p = testProvider({
281
497
  model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
282
498
  countPromptTokens: (messages) => {
283
499
  seen.push(...messages.map(({ content }) => content));
284
500
  return { kind: "upper_bound", tokens: 7, source: "test:proven-bound" };
285
501
  },
286
- calculateCost: (u) => u.total * 2,
502
+ streaming: false,
503
+ estimateCost: (usage) => ({
504
+ kind: "estimated",
505
+ amount: { amount: String((usage?.totalTokens ?? 0) * 2), currency: "USD" },
506
+ source: "test estimator",
507
+ }),
287
508
  });
288
509
  assert.deepEqual(
289
510
  await p.countPromptTokens([{ role: "system", content: "system" }, { role: "user", content: "user" }]),
290
511
  { kind: "upper_bound", tokens: 7, source: "test:proven-bound" },
291
512
  );
292
513
  assert.deepEqual(seen, ["system", "user"]);
293
- assert.equal(p.calculateCost({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 5 }), 10);
514
+ const response = await p.generate({ workerId: "accounted", messages: [] });
515
+ assert.deepEqual(response.accounting[0]?.cost, {
516
+ kind: "estimated",
517
+ amount: { amount: "4", currency: "USD" },
518
+ source: "test estimator",
519
+ });
294
520
  });
295
521
 
296
522
  test("generate maps a streamed response into ProviderResponse", async () => {
297
- const p = new AiSdkProvider({ model: "req-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
523
+ const p = testProvider({ model: "req-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
298
524
  installFetch([
299
525
  { model: "wire-model", choices: [{ delta: { content: "hel" } }] },
300
526
  { choices: [{ delta: { content: "lo" }, finish_reason: "stop" }] },
301
527
  { usage: { prompt_tokens: 3, completion_tokens: 2, total_tokens: 5, cached_tokens: 1 } },
302
528
  ]);
303
- const { assistant, assistantRaw } = await p.generate({ workerId: "r", messages: [{ role: "user", content: "hi" }] });
529
+ const { assistant, assistantRaw, accounting } = await p.generate({ workerId: "r", messages: [{ role: "user", content: "hi" }] });
304
530
  assert.equal(assistant.content, "hello");
305
531
  assert.equal(assistant.model, "wire-model"); // wire-reported wins
306
532
  assert.equal(assistant.finishReason, "stop");
307
- assert.deepEqual(assistant.usage, { prompt: 3, completion: 2, reasoning: 0, cached: 1, total: 5 });
533
+ assert.deepEqual(accounting[0]?.usage, {
534
+ inputTokens: 3,
535
+ outputTokens: 2,
536
+ totalTokens: 5,
537
+ inputTokenDetails: { cacheReadTokens: 1 },
538
+ });
308
539
  assert.equal(assistant.reasoning, null); // none emitted
309
540
  assert.notEqual(assistantRaw, undefined);
310
541
  });
311
542
 
543
+ test("native SDK accounting metadata becomes a normalized charge in buffered and streamed responses", async (t) => {
544
+ const usage = {
545
+ inputTokens: { total: 2, noCache: 2, cacheRead: 0, cacheWrite: 0 },
546
+ outputTokens: { total: 1, text: 1, reasoning: 0 },
547
+ };
548
+ const providerMetadata = { openrouter: { usage: { cost: 0.00154935 } } };
549
+ const charge = {
550
+ kind: "charged",
551
+ amount: { amount: "0.00154935", currency: "USD" },
552
+ source: "OpenRouter response usage.cost",
553
+ };
554
+ const languageModel = {
555
+ specificationVersion: "v4",
556
+ provider: "openrouter.chat",
557
+ modelId: "router-test",
558
+ supportedUrls: {},
559
+ doGenerate: async () => ({
560
+ content: [{ type: "text", text: "ok" }],
561
+ finishReason: { unified: "stop", raw: "completed" },
562
+ usage,
563
+ providerMetadata,
564
+ response: { id: "response-buffered", modelId: "router-test" },
565
+ warnings: [],
566
+ }),
567
+ doStream: async () => ({
568
+ stream: new ReadableStream({
569
+ start(controller) {
570
+ controller.enqueue({ type: "stream-start", warnings: [] });
571
+ controller.enqueue({ type: "response-metadata", id: "response-streamed", modelId: "router-test" });
572
+ controller.enqueue({ type: "text-start", id: "text-1" });
573
+ controller.enqueue({ type: "text-delta", id: "text-1", delta: "ok" });
574
+ controller.enqueue({ type: "text-end", id: "text-1" });
575
+ controller.enqueue({
576
+ type: "finish",
577
+ finishReason: { unified: "stop", raw: "completed" },
578
+ usage,
579
+ providerMetadata,
580
+ });
581
+ controller.close();
582
+ },
583
+ }),
584
+ response: {},
585
+ }),
586
+ } as unknown as LanguageModel;
587
+ const config = {
588
+ model: "router-test",
589
+ languageModel,
590
+ fetchTimeoutMs: 5_000,
591
+ temperature: 0.2,
592
+ repeatPenalty: 1.15,
593
+ reasoning: { mode: "off" as const, budget: null },
594
+ retryAttempts: 0,
595
+ normalizeCost: providerCostNormalizer("@openrouter/ai-sdk-provider"),
596
+ };
597
+
598
+ await t.test("buffered", async () => {
599
+ const response = await testProvider({ ...config, streaming: false })
600
+ .generate({ workerId: "buffered", messages: [] });
601
+ assert.deepEqual(response.accounting[0]?.cost, charge);
602
+ });
603
+ await t.test("streamed", async () => {
604
+ const response = await testProvider(config)
605
+ .generate({ workerId: "streamed", messages: [] });
606
+ assert.deepEqual(response.accounting[0]?.cost, charge);
607
+ });
608
+ });
609
+
610
+ test("native SDK providers share the first-content retry contract", async () => {
611
+ let calls = 0;
612
+ const usage = {
613
+ inputTokens: { total: 1, noCache: 1, cacheRead: 0, cacheWrite: 0 },
614
+ outputTokens: { total: 1, text: 1, reasoning: 0 },
615
+ };
616
+ const languageModel = {
617
+ specificationVersion: "v4",
618
+ provider: "native.test",
619
+ modelId: "native-timeout",
620
+ supportedUrls: {},
621
+ doGenerate: async () => { throw new Error("buffered generation is not under test"); },
622
+ doStream: async ({ abortSignal }: { abortSignal?: AbortSignal }) => {
623
+ calls++;
624
+ if (calls > 1) {
625
+ return {
626
+ stream: new ReadableStream({
627
+ start(controller) {
628
+ controller.enqueue({ type: "stream-start", warnings: [] });
629
+ controller.enqueue({ type: "response-metadata", id: "native-retry", modelId: "native-timeout" });
630
+ controller.enqueue({ type: "text-start", id: "text-1" });
631
+ controller.enqueue({ type: "text-delta", id: "text-1", delta: "recovered" });
632
+ controller.enqueue({ type: "text-end", id: "text-1" });
633
+ controller.enqueue({
634
+ type: "finish",
635
+ finishReason: { unified: "stop", raw: "completed" },
636
+ usage,
637
+ });
638
+ controller.close();
639
+ },
640
+ }),
641
+ response: {},
642
+ };
643
+ }
644
+ return {
645
+ stream: new ReadableStream({
646
+ start(controller) {
647
+ controller.enqueue({ type: "stream-start", warnings: [] });
648
+ const timer = setTimeout(() => controller.close(), 100);
649
+ abortSignal?.addEventListener("abort", () => {
650
+ clearTimeout(timer);
651
+ controller.error(abortSignal.reason);
652
+ }, { once: true });
653
+ },
654
+ }),
655
+ response: {},
656
+ };
657
+ },
658
+ } as unknown as LanguageModel;
659
+ const provider = testProvider({
660
+ model: "native-timeout",
661
+ languageModel,
662
+ fetchTimeoutMs: 5_000,
663
+ operationTimeoutMs: 5_000,
664
+ firstContentTimeoutMs: 10,
665
+ temperature: 0.2,
666
+ repeatPenalty: 1.15,
667
+ reasoning: { mode: "off", budget: null },
668
+ retryAttempts: 1,
669
+ source: "provider:test-native",
670
+ });
671
+
672
+ const result = await provider.generate({ workerId: "native-retry", messages: [] });
673
+ assert.equal(result.assistant.content, "recovered");
674
+ assert.equal(calls, 2);
675
+ assert.deepEqual(result.accounting.map(({ outcome }) => outcome), ["error", "response"]);
676
+ });
677
+
678
+ test("compatible xAI wire usage becomes an exact tick charge without raw-body capture", async () => {
679
+ const p = testProvider({
680
+ model: "grok-test",
681
+ url: "http://x/v1/chat/completions",
682
+ fetchTimeoutMs: 5_000,
683
+ temperature: 0.2,
684
+ repeatPenalty: 1.15,
685
+ reasoning: { mode: "off", budget: null },
686
+ retryAttempts: 0,
687
+ streaming: false,
688
+ normalizeCost: providerCostNormalizer("@ai-sdk/xai"),
689
+ });
690
+ installFetchJson({
691
+ id: "response-1",
692
+ model: "grok-test",
693
+ choices: [{ message: { content: "ok" }, finish_reason: "stop" }],
694
+ usage: {
695
+ prompt_tokens: 2,
696
+ completion_tokens: 1,
697
+ total_tokens: 3,
698
+ cost_in_usd_ticks: 15_493_500,
699
+ },
700
+ });
701
+ const response = await p.generate({ workerId: "xai", messages: [] });
702
+ assert.deepEqual(response.accounting[0]?.cost, {
703
+ kind: "charged",
704
+ amount: { amount: "15493500", currency: "USDTICK" },
705
+ usdEquivalent: "0.00154935",
706
+ source: "xAI response usage.cost_in_usd_ticks",
707
+ });
708
+ assert.equal(response.rawBody, undefined);
709
+ });
710
+
711
+ test("streamed xAI final usage retains its exact tick charge", async () => {
712
+ const p = testProvider({
713
+ model: "grok-test",
714
+ url: "http://x/v1/chat/completions",
715
+ fetchTimeoutMs: 5_000,
716
+ temperature: 0.2,
717
+ repeatPenalty: 1.15,
718
+ reasoning: { mode: "off", budget: null },
719
+ retryAttempts: 0,
720
+ normalizeCost: providerCostNormalizer("@ai-sdk/xai"),
721
+ });
722
+ installFetch([
723
+ { choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] },
724
+ {
725
+ choices: [],
726
+ usage: {
727
+ prompt_tokens: 2,
728
+ completion_tokens: 1,
729
+ total_tokens: 3,
730
+ cost_in_usd_ticks: 15_493_500,
731
+ },
732
+ },
733
+ ]);
734
+ const response = await p.generate({ workerId: "xai", messages: [] });
735
+ assert.deepEqual(response.accounting[0]?.cost, {
736
+ kind: "charged",
737
+ amount: { amount: "15493500", currency: "USDTICK" },
738
+ usdEquivalent: "0.00154935",
739
+ source: "xAI response usage.cost_in_usd_ticks",
740
+ });
741
+ });
742
+
312
743
  test("generate surfaces and normalizes an out-of-set finish_reason", async () => {
313
744
  const warnings: Array<{ message: string; code?: string }> = [];
314
745
  mock.method(process, "emitWarning", (message: string | Error, options?: string | { code?: string }) => {
@@ -317,7 +748,7 @@ test("generate surfaces and normalizes an out-of-set finish_reason", async () =>
317
748
  ...(typeof options === "object" && options.code !== undefined ? { code: options.code } : {}),
318
749
  });
319
750
  });
320
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
751
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
321
752
  installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "function_call" }] }]);
322
753
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
323
754
  assert.equal(assistant.finishReason, null);
@@ -335,7 +766,7 @@ test("#161: a streamed resource interruption is a failed exchange with complete
335
766
  usage: { prompt_tokens: 7, completion_tokens: 5, total_tokens: 12 },
336
767
  },
337
768
  ]);
338
- const provider = new AiSdkProvider({
769
+ const provider = testProvider({
339
770
  ...injectedBase,
340
771
  retryAttempts: 2,
341
772
  rawBody: true,
@@ -354,12 +785,10 @@ test("#161: a streamed resource interruption is a failed exchange with complete
354
785
  assert.equal(error.attempt?.assistant.content, "partial answer");
355
786
  assert.equal(error.attempt?.assistant.reasoning, "partial thought");
356
787
  assert.equal(error.attempt?.assistant.finishReason, "resource_interrupted");
357
- assert.deepEqual(error.attempt?.assistant.usage, {
358
- prompt: 7,
359
- completion: 2,
360
- reasoning: 3,
361
- cached: 0,
362
- total: 12,
788
+ assert.deepEqual(error.accounting[0]?.usage, {
789
+ inputTokens: 7,
790
+ outputTokens: 5,
791
+ totalTokens: 12,
363
792
  });
364
793
  assert.equal(
365
794
  (error.attempt?.assistantRaw as { rawFinishReason?: string }).rawFinishReason,
@@ -382,7 +811,7 @@ test("#161: a buffered resource interruption preserves the successful wire respo
382
811
  usage: { prompt_tokens: 7, completion_tokens: 5, total_tokens: 12 },
383
812
  };
384
813
  const calls = installFetchJson(wire);
385
- const provider = new AiSdkProvider({
814
+ const provider = testProvider({
386
815
  ...injectedBase,
387
816
  streaming: false,
388
817
  retryAttempts: 2,
@@ -411,37 +840,37 @@ test("#161: a buffered resource interruption preserves the successful wire respo
411
840
  test("generate translates a backend cap synonym to canonical length", async () => {
412
841
  // gemini shouts MAX_TOKENS, anthropic says max_tokens -- both must reach core as
413
842
  // "length" so its truncation check (=== "length") is a cross-backend invariant.
414
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
843
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
415
844
  installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "MAX_TOKENS" }] }]);
416
845
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
417
846
  assert.equal(assistant.finishReason, "length");
418
847
  });
419
848
 
420
849
  test("generate translates end_turn to canonical stop", async () => {
421
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
850
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
422
851
  installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "end_turn" }] }]);
423
852
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
424
853
  assert.equal(assistant.finishReason, "stop");
425
854
  });
426
855
 
427
856
  test("generate translates xAI completed to canonical stop", async () => {
428
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
857
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
429
858
  installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "completed" }] }]);
430
859
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
431
860
  assert.equal(assistant.finishReason, "stop");
432
861
  });
433
862
 
434
863
  test("generate aggregates reasoning deltas under multiple field names", async () => {
435
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
864
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
436
865
  installFetch([{ choices: [{ delta: { reasoning_content: "be", thinking: "cause" } }] }]);
437
866
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
438
867
  assert.equal(assistant.reasoning, "because");
439
868
  assert.equal("reasoningEncrypted" in assistant, false); // open reasoning only -> field absent
440
869
  });
441
870
 
442
- test("{§provider-tagged-reasoning} explicit think-tags projects one streamed leading envelope and reclassifies usage", async () => {
871
+ test("{§provider-tagged-reasoning} explicit think-tags project content without estimating token attribution", async () => {
443
872
  const config = { ...injectedBase, reasoningResponseStyle: "think-tags" as const, rawBody: true };
444
- const p = new AiSdkProvider(config);
873
+ const p = testProvider(config);
445
874
  installFetch([
446
875
  { choices: [{ delta: { content: "<thi" } }] },
447
876
  { choices: [{ delta: { content: "nk>12345</th" } }] },
@@ -453,12 +882,10 @@ test("{§provider-tagged-reasoning} explicit think-tags projects one streamed le
453
882
 
454
883
  assert.equal(response.assistant.reasoning, "12345");
455
884
  assert.equal(response.assistant.content, "abcde");
456
- assert.deepEqual(response.assistant.usage, {
457
- prompt: 3,
458
- completion: 5,
459
- reasoning: 5,
460
- cached: 0,
461
- total: 13,
885
+ assert.deepEqual(response.accounting[0]?.usage, {
886
+ inputTokens: 3,
887
+ outputTokens: 10,
888
+ totalTokens: 13,
462
889
  });
463
890
  assert.deepEqual(
464
891
  ((response.assistantRaw as { content: string; reasoning: string }).content),
@@ -476,12 +903,12 @@ test("{§provider-tagged-reasoning} explicit think-tags projects one buffered le
476
903
  usage: { prompt_tokens: 3, completion_tokens: 10, total_tokens: 13 },
477
904
  });
478
905
  const config = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
479
- const response = await new AiSdkProvider(config).generate({ workerId: "tagged-buffer", messages: [] });
906
+ const response = await testProvider(config).generate({ workerId: "tagged-buffer", messages: [] });
480
907
 
481
908
  assert.equal(response.assistant.reasoning, "12345");
482
909
  assert.equal(response.assistant.content, "abcde");
483
- assert.equal(response.assistant.usage.completion, 5);
484
- assert.equal(response.assistant.usage.reasoning, 5);
910
+ assert.equal(response.accounting[0]?.usage?.outputTokens, 10);
911
+ assert.equal(response.accounting[0]?.usage?.outputTokenDetails, undefined);
485
912
  });
486
913
 
487
914
  test("{§provider-tagged-reasoning} tagged text does not overwrite itemized reasoning usage", async () => {
@@ -496,12 +923,14 @@ test("{§provider-tagged-reasoning} tagged text does not overwrite itemized reas
496
923
  },
497
924
  });
498
925
  const config = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
499
- const response = await new AiSdkProvider(config).generate({ workerId: "tagged-itemized", messages: [] });
926
+ const response = await testProvider(config).generate({ workerId: "tagged-itemized", messages: [] });
500
927
 
501
928
  assert.equal(response.assistant.reasoning, "12345");
502
929
  assert.equal(response.assistant.content, "abcde");
503
- assert.equal(response.assistant.usage.completion, 7);
504
- assert.equal(response.assistant.usage.reasoning, 3);
930
+ assert.deepEqual(response.accounting[0]?.usage?.outputTokenDetails, {
931
+ textTokens: 7,
932
+ reasoningTokens: 3,
933
+ });
505
934
  });
506
935
 
507
936
  test("{§provider-tagged-reasoning} an unclosed capped envelope is wholly reasoning in streamed and buffered responses", async () => {
@@ -510,15 +939,13 @@ test("{§provider-tagged-reasoning} an unclosed capped envelope is wholly reason
510
939
  { choices: [{ delta: { content: "<think>unfinished" }, finish_reason: "length" }] },
511
940
  { usage: { prompt_tokens: 3, completion_tokens: 8, total_tokens: 11 } },
512
941
  ]);
513
- const streamed = await new AiSdkProvider(config).generate({ workerId: "tagged-capped-stream", messages: [] });
942
+ const streamed = await testProvider(config).generate({ workerId: "tagged-capped-stream", messages: [] });
514
943
  assert.equal(streamed.assistant.reasoning, "unfinished");
515
944
  assert.equal(streamed.assistant.content, "");
516
- assert.deepEqual(streamed.assistant.usage, {
517
- prompt: 3,
518
- completion: 0,
519
- reasoning: 8,
520
- cached: 0,
521
- total: 11,
945
+ assert.deepEqual(streamed.accounting[0]?.usage, {
946
+ inputTokens: 3,
947
+ outputTokens: 8,
948
+ totalTokens: 11,
522
949
  });
523
950
 
524
951
  mock.restoreAll();
@@ -528,11 +955,11 @@ test("{§provider-tagged-reasoning} an unclosed capped envelope is wholly reason
528
955
  usage: { prompt_tokens: 3, completion_tokens: 8, total_tokens: 11 },
529
956
  });
530
957
  const bufferedConfig = { ...config, streaming: false };
531
- const buffered = await new AiSdkProvider(bufferedConfig).generate({ workerId: "tagged-capped-buffer", messages: [] });
958
+ const buffered = await testProvider(bufferedConfig).generate({ workerId: "tagged-capped-buffer", messages: [] });
532
959
  assert.equal(buffered.assistant.reasoning, "unfinished");
533
960
  assert.equal(buffered.assistant.content, "");
534
- assert.equal(buffered.assistant.usage.completion, 0);
535
- assert.equal(buffered.assistant.usage.reasoning, 8);
961
+ assert.equal(buffered.accounting[0]?.usage?.outputTokens, 8);
962
+ assert.equal(buffered.accounting[0]?.usage?.outputTokenDetails, undefined);
536
963
  });
537
964
 
538
965
  test("{§provider-tagged-reasoning} verbatim, non-leading, and structured-reasoning controls preserve literal tags", async () => {
@@ -541,11 +968,11 @@ test("{§provider-tagged-reasoning} verbatim, non-leading, and structured-reason
541
968
  choices: [{ message: { content: "<think>literal</think>answer" }, finish_reason: "stop" }],
542
969
  usage: { prompt_tokens: 1, completion_tokens: 4, total_tokens: 5 },
543
970
  });
544
- const verbatim = await new AiSdkProvider({ ...injectedBase, streaming: false })
971
+ const verbatim = await testProvider({ ...injectedBase, streaming: false })
545
972
  .generate({ workerId: "verbatim", messages: [] });
546
973
  assert.equal(verbatim.assistant.content, "<think>literal</think>answer");
547
974
  assert.equal(verbatim.assistant.reasoning, null);
548
- assert.equal(verbatim.assistant.usage.completion, 4);
975
+ assert.equal(verbatim.accounting[0]?.usage?.outputTokens, 4);
549
976
 
550
977
  mock.restoreAll();
551
978
  installFetchJson({
@@ -554,7 +981,7 @@ test("{§provider-tagged-reasoning} verbatim, non-leading, and structured-reason
554
981
  usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 },
555
982
  });
556
983
  const taggedConfig = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
557
- const nonLeading = await new AiSdkProvider(taggedConfig)
984
+ const nonLeading = await testProvider(taggedConfig)
558
985
  .generate({ workerId: "non-leading", messages: [] });
559
986
  assert.equal(nonLeading.assistant.content, "show <think>literal</think> exactly");
560
987
  assert.equal(nonLeading.assistant.reasoning, null);
@@ -568,14 +995,14 @@ test("{§provider-tagged-reasoning} verbatim, non-leading, and structured-reason
568
995
  }, finish_reason: "stop" }],
569
996
  usage: { prompt_tokens: 1, completion_tokens: 7, total_tokens: 8 },
570
997
  });
571
- const structured = await new AiSdkProvider(taggedConfig)
998
+ const structured = await testProvider(taggedConfig)
572
999
  .generate({ workerId: "structured", messages: [] });
573
1000
  assert.equal(structured.assistant.content, "<think>literal visible bytes</think>");
574
1001
  assert.equal(structured.assistant.reasoning, "structured reasoning");
575
1002
  });
576
1003
 
577
1004
  test("{§provider-tagged-reasoning} grammar evidence retains the exact pre-projection tagged sentence", async () => {
578
- const content = "<think>🧠reason</think><<PLAN::PLAN\n<<SEND[200]:done:SEND";
1005
+ const content = "<think>🧠reason</think># PLAN0\n\n## SEND0 [200]\ndone";
579
1006
  const config = {
580
1007
  ...injectedBase,
581
1008
  contextWindow: 640,
@@ -586,14 +1013,14 @@ test("{§provider-tagged-reasoning} grammar evidence retains the exact pre-proje
586
1013
  };
587
1014
  installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
588
1015
 
589
- const response = await new AiSdkProvider(config).generate({
1016
+ const response = await testProvider(config).generate({
590
1017
  workerId: "tagged-grammar",
591
1018
  messages: [],
592
1019
  grammar: `root ::= ${JSON.stringify(content)}`,
593
1020
  });
594
1021
 
595
1022
  assert.equal(response.assistant.reasoning, "🧠reason");
596
- assert.equal(response.assistant.content, "<<PLAN::PLAN\n<<SEND[200]:done:SEND");
1023
+ assert.equal(response.assistant.content, "# PLAN0\n\n## SEND0 [200]\ndone");
597
1024
  assert.deepEqual(response.grammarEvidence, {
598
1025
  input: content,
599
1026
  contentStart: [..."<think>🧠reason</think>"].length,
@@ -610,7 +1037,7 @@ test("encrypted reasoning (non-streamed): encrypted entries normalize and text e
610
1037
  { type: "reasoning.text", text: "never surfaced here" },
611
1038
  ],
612
1039
  }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
613
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
1040
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
614
1041
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
615
1042
  // Wire detail ID is preserved; the assistant-message location supports the
616
1043
  // derived classification but supplies no downstream client entity ID.
@@ -624,7 +1051,7 @@ test("distinct encrypted-reasoning wire ids stay distinct items", async () => {
624
1051
  { type: "reasoning.encrypted", data: "AAA", format: "openai-responses-v1", id: "rs_1" },
625
1052
  { type: "reasoning.encrypted", data: "BBB", format: "openai-responses-v1", id: "rs_2" },
626
1053
  ] }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
627
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
1054
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
628
1055
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
629
1056
  assert.equal(assistant.reasoningEncrypted?.length, 2);
630
1057
  assert.deepEqual(assistant.reasoningEncrypted?.map((i) => i.id), ["rs_1", "rs_2"]);
@@ -634,7 +1061,7 @@ test("assistant-message location classifies encrypted reasoning without inventin
634
1061
  installFetchJson({ model: "m", choices: [{ message: { content: "ok", reasoning_details: [
635
1062
  { type: "reasoning.encrypted", data: "OPAQUE", format: "openai-responses-v1", id: null, index: 0 },
636
1063
  ] }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
637
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
1064
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
638
1065
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
639
1066
  assert.deepEqual(assistant.reasoningEncrypted, [{
640
1067
  id: null,
@@ -644,7 +1071,7 @@ test("assistant-message location classifies encrypted reasoning without inventin
644
1071
  });
645
1072
 
646
1073
  test("encrypted reasoning (streamed): chunked blob concatenates per entry index", async () => {
647
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1074
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
648
1075
  installFetch([
649
1076
  { choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "gAAAA", format: "openai-responses-v1", id: "rs_1", index: 0 }] } }] },
650
1077
  { choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "BqXYZ", id: "rs_1", index: 0 }] } }] },
@@ -655,32 +1082,35 @@ test("encrypted reasoning (streamed): chunked blob concatenates per entry index"
655
1082
  assert.equal(assistant.content, "4");
656
1083
  });
657
1084
 
658
- test("reasoningStyle 'think' gates on budget != 0 (magnitude irrelevant for native)", async () => {
659
- const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
1085
+ test("reasoningStyle 'think' follows activation (magnitude is irrelevant to the boolean wire control)", async () => {
1086
+ const on = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
660
1087
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
661
1088
  await on.generate({ workerId: "r", messages: [] });
662
1089
  assert.equal(JSON.parse(calls[0].init.body as string).think, true);
663
1090
 
664
1091
  mock.restoreAll();
665
- const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
1092
+ const off = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
666
1093
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
667
1094
  await off.generate({ workerId: "r", messages: [] });
668
1095
  assert.equal("think" in JSON.parse(calls[0].init.body as string), false);
669
1096
  });
670
1097
 
671
- test("reasoningStyle 'effort' sends a reasoning_effort tier from the budget", async () => {
672
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: 5000 }, retryAttempts: 0, reasoningStyle: "effort" });
673
- const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
674
- await p.generate({ workerId: "r", messages: [] });
675
- assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, "high");
1098
+ test("reasoningStyle 'effort' enables at the portable default without inventing a budget", async () => {
1099
+ for (const [budget, expected] of [[null, "medium"], [5000, "high"]] as const) {
1100
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget }, retryAttempts: 0, reasoningStyle: "effort" });
1101
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1102
+ await p.generate({ workerId: "r", messages: [] });
1103
+ assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, expected);
1104
+ mock.restoreAll();
1105
+ }
676
1106
  });
677
1107
 
678
1108
  test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS, on sends the tier", async () => {
679
1109
  // expected === null → the field must be ABSENT from the wire body. Fireworks
680
1110
  // 400s reasoning_effort='adaptive' for non-MiniMax models (wire-verified,
681
1111
  // Adaptive = the backend's own default posture = omission.
682
- for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "on", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "on"; budget: number | null }, string | null]>) {
683
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
1112
+ for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "on", budget: null }, "medium"], [{ mode: "on", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "on"; budget: number | null }, string | null]>) {
1113
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
684
1114
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
685
1115
  await p.generate({ workerId: "r", messages: [] });
686
1116
  const body = JSON.parse(calls[0].init.body as string);
@@ -694,10 +1124,11 @@ test("{§deepseek-reasoning-request} #157: thinking_effort maps the complete Dee
694
1124
  const cases = [
695
1125
  [{ mode: "off", budget: null }, { thinking: { type: "disabled" } }],
696
1126
  [{ mode: "adaptive", budget: null }, {}],
1127
+ [{ mode: "on", budget: null }, { thinking: { type: "enabled" } }],
697
1128
  [{ mode: "on", budget: 5000 }, { thinking: { type: "enabled" }, reasoning_effort: "high" }],
698
1129
  ] as const;
699
1130
  for (const [reasoning, expected] of cases) {
700
- const p = new AiSdkProvider({
1131
+ const p = testProvider({
701
1132
  model: "m",
702
1133
  url: "http://x/v1/chat/completions",
703
1134
  fetchTimeoutMs: 5000,
@@ -723,7 +1154,7 @@ test("{§deepseek-reasoning-request} #157: thinking_effort maps the complete Dee
723
1154
  });
724
1155
 
725
1156
  test("the family temperature default rides every request; caller sampling overrides it", async () => {
726
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1157
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
727
1158
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
728
1159
  await p.generate({ workerId: "r", messages: [] });
729
1160
  assert.equal(JSON.parse(calls[0].init.body as string).temperature, 0.2);
@@ -742,7 +1173,7 @@ test("the family temperature default rides every request; caller sampling overri
742
1173
  test("DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box default; never on cloud", async () => {
743
1174
  const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off" as const, budget: null }, retryAttempts: 0 };
744
1175
  // set + llamacpp -> the loop-breakers ride the wire
745
- const p = new AiSdkProvider({ ...base, grammarStyle: "llamacpp", dryMultiplier: 0.8, dryBase: 1.75, dryAllowedLength: 2, repeatLastN: 512 });
1176
+ const p = testProvider({ ...base, grammarStyle: "llamacpp", dryMultiplier: 0.8, dryBase: 1.75, dryAllowedLength: 2, repeatLastN: 512 });
746
1177
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
747
1178
  await p.generate({ workerId: "r", messages: [] });
748
1179
  let body = JSON.parse(calls[0].init.body as string);
@@ -753,7 +1184,7 @@ test("DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box
753
1184
  assert.equal(body.repeat_penalty, 1.15); // repeat_penalty always rides the llamacpp path
754
1185
  mock.restoreAll();
755
1186
  // unset -> no dry_*/repeat_last_n on the wire (box keeps its own defaults)
756
- const p2 = new AiSdkProvider({ ...base, grammarStyle: "llamacpp" });
1187
+ const p2 = testProvider({ ...base, grammarStyle: "llamacpp" });
757
1188
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
758
1189
  await p2.generate({ workerId: "r", messages: [] });
759
1190
  body = JSON.parse(calls[0].init.body as string);
@@ -761,7 +1192,7 @@ test("DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box
761
1192
  assert.equal("repeat_last_n" in body, false);
762
1193
  mock.restoreAll();
763
1194
  // DRY is a llama.cpp sampler: a cloud ("none") provider never emits it, even if configured
764
- const p3 = new AiSdkProvider({ ...base, grammarStyle: "none", dryMultiplier: 0.8, repeatLastN: 512 });
1195
+ const p3 = testProvider({ ...base, grammarStyle: "none", dryMultiplier: 0.8, repeatLastN: 512 });
765
1196
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
766
1197
  await p3.generate({ workerId: "r", messages: [] });
767
1198
  body = JSON.parse(calls[0].init.body as string);
@@ -771,7 +1202,7 @@ test("DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box
771
1202
  });
772
1203
 
773
1204
  test("llamacpp grammar path: temperature default + the managed repeat-penalty floor", async () => {
774
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
1205
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
775
1206
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
776
1207
  await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
777
1208
  const body = JSON.parse(calls[0].init.body as string);
@@ -781,13 +1212,13 @@ test("llamacpp grammar path: temperature default + the managed repeat-penalty fl
781
1212
 
782
1213
  test("the repeat penalty rides every request rail-off, keyed per backend", async () => {
783
1214
  // llama.cpp with NO grammar carries its key too (unconstrained local is guarded)
784
- const llama = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
1215
+ const llama = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
785
1216
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
786
1217
  await llama.generate({ workerId: "r", messages: [] });
787
1218
  assert.equal(JSON.parse(calls[0].init.body as string).repeat_penalty, 1.15);
788
1219
  mock.restoreAll();
789
1220
  // A `none`-style cloud backend with a frequency penalty gets frequency_penalty.
790
- const cloud = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1221
+ const cloud = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
791
1222
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
792
1223
  await cloud.generate({ workerId: "r", messages: [] });
793
1224
  const cloudBody = JSON.parse(calls[0].init.body as string);
@@ -796,14 +1227,14 @@ test("the repeat penalty rides every request rail-off, keyed per backend", async
796
1227
  assert.equal("repeat_penalty" in cloudBody, false);
797
1228
  mock.restoreAll();
798
1229
  // frequencyPenalty unset (default 0) opts out cleanly - sends nothing (an out-of-date plugin runs unguarded, never breaks)
799
- const bare = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1230
+ const bare = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
800
1231
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
801
1232
  await bare.generate({ workerId: "r", messages: [] });
802
1233
  assert.equal("frequency_penalty" in JSON.parse(calls[0].init.body as string), false);
803
1234
  });
804
1235
 
805
1236
  test("sampling passthrough forwards caller params; managed + reserved keys win", async () => {
806
- const p = new AiSdkProvider({ model: "managed-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1237
+ const p = testProvider({ model: "managed-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
807
1238
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
808
1239
  await p.generate({
809
1240
  workerId: "r",
@@ -826,7 +1257,7 @@ test("sampling passthrough forwards caller params; managed + reserved keys win",
826
1257
  });
827
1258
 
828
1259
  test("sampling passthrough guards contract invariants: n/tools/caps stripped, platform knobs pass", async () => {
829
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1260
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
830
1261
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
831
1262
  await p.generate({
832
1263
  workerId: "r",
@@ -851,15 +1282,17 @@ test("sampling passthrough guards contract invariants: n/tools/caps stripped, pl
851
1282
  });
852
1283
 
853
1284
  test("template reasoning returns the exact pre-projection grammar sentence ({§gbnf-response-observation})", async () => {
854
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
855
- const calls = installFetch([{ choices: [{ delta: { reasoning_content: "con🙂sider", content: "x" } }] }]);
1285
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
856
1286
  const grammarInput = "<|channel>thought\ncon🙂sider<channel|>x";
1287
+ const calls = installFetch([{ choices: [{ delta: { content: grammarInput } }] }]);
857
1288
  const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(grammarInput)}` });
858
1289
  const body = JSON.parse(calls[0].init.body as string);
859
1290
  assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true });
860
- assert.equal(body.reasoning_format, "auto");
1291
+ assert.equal(body.reasoning_format, "none");
861
1292
  assert.equal(body.thinking_budget_tokens, 64);
862
1293
  assert.equal(body.grammar, `root ::= ${JSON.stringify(grammarInput)}`);
1294
+ assert.equal(res.assistant.reasoning, "con🙂sider");
1295
+ assert.equal(res.assistant.content, "x");
863
1296
  assert.deepEqual(res.grammarEvidence, {
864
1297
  input: grammarInput,
865
1298
  contentStart: [..."<|channel>thought\ncon🙂sider<channel|>"].length,
@@ -868,13 +1301,64 @@ test("template reasoning returns the exact pre-projection grammar sentence ({§g
868
1301
  assert.equal(res.meta?.railsVerdict, undefined, "the provider represents evidence but does not grade itself");
869
1302
  });
870
1303
 
871
- test("template reasoning does not invent pre-projection evidence when the wire omits its reasoning field", async () => {
872
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
873
- installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1304
+ test("template reasoning projects a leading think envelope without losing grammar evidence", async () => {
1305
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1306
+ const input = "<think>\ncon🙂sider</think>x";
1307
+ const calls = installFetch([{ choices: [{ delta: { content: input } }] }]);
1308
+ const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(input)}` });
1309
+ const body = JSON.parse(calls[0].init.body as string);
1310
+ assert.equal(body.reasoning_format, "none");
1311
+ assert.equal(res.assistant.reasoning, "con🙂sider");
1312
+ assert.equal(res.assistant.content, "x");
1313
+ assert.deepEqual(res.grammarEvidence, {
1314
+ input,
1315
+ contentStart: [..."<think>\ncon🙂sider</think>"].length,
1316
+ transported: true,
1317
+ });
1318
+ });
1319
+
1320
+ test("a verbatim template response remains exact evidence when it has no channel envelope", async () => {
1321
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1322
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1323
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
1324
+ const body = JSON.parse(calls[0].init.body as string);
1325
+ assert.equal(body.reasoning_format, "none");
1326
+ assert.deepEqual(res.grammarEvidence, { input: "x", contentStart: 0, transported: true });
1327
+ });
1328
+
1329
+ test("a template grammar preserves exact evidence when reasoning is disabled", async () => {
1330
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1331
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1332
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
1333
+ const body = JSON.parse(calls[0].init.body as string);
1334
+ assert.deepEqual(body.chat_template_kwargs, { enable_thinking: false });
1335
+ assert.equal(body.reasoning_format, "none");
1336
+ assert.deepEqual(res.grammarEvidence, { input: "x", contentStart: 0, transported: true });
1337
+ });
1338
+
1339
+ test("an unexpectedly projected template response cannot claim pre-projection evidence", async () => {
1340
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1341
+ installFetch([{ choices: [{ delta: { reasoning_content: "reason", content: "x" } }] }]);
874
1342
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
875
1343
  assert.equal(res.grammarEvidence, undefined);
876
1344
  });
877
1345
 
1346
+ test("template reasoning preserves an empty grammar-required channel as exact evidence", async () => {
1347
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1348
+ const input = "<|channel>thought\n<channel|>x";
1349
+ const calls = installFetch([{ choices: [{ delta: { content: input } }] }]);
1350
+ const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(input)}` });
1351
+ const body = JSON.parse(calls[0].init.body as string);
1352
+ assert.equal(body.reasoning_format, "none");
1353
+ assert.equal(res.assistant.reasoning, null);
1354
+ assert.equal(res.assistant.content, "x");
1355
+ assert.deepEqual(res.grammarEvidence, {
1356
+ input,
1357
+ contentStart: [..."<|channel>thought\n<channel|>"].length,
1358
+ transported: true,
1359
+ });
1360
+ });
1361
+
878
1362
  test("channel-escape detector: billed completion tokens vastly beyond visible channels attach grammar_unenforced", async () => {
879
1363
  // The run105 shape: tiny visible content, no reasoning, thousands billed — the decode
880
1364
  // escaped into a discarded reasoning block, unconstrained.
@@ -891,16 +1375,16 @@ test("channel-escape detector: billed completion tokens vastly beyond visible ch
891
1375
  }
892
1376
  return new Response(sseStream(chunks), { status: 200 });
893
1377
  };
894
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetch, tokenizeUrl: "http://x/tokenize", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1378
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetch, tokenizeUrl: "http://x/tokenize", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
895
1379
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
896
1380
  const escape = res.notices?.find((e) => e.message.includes("escaped the grammar"));
897
1381
  assert.ok(escape, "escape notice attached");
898
1382
  assert.equal(escape!.kind, "grammar_unenforced");
899
- assert.match(escape!.message ?? "", /5000 completion tokens billed/);
1383
+ assert.match(escape!.message ?? "", /5000 output tokens billed/);
900
1384
  });
901
1385
 
902
1386
  test("channel-escape state is absent without a transported grammar", async () => {
903
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1387
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
904
1388
  installFetch([
905
1389
  { choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
906
1390
  { usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
@@ -911,7 +1395,7 @@ test("channel-escape state is absent without a transported grammar", async () =>
911
1395
  });
912
1396
 
913
1397
  test("reasoningStyle 'template' sends llama-server activation, parser, and response-wide allowance", async () => {
914
- const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
1398
+ const on = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
915
1399
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
916
1400
  await on.generate({ workerId: "r", messages: [] });
917
1401
  let body = JSON.parse(calls[0].init.body as string);
@@ -920,7 +1404,7 @@ test("reasoningStyle 'template' sends llama-server activation, parser, and respo
920
1404
  assert.equal(body.thinking_budget_tokens, 64);
921
1405
 
922
1406
  mock.restoreAll();
923
- const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
1407
+ const off = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
924
1408
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
925
1409
  await off.generate({ workerId: "r", messages: [] });
926
1410
  body = JSON.parse(calls[0].init.body as string);
@@ -931,33 +1415,54 @@ test("reasoningStyle 'template' sends llama-server activation, parser, and respo
931
1415
 
932
1416
  test("reasoningStyle 'template' explicit budget tightens the reserve and cannot exceed it", async () => {
933
1417
  const base = { model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 } as const, completionReserve: { tokens: 160 } as const, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryAttempts: 0, reasoningStyle: "template" as const };
934
- const p = new AiSdkProvider({ ...base, reasoning: { mode: "on", budget: 32 } });
1418
+ const p = testProvider({ ...base, reasoning: { mode: "on", budget: 32 } });
935
1419
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
936
1420
  await p.generate({ workerId: "r", messages: [], sampling: { thinking_budget_tokens: 999, reasoning_format: "none" } });
937
1421
  const body = JSON.parse(calls[0].init.body as string);
938
1422
  assert.equal(body.thinking_budget_tokens, 32);
939
1423
  assert.equal(body.reasoning_format, "auto");
940
1424
  assert.throws(
941
- () => new AiSdkProvider({ ...base, reasoning: { mode: "on", budget: 65 } }),
1425
+ () => testProvider({ ...base, reasoning: { mode: "on", budget: 65 } }),
942
1426
  /REASONING_BUDGET \(65\) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE \(64\)/,
943
1427
  );
944
1428
  });
945
1429
 
946
- test("budget 0 suppresses effort and include_reasoning", async () => {
947
- const effort = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "effort" });
1430
+ test("reasoningStyle 'template' explicit activation without a budget uses the resolved reserve", async () => {
1431
+ const p = testProvider({
1432
+ model: "m",
1433
+ url: "http://x/v1/chat/completions",
1434
+ contextWindow: 640,
1435
+ reasoningReserve: { tokens: 64 },
1436
+ completionReserve: { tokens: 160 },
1437
+ fetchTimeoutMs: 5000,
1438
+ temperature: 0.2,
1439
+ repeatPenalty: 1.15,
1440
+ reasoning: { mode: "on", budget: null },
1441
+ retryAttempts: 0,
1442
+ reasoningStyle: "template",
1443
+ });
1444
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1445
+ await p.generate({ workerId: "r", messages: [] });
1446
+ const body = JSON.parse(calls[0].init.body as string);
1447
+ assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true });
1448
+ assert.equal(body.thinking_budget_tokens, 64);
1449
+ });
1450
+
1451
+ test("reasoning off suppresses effort and include_reasoning controls", async () => {
1452
+ const effort = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "effort" });
948
1453
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
949
1454
  await effort.generate({ workerId: "r", messages: [] });
950
1455
  assert.equal("reasoning_effort" in JSON.parse(calls[0].init.body as string), false);
951
1456
 
952
1457
  mock.restoreAll();
953
- const relay = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
1458
+ const relay = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
954
1459
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
955
1460
  await relay.generate({ workerId: "r", messages: [] });
956
1461
  assert.equal("include_reasoning" in JSON.parse(calls[0].init.body as string), false);
957
1462
  });
958
1463
 
959
1464
  test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", async () => {
960
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
1465
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
961
1466
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
962
1467
  await p.generate({ workerId: "r", messages: [] });
963
1468
  assert.equal(JSON.parse(calls[0].init.body as string).include_reasoning, true);
@@ -966,7 +1471,7 @@ test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", asy
966
1471
  // — grammar-constrained sampling —
967
1472
 
968
1473
  test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor", async () => {
969
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
1474
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
970
1475
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
971
1476
  await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
972
1477
  const body = JSON.parse(calls[0].init.body as string);
@@ -976,7 +1481,7 @@ test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor
976
1481
  });
977
1482
 
978
1483
  test("grammar transport 'none' (default): the grammar is never sent — no silent unconstrained", async () => {
979
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1484
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
980
1485
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
981
1486
  await p.generate({ workerId: "r", messages: [], grammar: "root ::= statement" });
982
1487
  const body = JSON.parse(calls[0].init.body as string);
@@ -986,7 +1491,7 @@ test("grammar transport 'none' (default): the grammar is never sent — no silen
986
1491
 
987
1492
  // — exact pre-projection grammar evidence ({§gbnf-response-observation}) —
988
1493
 
989
- const grammarProvider = () => new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", source: "provider:test" });
1494
+ const grammarProvider = () => testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", source: "provider:test" });
990
1495
  const streamingContent = (content: string) => installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
991
1496
 
992
1497
  test("an unsplit grammar response carries the exact observed sentence", async () => {
@@ -1020,7 +1525,7 @@ test("empty unsplit content remains exact grammar evidence", async () => {
1020
1525
  });
1021
1526
 
1022
1527
  test("grammarStyle 'none' produces no grammar observation", async () => {
1023
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // grammarStyle defaults to "none"
1528
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // grammarStyle defaults to "none"
1024
1529
  streamingContent("anything goes");
1025
1530
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
1026
1531
  assert.equal(res.assistant.content, "anything goes");
@@ -1039,7 +1544,7 @@ test("provider evidence does not depend on the local validator understanding the
1039
1544
  // — PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar, withhold it, and preserve the observation —
1040
1545
 
1041
1546
  test("gbnfDebug marks an unconstrained observation as not transported", async () => {
1042
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
1547
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
1043
1548
  const calls = installFetch([{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }]);
1044
1549
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
1045
1550
  const body = JSON.parse(calls[0].init.body as string);
@@ -1051,7 +1556,7 @@ test("gbnfDebug marks an unconstrained observation as not transported", async ()
1051
1556
  });
1052
1557
 
1053
1558
  test("gbnfDebug preserves conflicting bytes without a provider verdict", async () => {
1054
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
1559
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
1055
1560
  const calls = installFetch([{ choices: [{ delta: { content: "xon-conforming output" }, finish_reason: "stop" }] }]);
1056
1561
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
1057
1562
  assert.equal(res.assistant.content, "xon-conforming output");
@@ -1062,7 +1567,7 @@ test("gbnfDebug preserves conflicting bytes without a provider verdict", async (
1062
1567
  });
1063
1568
 
1064
1569
  test("gbnfDebug: an INVALID grammar throws before any wire call — it never reaches the model", async () => {
1065
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true });
1570
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true });
1066
1571
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1067
1572
  await assert.rejects(
1068
1573
  () => p.generate({ workerId: "r", messages: [], grammar: 'foo ::= "a"' }), // no `root` rule → invalid GBNF
@@ -1074,7 +1579,7 @@ test("gbnfDebug: an INVALID grammar throws before any wire call — it never rea
1074
1579
  // — meta bag: verbatim provider metadata —
1075
1580
 
1076
1581
  test("meta: passes backend fields through without reinterpreting monetary values", async () => {
1077
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
1582
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
1078
1583
  const balance = { amount: "0.0000042", currency: "XMR" };
1079
1584
  installFetchJson({ ...jsonChoice, balance, system_fingerprint: "fp_abc" });
1080
1585
  const res = await p.generate({ workerId: "r", messages: [] });
@@ -1088,15 +1593,34 @@ const headerVal = (init: RequestInit, name: string): string | undefined =>
1088
1593
  new Headers(init.headers).get(name) ?? undefined;
1089
1594
 
1090
1595
  test("firstPartyMetadata: attributions + client ride as Plurnk-* headers", async () => {
1091
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1596
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1092
1597
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1093
1598
  await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0", "@foo/y@0.3.1"], client: "plurnk.nvim/1.4.0" });
1094
1599
  assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), '["@acme/x@1.2.0","@foo/y@0.3.1"]');
1095
1600
  assert.equal(headerVal(calls[0].init, "Plurnk-Client"), "plurnk.nvim/1.4.0");
1096
1601
  });
1097
1602
 
1603
+ test("Plurnk-Call-Kind carries the caller's emission or bare output contract under the first-party gate", async () => {
1604
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1605
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1606
+ await p.generate({ workerId: "emission", messages: [], callKind: "emission" });
1607
+ await p.generate({ workerId: "bare", messages: [], callKind: "bare" });
1608
+ assert.equal(headerVal(calls[0].init, "Plurnk-Call-Kind"), "emission");
1609
+ assert.equal(headerVal(calls[1].init, "Plurnk-Call-Kind"), "bare");
1610
+ });
1611
+
1612
+ test("generate rejects an unknown call kind before provider I/O", async () => {
1613
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1614
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1615
+ await assert.rejects(
1616
+ p.generate({ workerId: "invalid", messages: [], callKind: "unknown" as never }),
1617
+ /unsupported callKind "unknown"/,
1618
+ );
1619
+ assert.equal(calls.length, 0);
1620
+ });
1621
+
1098
1622
  test("Plurnk-Worker-Primary: the lineage root rides under the gate; emitted even when it equals workerId", async () => {
1099
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1623
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1100
1624
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1101
1625
  await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
1102
1626
  assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), "w-root"); // a descendant: Primary != Worker-Id
@@ -1115,22 +1639,23 @@ test("Plurnk-Worker-Primary: the lineage root rides under the gate; emitted even
1115
1639
  });
1116
1640
 
1117
1641
  test("Plurnk-Worker-Primary is structurally dropped when firstPartyMetadata is off", async () => {
1118
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1642
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1119
1643
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1120
1644
  await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
1121
1645
  assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), undefined); // never reaches a third-party backend
1122
1646
  });
1123
1647
 
1124
1648
  test("firstPartyMetadata off (default): the headers are structurally dropped even when values are passed", async () => {
1125
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1649
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1126
1650
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1127
- await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0"], client: "plurnk-cli/2.0.0" });
1651
+ await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0"], client: "plurnk-cli/2.0.0", callKind: "bare" });
1128
1652
  assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), undefined); // never leaks to a non-first-party backend
1129
1653
  assert.equal(headerVal(calls[0].init, "Plurnk-Client"), undefined);
1654
+ assert.equal(headerVal(calls[0].init, "Plurnk-Call-Kind"), undefined);
1130
1655
  });
1131
1656
 
1132
1657
  test("firstPartyMetadata on but empty values: no header emitted", async () => {
1133
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1658
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1134
1659
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1135
1660
  await p.generate({ workerId: "r", messages: [], attributions: [], client: "" });
1136
1661
  assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), undefined);
@@ -1138,7 +1663,7 @@ test("firstPartyMetadata on but empty values: no header emitted", async () => {
1138
1663
  });
1139
1664
 
1140
1665
  test("grammar transport: no grammar passed sends no grammar field, but the penalty rides", async () => {
1141
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
1666
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
1142
1667
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1143
1668
  await p.generate({ workerId: "r", messages: [] });
1144
1669
  const body = JSON.parse(calls[0].init.body as string);
@@ -1147,7 +1672,7 @@ test("grammar transport: no grammar passed sends no grammar field, but the penal
1147
1672
  });
1148
1673
 
1149
1674
  test("maxTokens transports as max_tokens; absent → no wire field (server default)", async () => {
1150
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1675
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1151
1676
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1152
1677
  await p.generate({ workerId: "r", messages: [], maxTokens: 2048 });
1153
1678
  assert.equal(JSON.parse(calls[0].init.body as string).max_tokens, 2048);
@@ -1159,7 +1684,7 @@ test("maxTokens transports as max_tokens; absent → no wire field (server defau
1159
1684
  });
1160
1685
 
1161
1686
  test("slot affinity is internal: sticky per workerId, distinct workers spread across slots", async () => {
1162
- const pinning = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
1687
+ const pinning = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
1163
1688
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1164
1689
  await pinning.generate({ workerId: "run-A", messages: [] });
1165
1690
  await pinning.generate({ workerId: "run-B", messages: [] });
@@ -1170,20 +1695,20 @@ test("slot affinity is internal: sticky per workerId, distinct workers spread ac
1170
1695
  });
1171
1696
 
1172
1697
  test("slot affinity: no pinning backend or unknown slotCount → no id_slot ever", async () => {
1173
- const cloud = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // default: no pinning
1698
+ const cloud = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // default: no pinning
1174
1699
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1175
1700
  await cloud.generate({ workerId: "run-A", messages: [] });
1176
1701
  assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
1177
1702
 
1178
1703
  mock.restoreAll();
1179
- const noCount = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true }); // slotCount null
1704
+ const noCount = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true }); // slotCount null
1180
1705
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1181
1706
  await noCount.generate({ workerId: "run-A", messages: [] });
1182
1707
  assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
1183
1708
  });
1184
1709
 
1185
1710
  test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; recent workers stay sticky", async () => {
1186
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
1711
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
1187
1712
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1188
1713
  const slotOf = (i: number) => JSON.parse(calls[i].init.body as string).id_slot;
1189
1714
  for (let i = 0; i < 16; i++) await p.generate({ workerId: `r${i}`, messages: [] }); // fills the 16-entry window {r0..r15}
@@ -1197,7 +1722,7 @@ test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; r
1197
1722
 
1198
1723
  test("streaming:false: a non-ok response rejects as a classified ProviderError (covers the non-streamed transport)", async () => {
1199
1724
  const { ProviderError } = await import("./errors.ts");
1200
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, source: "provider:test" });
1725
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, source: "provider:test" });
1201
1726
  mock.method(globalThis, "fetch", async () => new Response("boom", { status: 500 }));
1202
1727
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
1203
1728
  assert.ok(err instanceof ProviderError, `expected ProviderError, got ${String(err)}`);
@@ -1208,14 +1733,14 @@ test("streaming:false: a non-ok response rejects as a classified ProviderError (
1208
1733
  });
1209
1734
 
1210
1735
  test("generate fail-hards on a missing or empty workerId", async () => {
1211
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1736
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1212
1737
  installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1213
1738
  await assert.rejects(() => p.generate({ workerId: "", messages: [] }), /workerId is required/);
1214
1739
  await assert.rejects(() => (p.generate as (a: object) => Promise<unknown>)({ messages: [] }), /workerId is required/);
1215
1740
  });
1216
1741
 
1217
1742
  test("messages pass through verbatim — the provider injects no turn (PLAN lives in the grammar, never a provider prefill)", async () => {
1218
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1743
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1219
1744
  const calls = installFetch([{ choices: [{ delta: { content: "out" } }] }]);
1220
1745
  const input = [{ role: "user" as const, content: "hi" }];
1221
1746
  const res = await p.generate({ workerId: "r", messages: input });
@@ -1225,7 +1750,7 @@ test("messages pass through verbatim — the provider injects no turn (PLAN live
1225
1750
 
1226
1751
  test("generate wraps an HTTP failure as a ProviderError carrying Problem Details", async () => {
1227
1752
  const { ProviderError } = await import("./errors.ts");
1228
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, source: "provider:test" });
1753
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, source: "provider:test" });
1229
1754
  mock.method(globalThis, "fetch", async () => new Response("rate limited", { status: 429 }));
1230
1755
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
1231
1756
  assert.ok(err instanceof ProviderError, `expected ProviderError, got ${String(err)}`);
@@ -1239,14 +1764,14 @@ test("generate wraps an HTTP failure as a ProviderError carrying Problem Details
1239
1764
  });
1240
1765
 
1241
1766
  test("generate rejects on a pre-aborted external signal", async () => {
1242
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1767
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1243
1768
  installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1244
1769
  const signal = AbortSignal.abort(new Error("nope"));
1245
1770
  await assert.rejects(() => p.generate({ workerId: "r", messages: [], signal }));
1246
1771
  });
1247
1772
 
1248
1773
  test("configured headers and url are sent verbatim", async () => {
1249
- const p = new AiSdkProvider({
1774
+ const p = testProvider({
1250
1775
  model: "m", url: "http://host/custom/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
1251
1776
  headers: { Authorization: "Bearer secret", "X-Title": "plurnk" },
1252
1777
  });
@@ -1262,32 +1787,34 @@ test("configured headers and url are sent verbatim", async () => {
1262
1787
 
1263
1788
  const retryCfg = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const };
1264
1789
 
1790
+ const stalledStreamResponse = (): Response => new Response(new ReadableStream({
1791
+ start(controller) {
1792
+ controller.enqueue(new TextEncoder().encode(
1793
+ 'data: {"id":"stalled","object":"chat.completion.chunk","created":1,"model":"m","choices":[{"index":0,"delta":{"content":"partial"},"finish_reason":null}]}\n\n',
1794
+ ));
1795
+ setTimeout(() => controller.close(), 100);
1796
+ },
1797
+ }), { status: 200 });
1798
+
1265
1799
  test("retry: a transient failure retries and a later success resolves", async () => {
1266
1800
  const calls = installFetchScript([
1801
+ { status: 408, retryAfter: 0 },
1802
+ { status: 409, retryAfter: 0 },
1267
1803
  { status: 429, retryAfter: 0 },
1268
1804
  { status: 503, retryAfter: 0 },
1269
1805
  { status: 200, chunks: [{ choices: [{ delta: { content: "ok" } }] }] },
1270
1806
  ]);
1271
- const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 3 });
1807
+ const p = testProvider({ ...retryCfg, retryAttempts: 4 });
1272
1808
  const res = await p.generate({ workerId: "r", messages: [] });
1273
1809
  assert.equal(res.assistant.content, "ok");
1274
- assert.equal(calls.length, 3); // 429 → 503 → 200
1810
+ assert.equal(calls.length, 5); // 408 → 409 → 429 → 503 → 200
1275
1811
  });
1276
1812
 
1277
- test("streamed-body silence fails the exchange without replaying partial output", async () => {
1813
+ test("streamed-body silence retries and returns the retry's complete output", async () => {
1278
1814
  let calls = 0;
1279
1815
  mock.method(globalThis, "fetch", async () => {
1280
1816
  calls++;
1281
- if (calls === 1) {
1282
- return new Response(new ReadableStream({
1283
- start(controller) {
1284
- controller.enqueue(new TextEncoder().encode(
1285
- 'data: {"id":"first","object":"chat.completion.chunk","created":1,"model":"m","choices":[{"index":0,"delta":{"content":"partial"},"finish_reason":null}]}\n\n',
1286
- ));
1287
- setTimeout(() => controller.close(), 100);
1288
- },
1289
- }), { status: 200 });
1290
- }
1817
+ if (calls === 1) return stalledStreamResponse();
1291
1818
  return new Response(new ReadableStream({
1292
1819
  start(controller) {
1293
1820
  controller.enqueue(new TextEncoder().encode(
@@ -1297,7 +1824,30 @@ test("streamed-body silence fails the exchange without replaying partial output"
1297
1824
  },
1298
1825
  }), { status: 200 });
1299
1826
  });
1300
- const p = new AiSdkProvider({
1827
+ const p = testProvider({
1828
+ model: "m",
1829
+ url: "http://x/v1/chat/completions",
1830
+ fetchTimeoutMs: 5000,
1831
+ streamIdleTimeoutMs: 10,
1832
+ temperature: 0.2,
1833
+ repeatPenalty: 1.15,
1834
+ reasoning: { mode: "off", budget: null },
1835
+ retryAttempts: 1,
1836
+ source: "provider:test",
1837
+ });
1838
+ const result = await p.generate({ workerId: "r", messages: [] });
1839
+ assert.equal(result.assistant.content, "recovered", "the retry's complete output, not the stalled partial");
1840
+ assert.equal(calls, 2, "the stall retried once and the retry succeeded");
1841
+ mock.restoreAll();
1842
+ });
1843
+
1844
+ test("streamed-body silence does not replay when retries are disabled", async () => {
1845
+ let calls = 0;
1846
+ mock.method(globalThis, "fetch", async () => {
1847
+ calls++;
1848
+ return stalledStreamResponse();
1849
+ });
1850
+ const p = testProvider({
1301
1851
  model: "m",
1302
1852
  url: "http://x/v1/chat/completions",
1303
1853
  fetchTimeoutMs: 1000,
@@ -1305,18 +1855,196 @@ test("streamed-body silence fails the exchange without replaying partial output"
1305
1855
  temperature: 0.2,
1306
1856
  repeatPenalty: 1.15,
1307
1857
  reasoning: { mode: "off", budget: null },
1858
+ retryAttempts: 0,
1859
+ source: "provider:test",
1860
+ });
1861
+ await assert.rejects(
1862
+ p.generate({ workerId: "r", messages: [] }),
1863
+ (error: ProviderError) => error.kind === "network_failure"
1864
+ && error.problem.timeoutPhase === "stream_idle"
1865
+ && error.problem.timeoutMs === 10,
1866
+ );
1867
+ assert.equal(calls, 1, "zero retries permits exactly one provider request");
1868
+ mock.restoreAll();
1869
+ });
1870
+
1871
+ test("streamed-body silence exhausts the configured retry budget once", async () => {
1872
+ let calls = 0;
1873
+ mock.method(globalThis, "fetch", async () => {
1874
+ calls++;
1875
+ return stalledStreamResponse();
1876
+ });
1877
+ const p = testProvider({
1878
+ model: "m",
1879
+ url: "http://x/v1/chat/completions",
1880
+ fetchTimeoutMs: 5000,
1881
+ streamIdleTimeoutMs: 10,
1882
+ temperature: 0.2,
1883
+ repeatPenalty: 1.15,
1884
+ reasoning: { mode: "off", budget: null },
1308
1885
  retryAttempts: 1,
1309
1886
  source: "provider:test",
1310
1887
  });
1311
1888
  await assert.rejects(
1312
1889
  p.generate({ workerId: "r", messages: [] }),
1313
1890
  (error: ProviderError) => error.kind === "network_failure"
1314
- && /chunk timeout/i.test(error.message),
1891
+ && error.problem.attempts === 2
1892
+ && error.problem.retryExhausted === true
1893
+ && error.problem.retryable === false,
1894
+ );
1895
+ assert.equal(calls, 2, "one configured retry permits exactly two provider requests");
1896
+ mock.restoreAll();
1897
+ });
1898
+
1899
+ test("an attempt timeout retries within the larger operation deadline and settles every physical request", async () => {
1900
+ let calls = 0;
1901
+ mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
1902
+ calls++;
1903
+ if (calls > 1) {
1904
+ return new Response(sseStream([
1905
+ { choices: [{ delta: { content: "recovered" }, finish_reason: "stop" }] },
1906
+ ]), { status: 200 });
1907
+ }
1908
+ return await new Promise<Response>((_resolve, reject) => {
1909
+ const signal = init?.signal;
1910
+ signal?.addEventListener("abort", () => reject(signal.reason), { once: true });
1911
+ });
1912
+ });
1913
+ const connectivity = { operationTimeoutMs: 5_000 };
1914
+ const settled: Array<{ outcome: string }> = [];
1915
+ const p = testProvider({
1916
+ model: "m",
1917
+ url: "http://x/v1/chat/completions",
1918
+ fetchTimeoutMs: 10,
1919
+ streamIdleTimeoutMs: 0,
1920
+ temperature: 0.2,
1921
+ repeatPenalty: 1.15,
1922
+ reasoning: { mode: "off", budget: null },
1923
+ retryAttempts: 1,
1924
+ source: "provider:test",
1925
+ ...connectivity,
1926
+ });
1927
+ const result = await p.generate({
1928
+ workerId: "r",
1929
+ messages: [],
1930
+ observeRequest: async () => async (accounting) => { settled.push(accounting); },
1931
+ });
1932
+ assert.equal(result.assistant.content, "recovered");
1933
+ assert.equal(calls, 2);
1934
+ assert.deepEqual(settled.map(({ outcome }) => outcome), ["error", "response"]);
1935
+ assert.deepEqual(result.accounting.map(({ outcome }) => outcome), ["error", "response"]);
1936
+ mock.restoreAll();
1937
+ });
1938
+
1939
+ test("first-content silence retries independently of the stream-idle deadline", async () => {
1940
+ let calls = 0;
1941
+ mock.method(globalThis, "fetch", async () => {
1942
+ calls++;
1943
+ if (calls > 1) {
1944
+ return new Response(sseStream([
1945
+ { choices: [{ delta: { content: "recovered" }, finish_reason: "stop" }] },
1946
+ ]), { status: 200 });
1947
+ }
1948
+ return new Response(new ReadableStream({
1949
+ start(controller) {
1950
+ setTimeout(() => controller.close(), 100);
1951
+ },
1952
+ }), { status: 200 });
1953
+ });
1954
+ const connectivity = { operationTimeoutMs: 5_000, firstContentTimeoutMs: 10 };
1955
+ const p = testProvider({
1956
+ model: "m",
1957
+ url: "http://x/v1/chat/completions",
1958
+ fetchTimeoutMs: 5_000,
1959
+ streamIdleTimeoutMs: 0,
1960
+ temperature: 0.2,
1961
+ repeatPenalty: 1.15,
1962
+ reasoning: { mode: "off", budget: null },
1963
+ retryAttempts: 1,
1964
+ source: "provider:test",
1965
+ ...connectivity,
1966
+ });
1967
+ const result = await p.generate({ workerId: "r", messages: [] });
1968
+ assert.equal(result.assistant.content, "recovered");
1969
+ assert.equal(calls, 2);
1970
+ mock.restoreAll();
1971
+ });
1972
+
1973
+ test("operation-deadline exhaustion is a distinct non-retryable failure", async () => {
1974
+ let calls = 0;
1975
+ mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
1976
+ calls++;
1977
+ return await new Promise<Response>((_resolve, reject) => {
1978
+ const signal = init?.signal;
1979
+ signal?.addEventListener("abort", () => reject(signal.reason), { once: true });
1980
+ });
1981
+ });
1982
+ const connectivity = { operationTimeoutMs: 10 };
1983
+ const p = testProvider({
1984
+ model: "m",
1985
+ url: "http://x/v1/chat/completions",
1986
+ fetchTimeoutMs: 50,
1987
+ streamIdleTimeoutMs: 0,
1988
+ temperature: 0.2,
1989
+ repeatPenalty: 1.15,
1990
+ reasoning: { mode: "off", budget: null },
1991
+ retryAttempts: 3,
1992
+ source: "provider:test",
1993
+ ...connectivity,
1994
+ });
1995
+ await assert.rejects(
1996
+ p.generate({ workerId: "r", messages: [] }),
1997
+ (error: ProviderError) => error.kind === "deadline_exceeded"
1998
+ && error.status === 504
1999
+ && error.problem.retryable === false
2000
+ && error.problem.timeoutPhase === "operation"
2001
+ && error.problem.timeoutMs === 10
2002
+ && error.accounting.length === 1
2003
+ && error.accounting[0]?.outcome === "error",
1315
2004
  );
1316
2005
  assert.equal(calls, 1);
1317
2006
  mock.restoreAll();
1318
2007
  });
1319
2008
 
2009
+ test("the total generation deadline spans stalled-stream retry scheduling", async () => {
2010
+ let calls = 0;
2011
+ mock.method(globalThis, "fetch", async () => {
2012
+ calls++;
2013
+ if (calls > 1) {
2014
+ return new Response(new ReadableStream({
2015
+ start(controller) {
2016
+ controller.enqueue(new TextEncoder().encode(
2017
+ 'data: {"id":"second","object":"chat.completion.chunk","created":2,"model":"m","choices":[{"index":0,"delta":{"content":"late"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n',
2018
+ ));
2019
+ controller.close();
2020
+ },
2021
+ }), { status: 200 });
2022
+ }
2023
+ return stalledStreamResponse();
2024
+ });
2025
+ const p = testProvider({
2026
+ model: "m",
2027
+ url: "http://x/v1/chat/completions",
2028
+ fetchTimeoutMs: 5000,
2029
+ operationTimeoutMs: 50,
2030
+ streamIdleTimeoutMs: 10,
2031
+ temperature: 0.2,
2032
+ repeatPenalty: 1.15,
2033
+ reasoning: { mode: "off", budget: null },
2034
+ retryAttempts: 3,
2035
+ source: "provider:test",
2036
+ });
2037
+ const started = Date.now();
2038
+ await assert.rejects(
2039
+ p.generate({ workerId: "r", messages: [] }),
2040
+ (error: ProviderError) => error.kind === "deadline_exceeded"
2041
+ && error.problem.timeoutPhase === "operation",
2042
+ );
2043
+ assert.ok(Date.now() - started < 500, "the configured total deadline ends retry scheduling");
2044
+ assert.equal(calls, 1, "the total deadline expires before another request begins");
2045
+ mock.restoreAll();
2046
+ });
2047
+
1320
2048
  test("a zero stream-idle timeout permits a slow inter-chunk pause", async () => {
1321
2049
  mock.method(globalThis, "fetch", async () => new Response(new ReadableStream({
1322
2050
  async start(controller) {
@@ -1326,7 +2054,7 @@ test("a zero stream-idle timeout permits a slow inter-chunk pause", async () =>
1326
2054
  controller.close();
1327
2055
  },
1328
2056
  }), { status: 200 }));
1329
- const p = new AiSdkProvider({
2057
+ const p = testProvider({
1330
2058
  model: "m",
1331
2059
  url: "http://x/v1/chat/completions",
1332
2060
  fetchTimeoutMs: 1000,
@@ -1344,7 +2072,7 @@ test("a zero stream-idle timeout permits a slow inter-chunk pause", async () =>
1344
2072
  test("retry: exhausting the budget surfaces the classified ProviderError", async () => {
1345
2073
  const { ProviderError } = await import("./errors.ts");
1346
2074
  const calls = installFetchScript([{ status: 429, retryAfter: 0 }]); // always rate-limited
1347
- const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 2 });
2075
+ const p = testProvider({ ...retryCfg, retryAttempts: 2 });
1348
2076
  await assert.rejects(
1349
2077
  () => p.generate({ workerId: "r", messages: [] }),
1350
2078
  (err: unknown) => { assert.ok(err instanceof ProviderError); assert.equal(err.kind, "rate_limit"); return true; },
@@ -1357,7 +2085,7 @@ test("retry: a Retry-After HTTP-date is honored — a past date parses to a 0ms
1357
2085
  { status: 503, retryAfter: "Wed, 21 Oct 2015 07:28:00 GMT" }, // date form, in the past → max(0, past−now) = 0
1358
2086
  { status: 200, chunks: [{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }] },
1359
2087
  ]);
1360
- const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 1 });
2088
+ const p = testProvider({ ...retryCfg, retryAttempts: 1 });
1361
2089
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
1362
2090
  assert.equal(assistant.content, "ok");
1363
2091
  assert.equal(calls.length, 2); // initial 503 + one retry, no real wall-clock wait
@@ -1365,14 +2093,14 @@ test("retry: a Retry-After HTTP-date is honored — a past date parses to a 0ms
1365
2093
 
1366
2094
  test("retry: a terminal error (401 unauthorized) is never retried", async () => {
1367
2095
  const calls = installFetchScript([{ status: 401 }]);
1368
- const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 5 });
2096
+ const p = testProvider({ ...retryCfg, retryAttempts: 5 });
1369
2097
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), /401/);
1370
2098
  assert.equal(calls.length, 1); // terminal — no retry despite budget
1371
2099
  });
1372
2100
 
1373
2101
  test("retry: retryAttempts 0 surfaces the first transient failure immediately", async () => {
1374
2102
  const calls = installFetchScript([{ status: 503, retryAfter: 0 }]);
1375
- const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 0 });
2103
+ const p = testProvider({ ...retryCfg, retryAttempts: 0 });
1376
2104
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }));
1377
2105
  assert.equal(calls.length, 1); // no retry budget
1378
2106
  });
@@ -1380,7 +2108,7 @@ test("retry: retryAttempts 0 surfaces the first transient failure immediately",
1380
2108
  test("retry: a caller abort during backoff rejects promptly with no further attempt", async () => {
1381
2109
  const ac = new AbortController();
1382
2110
  const calls = installFetchScript([{ status: 503, retryAfter: 5 }]); // 5s backoff we never wait out
1383
- const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 3 });
2111
+ const p = testProvider({ ...retryCfg, retryAttempts: 3 });
1384
2112
  const promise = p.generate({ workerId: "r", messages: [], signal: ac.signal });
1385
2113
  await flush(); // attempt 0 fails, enters the backoff sleep
1386
2114
  assert.equal(calls.length, 1);
@@ -1391,23 +2119,29 @@ test("retry: a caller abort during backoff rejects promptly with no further atte
1391
2119
 
1392
2120
  // — Anthropic reasoning style (wire `thinking` parameter) —
1393
2121
 
1394
- test("reasoningStyle 'anthropic' maps the budget to the thinking param", async () => {
2122
+ test("reasoningStyle 'anthropic' maps an optional budget or the resolved reserve to the thinking param", async () => {
1395
2123
  // N>0 → enabled with budget_tokens
1396
- const capped = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: 4096 }, reasoningStyle: "anthropic" });
2124
+ const capped = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: 4096 }, reasoningStyle: "anthropic" });
1397
2125
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1398
2126
  await capped.generate({ workerId: "r", messages: [] });
1399
2127
  assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "enabled", budget_tokens: 4096 });
1400
2128
 
2129
+ mock.restoreAll();
2130
+ const unbudgeted = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 8192, reasoningReserve: { tokens: 2048 }, fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: null }, reasoningStyle: "anthropic" });
2131
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
2132
+ await unbudgeted.generate({ workerId: "r", messages: [] });
2133
+ assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "enabled", budget_tokens: 2048 });
2134
+
1401
2135
  mock.restoreAll();
1402
2136
  // 0 → explicit disabled
1403
- const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, reasoningStyle: "anthropic" });
2137
+ const off = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, reasoningStyle: "anthropic" });
1404
2138
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1405
2139
  await off.generate({ workerId: "r", messages: [] });
1406
2140
  assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "disabled" });
1407
2141
 
1408
2142
  mock.restoreAll();
1409
2143
  // -1 adaptive → omit (API default depth)
1410
- const adaptive = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, reasoningStyle: "anthropic" });
2144
+ const adaptive = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, reasoningStyle: "anthropic" });
1411
2145
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1412
2146
  await adaptive.generate({ workerId: "r", messages: [] });
1413
2147
  assert.equal("thinking" in JSON.parse(calls[0].init.body as string), false);
@@ -1425,14 +2159,14 @@ test("streaming:false posts without stream and parses the single JSON response",
1425
2159
  usage: { prompt_tokens: 3, completion_tokens: 1, total_tokens: 4 },
1426
2160
  }), { status: 200, headers: { "Content-Type": "application/json" } });
1427
2161
  });
1428
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
2162
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
1429
2163
  const res = await p.generate({ workerId: "r", messages: [] });
1430
2164
  const sent = JSON.parse(calls[0].body);
1431
2165
  assert.equal("stream" in sent, false); // no streaming flag
1432
2166
  assert.equal(res.assistant.content, "hello"); // content from message.content
1433
2167
  assert.equal(res.assistant.reasoning, "because"); // reasoning_content mapped
1434
2168
  assert.equal(res.assistant.finishReason, "stop");
1435
- assert.equal(res.assistant.usage.total, 4);
2169
+ assert.equal(res.accounting[0]?.usage?.totalTokens, 4);
1436
2170
  mock.restoreAll();
1437
2171
  });
1438
2172
 
@@ -1441,7 +2175,7 @@ const captureBase = { model: "m", url: "http://x/v1/chat/completions", fetchTime
1441
2175
 
1442
2176
  test("logprobs OFF by default: no wire request, no assistant.logprobs, no rawBody", async () => {
1443
2177
  const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
1444
- const p = new AiSdkProvider({ ...captureBase });
2178
+ const p = testProvider({ ...captureBase });
1445
2179
  const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
1446
2180
  const body = JSON.parse((calls[0].init.body as string));
1447
2181
  assert.equal("logprobs" in body, false);
@@ -1458,7 +2192,7 @@ test("logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw logpr
1458
2192
  { token: "no", logprob: -0.1, sampling_logprob: -0.1, top_logprobs: [{ token: "no", logprob: -0.1 }] },
1459
2193
  ] } }] };
1460
2194
  const calls = installFetch([chunk]);
1461
- const p = new AiSdkProvider({ ...captureBase, topLogprobs: 2 });
2195
+ const p = testProvider({ ...captureBase, topLogprobs: 2 });
1462
2196
  const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
1463
2197
  const body = JSON.parse((calls[0].init.body as string));
1464
2198
  assert.equal(body.logprobs, true);
@@ -1472,7 +2206,7 @@ test("logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw logpr
1472
2206
  test("rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob preserved", async () => {
1473
2207
  const wire = { model: "m", extra_top_level: "kept", choices: [{ message: { content: "no" }, finish_reason: "stop", logprobs: { content: [{ token: "no", logprob: -0.1, sampling_logprob: -0.1, token_id: 42 }] } }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } };
1474
2208
  installFetchJson(wire);
1475
- const p = new AiSdkProvider({ ...captureBase, streaming: false, topLogprobs: 0, rawBody: true });
2209
+ const p = testProvider({ ...captureBase, streaming: false, topLogprobs: 0, rawBody: true });
1476
2210
  const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
1477
2211
  assert.deepEqual(res.rawBody, wire); // verbatim
1478
2212
  assert.equal((res.rawBody as typeof wire).choices[0].logprobs.content[0].sampling_logprob, -0.1);
@@ -1483,7 +2217,7 @@ test("rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob prese
1483
2217
 
1484
2218
  test("caller sampling cannot forge logprobs (reserved keys): the env flag is the only control", async () => {
1485
2219
  const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
1486
- const p = new AiSdkProvider({ ...captureBase }); // logprobs OFF
2220
+ const p = testProvider({ ...captureBase }); // logprobs OFF
1487
2221
  await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }], sampling: { logprobs: true, top_logprobs: 5 } });
1488
2222
  const body = JSON.parse((calls[0].init.body as string));
1489
2223
  assert.equal("logprobs" in body, false); // sampling passthrough stripped it
@@ -1494,7 +2228,7 @@ test("caller sampling cannot forge logprobs (reserved keys): the env flag is the
1494
2228
  // — turn coordinate headers ({§lifecycle-terms}): same gate as every first-party signal —
1495
2229
 
1496
2230
  test("workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under the first-party gate", async () => {
1497
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
2231
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1498
2232
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1499
2233
  await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
1500
2234
  const headers = new Headers(calls[0].init.headers);
@@ -1504,7 +2238,7 @@ test("workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under the firs
1504
2238
  });
1505
2239
 
1506
2240
  test("third-party providers structurally DROP the coordinate (gate off by default)", async () => {
1507
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
2241
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1508
2242
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1509
2243
  await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
1510
2244
  const headers = new Headers(calls[0].init.headers);
@@ -1514,7 +2248,7 @@ test("third-party providers structurally DROP the coordinate (gate off by defaul
1514
2248
  });
1515
2249
 
1516
2250
  test("coordinates are 1-based — 0/absent/empty emit no header", async () => {
1517
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
2251
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1518
2252
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1519
2253
  await p.generate({ workerId: "r", messages: [], workspaceId: "", loop: 0, turn: 0 });
1520
2254
  const headers = new Headers(calls[0].init.headers);
@@ -1528,18 +2262,18 @@ test("coordinates are 1-based — 0/absent/empty emit no header", async () => {
1528
2262
 
1529
2263
  test("reserves derive from the detected window; absolutes stand alone; null window + percent = no claim", () => {
1530
2264
  const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
1531
- const derived = new AiSdkProvider({ ...base, contextWindow: 49152, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 } });
2265
+ const derived = testProvider({ ...base, contextWindow: 49152, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 } });
1532
2266
  assert.equal(derived.reasoningReserve, 4915); // jennifer/turboderp: 10% of 49152
1533
2267
  assert.equal(derived.completionReserve, 12288); // 25% of 49152
1534
- const pinned = new AiSdkProvider({ ...base, contextWindow: null, reasoningReserve: { tokens: 4096 }, completionReserve: { percent: 0.25 } });
2268
+ const pinned = testProvider({ ...base, contextWindow: null, reasoningReserve: { tokens: 4096 }, completionReserve: { percent: 0.25 } });
1535
2269
  assert.equal(pinned.reasoningReserve, 4096); // absolute pin needs no window
1536
2270
  assert.equal(pinned.completionReserve, null); // percent without a window = underivable
1537
- const legacy = new AiSdkProvider({ ...base, contextWindow: 49152 });
2271
+ const legacy = testProvider({ ...base, contextWindow: 49152 });
1538
2272
  assert.equal(legacy.reasoningReserve, null); // out-of-date sibling: no claim
1539
2273
  });
1540
2274
 
1541
2275
  test("router-owned tuning: tuningFloors:false drops the temperature/penalty floors, caller sampling still rides", async () => {
1542
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0, tuningFloors: false });
2276
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0, tuningFloors: false });
1543
2277
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1544
2278
  await p.generate({ workerId: "r", messages: [], sampling: { temperature: 0.9 } });
1545
2279
  const body = JSON.parse(calls[0].init.body as string);
@@ -1547,25 +2281,148 @@ test("router-owned tuning: tuningFloors:false drops the temperature/penalty floo
1547
2281
  assert.equal("frequency_penalty" in body, false); // the floor is suppressed; the router owns tuning
1548
2282
  });
1549
2283
 
1550
- // -- prompt-cache affinity (workerId -> prompt_cache_key) --
2284
+ // -- {§provider-cache-affinity} / {§provider-cache-write-policy} --
1551
2285
 
1552
- test("promptCacheKey on: body sends prompt_cache_key = workerId (serverless replica affinity)", async () => {
1553
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
2286
+ test("a compatible route's declared body affinity is managed by workerId", async () => {
2287
+ const p = testProvider({
2288
+ model: "m",
2289
+ url: "http://x/v1/chat/completions",
2290
+ fetchTimeoutMs: 5000,
2291
+ temperature: 0.2,
2292
+ repeatPenalty: 1.15,
2293
+ reasoning: { mode: "off", budget: null },
2294
+ retryAttempts: 0,
2295
+ cacheAffinity: { target: "body", name: "prompt_cache_key" },
2296
+ });
1554
2297
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1555
- await p.generate({ workerId: "worker-abc", messages: [] });
2298
+ await p.generate({ workerId: "worker-abc", messages: [], sampling: { prompt_cache_key: "hijack" } });
1556
2299
  assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc");
1557
2300
  });
1558
2301
 
1559
- test("promptCacheKey off (default): no prompt_cache_key on the wire", async () => {
1560
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
2302
+ test("an undeclared compatible route receives no guessed cache field", async () => {
2303
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1561
2304
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1562
2305
  await p.generate({ workerId: "worker-abc", messages: [] });
1563
2306
  assert.equal("prompt_cache_key" in JSON.parse(calls[0].init.body as string), false);
1564
2307
  });
1565
2308
 
1566
- test("prompt_cache_key is managed: caller sampling cannot forge/override the affinity key", async () => {
1567
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
2309
+ test("a compatible route's declared header affinity composes with static headers", async () => {
2310
+ const p = testProvider({
2311
+ model: "m",
2312
+ url: "http://x/v1/chat/completions",
2313
+ headers: { Authorization: "Bearer key" },
2314
+ fetchTimeoutMs: 5000,
2315
+ temperature: 0.2,
2316
+ repeatPenalty: 1.15,
2317
+ reasoning: { mode: "off", budget: null },
2318
+ retryAttempts: 0,
2319
+ cacheAffinity: { target: "header", name: "x-grok-conv-id" },
2320
+ });
1568
2321
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1569
- await p.generate({ workerId: "worker-abc", messages: [], sampling: { prompt_cache_key: "hijack" } });
1570
- assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc"); // managed wins
2322
+ await p.generate({ workerId: "worker-abc", messages: [] });
2323
+ const headers = new Headers(calls[0].init.headers);
2324
+ assert.equal(headers.get("authorization"), "Bearer key");
2325
+ assert.equal(headers.get("x-grok-conv-id"), "worker-abc");
2326
+ });
2327
+
2328
+ test("native request projections compose reasoning visibility, affinity, and system cache control", async () => {
2329
+ let request: Record<string, unknown> | undefined;
2330
+ const usage = {
2331
+ inputTokens: { total: 2, noCache: 2, cacheRead: 0, cacheWrite: 0 },
2332
+ outputTokens: { total: 1, text: 1, reasoning: 0 },
2333
+ };
2334
+ const languageModel = {
2335
+ specificationVersion: "v4",
2336
+ provider: "native.test",
2337
+ modelId: "native-cache",
2338
+ supportedUrls: {},
2339
+ doGenerate: async (options: Record<string, unknown>) => {
2340
+ request = options;
2341
+ return {
2342
+ content: [{ type: "text", text: "ok" }],
2343
+ finishReason: { unified: "stop", raw: "stop" },
2344
+ usage,
2345
+ response: { id: "response", modelId: "native-cache" },
2346
+ warnings: [],
2347
+ };
2348
+ },
2349
+ doStream: async () => { throw new Error("streaming is not under test"); },
2350
+ } as unknown as LanguageModel;
2351
+ const p = testProvider({
2352
+ model: "native-cache",
2353
+ languageModel,
2354
+ fetchTimeoutMs: 5000,
2355
+ temperature: 0.2,
2356
+ repeatPenalty: 1.15,
2357
+ reasoning: { mode: "adaptive", budget: null },
2358
+ retryAttempts: 0,
2359
+ streaming: false,
2360
+ cacheAffinity: { target: "provider-option", provider: "openai", name: "promptCacheKey" },
2361
+ reasoningResponseProviderOptions: {
2362
+ google: { thinkingConfig: { includeThoughts: true } },
2363
+ },
2364
+ systemCacheProviderOptions: {
2365
+ anthropic: { cacheControl: { type: "ephemeral" } },
2366
+ },
2367
+ });
2368
+ await p.generate({
2369
+ workerId: "worker-native",
2370
+ messages: [
2371
+ { role: "system", content: "stable definition" },
2372
+ { role: "system", content: "stable policy" },
2373
+ { role: "user", content: "changing packet" },
2374
+ ],
2375
+ });
2376
+
2377
+ assert.deepEqual(request?.providerOptions, {
2378
+ google: { thinkingConfig: { includeThoughts: true } },
2379
+ openai: { promptCacheKey: "worker-native" },
2380
+ });
2381
+ assert.deepEqual(request?.prompt, [
2382
+ { role: "system", content: "stable definition", providerOptions: undefined },
2383
+ {
2384
+ role: "system",
2385
+ content: "stable policy",
2386
+ providerOptions: { anthropic: { cacheControl: { type: "ephemeral" } } },
2387
+ },
2388
+ { role: "user", content: [{ type: "text", text: "changing packet" }], providerOptions: undefined },
2389
+ ]);
2390
+ });
2391
+
2392
+ test("native AI SDK reasoning turns on without an operator token budget", async () => {
2393
+ let request: Record<string, unknown> | undefined;
2394
+ const languageModel = {
2395
+ specificationVersion: "v4",
2396
+ provider: "native.test",
2397
+ modelId: "native-reasoning",
2398
+ supportedUrls: {},
2399
+ doGenerate: async (options: Record<string, unknown>) => {
2400
+ request = options;
2401
+ return {
2402
+ content: [{ type: "reasoning", text: "consider" }, { type: "text", text: "ok" }],
2403
+ finishReason: { unified: "stop", raw: "stop" },
2404
+ usage: {
2405
+ inputTokens: { total: 2, noCache: 2, cacheRead: 0, cacheWrite: 0 },
2406
+ outputTokens: { total: 2, text: 1, reasoning: 1 },
2407
+ },
2408
+ response: { id: "response", modelId: "native-reasoning" },
2409
+ warnings: [],
2410
+ };
2411
+ },
2412
+ doStream: async () => { throw new Error("streaming is not under test"); },
2413
+ } as unknown as LanguageModel;
2414
+ const p = testProvider({
2415
+ model: "native-reasoning",
2416
+ languageModel,
2417
+ fetchTimeoutMs: 5000,
2418
+ temperature: 0.2,
2419
+ repeatPenalty: 1.15,
2420
+ reasoning: { mode: "on", budget: null },
2421
+ retryAttempts: 0,
2422
+ streaming: false,
2423
+ });
2424
+ const response = await p.generate({ workerId: "worker-native", messages: [{ role: "user", content: "hello" }] });
2425
+
2426
+ assert.equal(request?.reasoning, "medium");
2427
+ assert.equal(response.assistant.reasoning, "consider");
1571
2428
  });