@plurnk/plurnk-providers 1.3.12 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (139) hide show
  1. package/.env.defaults +47 -38
  2. package/README.md +68 -4
  3. package/SPEC.md +245 -62
  4. package/dist/AiSdkProvider.d.ts +19 -5
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +264 -156
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +15 -17
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +27 -10
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +8 -2
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +41 -8
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/ProviderRegistry.d.ts +4 -1
  17. package/dist/ProviderRegistry.d.ts.map +1 -1
  18. package/dist/ProviderRegistry.js +7 -3
  19. package/dist/ProviderRegistry.js.map +1 -1
  20. package/dist/accounting.d.ts +3 -0
  21. package/dist/accounting.d.ts.map +1 -0
  22. package/dist/accounting.js +84 -0
  23. package/dist/accounting.js.map +1 -0
  24. package/dist/aiSdkTransport.d.ts +5 -2
  25. package/dist/aiSdkTransport.d.ts.map +1 -1
  26. package/dist/aiSdkTransport.js +91 -5
  27. package/dist/aiSdkTransport.js.map +1 -1
  28. package/dist/catalogProvider.d.ts +5 -2
  29. package/dist/catalogProvider.d.ts.map +1 -1
  30. package/dist/catalogProvider.js +24 -11
  31. package/dist/catalogProvider.js.map +1 -1
  32. package/dist/compatibleProvider.d.ts.map +1 -1
  33. package/dist/compatibleProvider.js +11 -4
  34. package/dist/compatibleProvider.js.map +1 -1
  35. package/dist/cost.d.ts +11 -0
  36. package/dist/cost.d.ts.map +1 -0
  37. package/dist/cost.js +61 -0
  38. package/dist/cost.js.map +1 -0
  39. package/dist/discover.d.ts +2 -0
  40. package/dist/discover.d.ts.map +1 -1
  41. package/dist/discover.js +15 -9
  42. package/dist/discover.js.map +1 -1
  43. package/dist/env.d.ts +4 -6
  44. package/dist/env.d.ts.map +1 -1
  45. package/dist/env.js +27 -29
  46. package/dist/env.js.map +1 -1
  47. package/dist/errors.d.ts +27 -0
  48. package/dist/errors.d.ts.map +1 -0
  49. package/dist/errors.js +152 -0
  50. package/dist/errors.js.map +1 -0
  51. package/dist/index.d.ts +12 -6
  52. package/dist/index.d.ts.map +1 -1
  53. package/dist/index.js +8 -5
  54. package/dist/index.js.map +1 -1
  55. package/dist/notices.d.ts +10 -0
  56. package/dist/notices.d.ts.map +1 -0
  57. package/dist/notices.js +11 -0
  58. package/dist/notices.js.map +1 -0
  59. package/dist/ollama.d.ts.map +1 -1
  60. package/dist/ollama.js +3 -3
  61. package/dist/ollama.js.map +1 -1
  62. package/dist/openai.d.ts +1 -1
  63. package/dist/openai.d.ts.map +1 -1
  64. package/dist/promptTokens.d.ts +4 -0
  65. package/dist/promptTokens.d.ts.map +1 -0
  66. package/dist/promptTokens.js +32 -0
  67. package/dist/promptTokens.js.map +1 -0
  68. package/dist/sdkModels.d.ts +2 -0
  69. package/dist/sdkModels.d.ts.map +1 -1
  70. package/dist/sdkModels.js +17 -6
  71. package/dist/sdkModels.js.map +1 -1
  72. package/dist/types.d.ts +52 -16
  73. package/dist/types.d.ts.map +1 -1
  74. package/dist/types.js +1 -1
  75. package/dist/types.js.map +1 -1
  76. package/dist/usage.d.ts +4 -0
  77. package/dist/usage.d.ts.map +1 -1
  78. package/dist/usage.js +64 -19
  79. package/dist/usage.js.map +1 -1
  80. package/dist/warnings.js +0 -0
  81. package/dist/warnings.js.map +1 -1
  82. package/package.json +15 -10
  83. package/src/AiSdkProvider.test.ts +750 -169
  84. package/src/AiSdkProvider.ts +354 -200
  85. package/src/Mock.test.ts +29 -14
  86. package/src/Mock.ts +36 -15
  87. package/src/Pool.test.ts +43 -6
  88. package/src/Pool.ts +56 -10
  89. package/src/ProviderRegistry.test.ts +158 -9
  90. package/src/ProviderRegistry.ts +19 -6
  91. package/src/accounting.test.ts +58 -0
  92. package/src/accounting.ts +88 -0
  93. package/src/aiSdkTransport.ts +101 -8
  94. package/src/boundaries.test.ts +9 -3
  95. package/src/catalogProvider.test.ts +43 -15
  96. package/src/catalogProvider.ts +32 -16
  97. package/src/compatibleProvider.test.ts +96 -0
  98. package/src/compatibleProvider.ts +15 -6
  99. package/src/cost.test.ts +64 -0
  100. package/src/cost.ts +78 -0
  101. package/src/defaults.test.ts +1 -0
  102. package/src/discover.test.ts +48 -7
  103. package/src/discover.ts +31 -21
  104. package/src/env.test.ts +38 -48
  105. package/src/env.ts +43 -40
  106. package/src/errors.test.ts +148 -0
  107. package/src/errors.ts +208 -0
  108. package/src/index.ts +30 -8
  109. package/src/lexicon-guard.test.ts +6 -6
  110. package/src/notices.ts +22 -0
  111. package/src/ollama.test.ts +64 -0
  112. package/src/ollama.ts +6 -3
  113. package/src/openai.ts +3 -0
  114. package/src/promptTokens.ts +41 -0
  115. package/src/sdkModels.test.ts +29 -3
  116. package/src/sdkModels.ts +19 -11
  117. package/src/types.ts +125 -64
  118. package/src/usage.test.ts +24 -5
  119. package/src/usage.ts +72 -21
  120. package/src/warnings.test.ts +10 -10
  121. package/src/warnings.ts +0 -0
  122. package/dist/OpenAICompat.d.ts +0 -76
  123. package/dist/OpenAICompat.d.ts.map +0 -1
  124. package/dist/OpenAICompat.js +0 -555
  125. package/dist/OpenAICompat.js.map +0 -1
  126. package/dist/openaiStream.d.ts +0 -47
  127. package/dist/openaiStream.d.ts.map +0 -1
  128. package/dist/openaiStream.js +0 -280
  129. package/dist/openaiStream.js.map +0 -1
  130. package/dist/standardProviders.d.ts +0 -31
  131. package/dist/standardProviders.d.ts.map +0 -1
  132. package/dist/standardProviders.js +0 -518
  133. package/dist/standardProviders.js.map +0 -1
  134. package/dist/telemetry.d.ts +0 -24
  135. package/dist/telemetry.d.ts.map +0 -1
  136. package/dist/telemetry.js +0 -85
  137. package/dist/telemetry.js.map +0 -1
  138. package/src/telemetry.test.ts +0 -69
  139. package/src/telemetry.ts +0 -116
@@ -1,7 +1,9 @@
1
1
  import test, { mock } from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
3
  import AiSdkProvider, { effortFromBudget } from "./AiSdkProvider.ts";
4
- import { ProviderError } from "./telemetry.ts";
4
+ import { ProviderError } from "./errors.ts";
5
+ import { authoritativeChargeNormalizer } from "./accounting.ts";
6
+ import type { LanguageModel } from "ai";
5
7
 
6
8
  // Build a fake fetch returning a one-chunk SSE stream, capturing the request
7
9
  // so tests can assert what the spine sent on the wire.
@@ -71,7 +73,7 @@ const injectedBase = {
71
73
  reasoning: { mode: "off" as const, budget: null },
72
74
  };
73
75
 
74
- test("#608: per-instance fetch owns streaming and buffered requests", async () => {
76
+ test("per-instance fetch owns streaming and buffered requests", async () => {
75
77
  const calls: Array<{ input: string | URL | Request; init?: RequestInit }> = [];
76
78
  const streamingFetch: typeof globalThis.fetch = async (input, init) => {
77
79
  calls.push({ input, init });
@@ -105,7 +107,7 @@ test("#608: per-instance fetch owns streaming and buffered requests", async () =
105
107
  assert.equal(JSON.parse(String(calls[1].init?.body)).stream, undefined);
106
108
  });
107
109
 
108
- test("#608: caller cancellation and provider timeout reach an injected fetch", async () => {
110
+ test("caller cancellation and provider timeout reach an injected fetch", async () => {
109
111
  const pendingFetch: typeof globalThis.fetch = async (_input, init) => {
110
112
  init?.signal?.throwIfAborted();
111
113
  return new Promise((_resolve, reject) => {
@@ -126,7 +128,7 @@ test("#608: caller cancellation and provider timeout reach an injected fetch", a
126
128
  );
127
129
  });
128
130
 
129
- test("#608: per-instance fetch owns tokenization and retry attempts", async () => {
131
+ test("per-instance fetch owns tokenization and retry attempts", async () => {
130
132
  const calls: string[] = [];
131
133
  let generationAttempts = 0;
132
134
  const providerFetch: typeof globalThis.fetch = async (input) => {
@@ -192,7 +194,7 @@ const installFetchScript = (responses: ScriptedResponse[]) => {
192
194
  const flush = () => new Promise<void>((r) => setImmediate(r));
193
195
 
194
196
  import { resetEmittedWarnings } from "./warnings.ts";
195
- test.afterEach(() => { mock.restoreAll(); resetEmittedWarnings(); }); // #40: warning-asserting tests stay order-independent
197
+ test.afterEach(() => { mock.restoreAll(); resetEmittedWarnings(); });
196
198
 
197
199
  test("effortFromBudget: maps budget to tiers", () => {
198
200
  assert.equal(effortFromBudget(1), "low");
@@ -202,7 +204,7 @@ test("effortFromBudget: maps budget to tiers", () => {
202
204
  assert.equal(effortFromBudget(4001), "high");
203
205
  });
204
206
 
205
- test("#543: a 524 Cloudflare edge timeout fails fast - not retried despite retryAttempts", async () => {
207
+ test("a 524 Cloudflare edge timeout fails fast - not retried despite retryAttempts", async () => {
206
208
  const calls = installFetchScript([{ status: 524, retryAfter: 120 }]);
207
209
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
208
210
  await assert.rejects(p.generate({ workerId: "r", messages: [] }));
@@ -211,7 +213,7 @@ test("#543: a 524 Cloudflare edge timeout fails fast - not retried despite retry
211
213
  mock.restoreAll();
212
214
  });
213
215
 
214
- test("#548: a 422 grammar_invalid is a failed exchange, not transport replay policy", async () => {
216
+ test("a 422 grammar_invalid is a failed exchange, not transport replay policy", async () => {
215
217
  const body = JSON.stringify({ error: { message: "non-conforming emission rejected: ...", type: "grammar_invalid" } });
216
218
  const calls = installFetchScript([{ status: 422, body }]);
217
219
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 2 });
@@ -237,43 +239,59 @@ test("an SSE error frame is a failed exchange, not an empty completion", async (
237
239
  assert.equal(calls.length, 1);
238
240
  });
239
241
 
240
- test("#539: a trailing eos_token (--special EOG leak) is stripped from content", async () => {
242
+ test("a trailing eos_token (--special EOG leak) is stripped from content", async () => {
241
243
  installFetchJson({ model: "m", choices: [{ message: { content: "the answer<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
242
244
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
243
245
  const res = await p.generate({ workerId: "r", messages: [] });
244
246
  assert.equal(res.assistant.content, "the answer"); // trailing <eos> gone; packet + verdict see clean bytes
245
247
  });
246
248
 
247
- test("#539: without a probed eos_token the content passes through untouched", async () => {
249
+ test("without a probed eos_token the content passes through untouched", async () => {
248
250
  installFetchJson({ model: "m", choices: [{ message: { content: "keeps <eos> literally" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
249
251
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
250
252
  const res = await p.generate({ workerId: "r", messages: [] });
251
253
  assert.equal(res.assistant.content, "keeps <eos> literally"); // no eosText (a cloud backend) -> no strip
252
254
  });
253
255
 
254
- test("#539: only the TRAILING eos_token is stripped; a quoted one mid-body survives", async () => {
256
+ test("only the trailing eos_token is stripped; a quoted one mid-body survives", async () => {
255
257
  installFetchJson({ model: "m", choices: [{ message: { content: "quotes <eos> in the body<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 } });
256
258
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
257
259
  const res = await p.generate({ workerId: "r", messages: [] });
258
260
  assert.equal(res.assistant.content, "quotes <eos> in the body"); // only the tail goes
259
261
  });
260
262
 
261
- test("identity getters and defaults", () => {
263
+ test("identity getters and default prompt estimate", async () => {
262
264
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
263
265
  assert.equal(p.model, "m");
264
266
  assert.equal(p.contextWindow, null); // default
265
- assert.equal(p.countTokens(""), 0);
266
- assert.equal(p.countTokens("four"), 2); // default heuristic ceil(4/2) upper bound
267
- assert.equal(p.calculateCost({ prompt: 9, completion: 9, reasoning: 0, cached: 0, total: 18 }), 0); // default free
267
+ assert.deepEqual(
268
+ await p.countPromptTokens([{ role: "user", content: "漢漢漢" }]),
269
+ {
270
+ kind: "estimate",
271
+ tokens: 2,
272
+ source: "heuristic:chars2",
273
+ detail: "chars/2 over message content; provider request framing is unknown",
274
+ },
275
+ "chars/2 is explicitly an estimate; high-token-density Unicode prevents an upper-bound claim",
276
+ );
277
+ assert.equal(p.calculateCost({ prompt: 9, completion: 9, reasoning: 0, cached: 0, total: 18 }), 0); // current unknown-rate sentinel
268
278
  });
269
279
 
270
- test("injected countTokens and calculateCost are used", () => {
280
+ test("injected prompt measurement preserves provenance and calculateCost is used", async () => {
281
+ const seen: string[] = [];
271
282
  const p = new AiSdkProvider({
272
283
  model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
273
- countTokens: (t) => t.length,
284
+ countPromptTokens: (messages) => {
285
+ seen.push(...messages.map(({ content }) => content));
286
+ return { kind: "upper_bound", tokens: 7, source: "test:proven-bound" };
287
+ },
274
288
  calculateCost: (u) => u.total * 2,
275
289
  });
276
- assert.equal(p.countTokens("abc"), 3);
290
+ assert.deepEqual(
291
+ await p.countPromptTokens([{ role: "system", content: "system" }, { role: "user", content: "user" }]),
292
+ { kind: "upper_bound", tokens: 7, source: "test:proven-bound" },
293
+ );
294
+ assert.deepEqual(seen, ["system", "user"]);
277
295
  assert.equal(p.calculateCost({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 5 }), 10);
278
296
  });
279
297
 
@@ -293,6 +311,139 @@ test("generate maps a streamed response into ProviderResponse", async () => {
293
311
  assert.notEqual(assistantRaw, undefined);
294
312
  });
295
313
 
314
+ test("native SDK accounting metadata becomes a normalized charge in buffered and streamed responses", async (t) => {
315
+ const usage = {
316
+ inputTokens: { total: 2, noCache: 2, cacheRead: 0, cacheWrite: 0 },
317
+ outputTokens: { total: 1, text: 1, reasoning: 0 },
318
+ };
319
+ const providerMetadata = { openrouter: { usage: { cost: 0.00154935 } } };
320
+ const charge = {
321
+ kind: "authoritative",
322
+ amount: { amount: "0.00154935", currency: "USD" },
323
+ usdEquivalent: "0.00154935",
324
+ source: "OpenRouter response usage.cost",
325
+ };
326
+ const languageModel = {
327
+ specificationVersion: "v4",
328
+ provider: "openrouter.chat",
329
+ modelId: "router-test",
330
+ supportedUrls: {},
331
+ doGenerate: async () => ({
332
+ content: [{ type: "text", text: "ok" }],
333
+ finishReason: { unified: "stop", raw: "completed" },
334
+ usage,
335
+ providerMetadata,
336
+ response: { id: "response-buffered", modelId: "router-test" },
337
+ warnings: [],
338
+ }),
339
+ doStream: async () => ({
340
+ stream: new ReadableStream({
341
+ start(controller) {
342
+ controller.enqueue({ type: "stream-start", warnings: [] });
343
+ controller.enqueue({ type: "response-metadata", id: "response-streamed", modelId: "router-test" });
344
+ controller.enqueue({ type: "text-start", id: "text-1" });
345
+ controller.enqueue({ type: "text-delta", id: "text-1", delta: "ok" });
346
+ controller.enqueue({ type: "text-end", id: "text-1" });
347
+ controller.enqueue({
348
+ type: "finish",
349
+ finishReason: { unified: "stop", raw: "completed" },
350
+ usage,
351
+ providerMetadata,
352
+ });
353
+ controller.close();
354
+ },
355
+ }),
356
+ response: {},
357
+ }),
358
+ } as unknown as LanguageModel;
359
+ const config = {
360
+ model: "router-test",
361
+ languageModel,
362
+ fetchTimeoutMs: 5_000,
363
+ temperature: 0.2,
364
+ repeatPenalty: 1.15,
365
+ reasoning: { mode: "off" as const, budget: null },
366
+ retryAttempts: 0,
367
+ normalizeCharge: authoritativeChargeNormalizer("@openrouter/ai-sdk-provider"),
368
+ };
369
+
370
+ await t.test("buffered", async () => {
371
+ const response = await new AiSdkProvider({ ...config, streaming: false })
372
+ .generate({ workerId: "buffered", messages: [] });
373
+ assert.deepEqual(response.charge, charge);
374
+ });
375
+ await t.test("streamed", async () => {
376
+ const response = await new AiSdkProvider(config)
377
+ .generate({ workerId: "streamed", messages: [] });
378
+ assert.deepEqual(response.charge, charge);
379
+ });
380
+ });
381
+
382
+ test("compatible xAI wire usage becomes an exact tick charge without raw-body capture", async () => {
383
+ const p = new AiSdkProvider({
384
+ model: "grok-test",
385
+ url: "http://x/v1/chat/completions",
386
+ fetchTimeoutMs: 5_000,
387
+ temperature: 0.2,
388
+ repeatPenalty: 1.15,
389
+ reasoning: { mode: "off", budget: null },
390
+ retryAttempts: 0,
391
+ streaming: false,
392
+ normalizeCharge: authoritativeChargeNormalizer("@ai-sdk/xai"),
393
+ });
394
+ installFetchJson({
395
+ id: "response-1",
396
+ model: "grok-test",
397
+ choices: [{ message: { content: "ok" }, finish_reason: "stop" }],
398
+ usage: {
399
+ prompt_tokens: 2,
400
+ completion_tokens: 1,
401
+ total_tokens: 3,
402
+ cost_in_usd_ticks: 15_493_500,
403
+ },
404
+ });
405
+ const response = await p.generate({ workerId: "xai", messages: [] });
406
+ assert.deepEqual(response.charge, {
407
+ kind: "authoritative",
408
+ amount: { amount: "15493500", currency: "USDTICK" },
409
+ usdEquivalent: "0.00154935",
410
+ source: "xAI response usage.cost_in_usd_ticks",
411
+ });
412
+ assert.equal(response.rawBody, undefined);
413
+ });
414
+
415
+ test("streamed xAI final usage retains its exact tick charge", async () => {
416
+ const p = new AiSdkProvider({
417
+ model: "grok-test",
418
+ url: "http://x/v1/chat/completions",
419
+ fetchTimeoutMs: 5_000,
420
+ temperature: 0.2,
421
+ repeatPenalty: 1.15,
422
+ reasoning: { mode: "off", budget: null },
423
+ retryAttempts: 0,
424
+ normalizeCharge: authoritativeChargeNormalizer("@ai-sdk/xai"),
425
+ });
426
+ installFetch([
427
+ { choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] },
428
+ {
429
+ choices: [],
430
+ usage: {
431
+ prompt_tokens: 2,
432
+ completion_tokens: 1,
433
+ total_tokens: 3,
434
+ cost_in_usd_ticks: 15_493_500,
435
+ },
436
+ },
437
+ ]);
438
+ const response = await p.generate({ workerId: "xai", messages: [] });
439
+ assert.deepEqual(response.charge, {
440
+ kind: "authoritative",
441
+ amount: { amount: "15493500", currency: "USDTICK" },
442
+ usdEquivalent: "0.00154935",
443
+ source: "xAI response usage.cost_in_usd_ticks",
444
+ });
445
+ });
446
+
296
447
  test("generate surfaces and normalizes an out-of-set finish_reason", async () => {
297
448
  const warnings: Array<{ message: string; code?: string }> = [];
298
449
  mock.method(process, "emitWarning", (message: string | Error, options?: string | { code?: string }) => {
@@ -311,7 +462,88 @@ test("generate surfaces and normalizes an out-of-set finish_reason", async () =>
311
462
  }]);
312
463
  });
313
464
 
314
- test("generate translates a backend cap synonym to canonical length (#425)", async () => {
465
+ test("#161: a streamed resource interruption is a failed exchange with complete attempt evidence", async () => {
466
+ const calls = installFetch([
467
+ { model: "served-model", choices: [{ delta: { reasoning_content: "partial thought", content: "partial answer" } }] },
468
+ {
469
+ choices: [{ delta: {}, finish_reason: "insufficient_system_resource" }],
470
+ usage: { prompt_tokens: 7, completion_tokens: 5, total_tokens: 12 },
471
+ },
472
+ ]);
473
+ const provider = new AiSdkProvider({
474
+ ...injectedBase,
475
+ retryAttempts: 2,
476
+ rawBody: true,
477
+ });
478
+
479
+ await assert.rejects(
480
+ provider.generate({ workerId: "interrupted", messages: [{ role: "user", content: "hello" }] }),
481
+ (error: unknown) => {
482
+ assert.ok(error instanceof ProviderError);
483
+ assert.equal(error.kind, "resource_interrupted");
484
+ assert.equal(error.status, 503);
485
+ assert.equal(error.problem.stage, "provider-response");
486
+ assert.equal(error.problem.retryable, false);
487
+ assert.equal(error.problem.finishReason, "resource_interrupted");
488
+ assert.equal(error.problem.rawFinishReason, "insufficient_system_resource");
489
+ assert.equal(error.attempt?.assistant.content, "partial answer");
490
+ assert.equal(error.attempt?.assistant.reasoning, "partial thought");
491
+ assert.equal(error.attempt?.assistant.finishReason, "resource_interrupted");
492
+ assert.deepEqual(error.attempt?.assistant.usage, {
493
+ prompt: 7,
494
+ completion: 2,
495
+ reasoning: 3,
496
+ cached: 0,
497
+ total: 12,
498
+ });
499
+ assert.equal(
500
+ (error.attempt?.assistantRaw as { rawFinishReason?: string }).rawFinishReason,
501
+ "insufficient_system_resource",
502
+ );
503
+ assert.ok(Array.isArray(error.attempt?.rawBody));
504
+ return true;
505
+ },
506
+ );
507
+ assert.equal(calls.length, 1, "a semantic interruption is not replayed as an HTTP failure");
508
+ });
509
+
510
+ test("#161: a buffered resource interruption preserves the successful wire response as failed-attempt evidence", async () => {
511
+ const wire = {
512
+ model: "served-model",
513
+ choices: [{
514
+ message: { content: "partial answer", reasoning_content: "partial thought" },
515
+ finish_reason: "insufficient_system_resource",
516
+ }],
517
+ usage: { prompt_tokens: 7, completion_tokens: 5, total_tokens: 12 },
518
+ };
519
+ const calls = installFetchJson(wire);
520
+ const provider = new AiSdkProvider({
521
+ ...injectedBase,
522
+ streaming: false,
523
+ retryAttempts: 2,
524
+ rawBody: true,
525
+ });
526
+
527
+ await assert.rejects(
528
+ provider.generate({ workerId: "interrupted", messages: [{ role: "user", content: "hello" }] }),
529
+ (error: unknown) => {
530
+ assert.ok(error instanceof ProviderError);
531
+ assert.equal(error.kind, "resource_interrupted");
532
+ assert.equal(error.attempt?.assistant.content, "partial answer");
533
+ assert.equal(error.attempt?.assistant.reasoning, "partial thought");
534
+ assert.equal(error.attempt?.assistant.finishReason, "resource_interrupted");
535
+ assert.deepEqual(error.attempt?.rawBody, wire);
536
+ assert.equal(
537
+ (error.attempt?.assistantRaw as { rawFinishReason?: string }).rawFinishReason,
538
+ "insufficient_system_resource",
539
+ );
540
+ return true;
541
+ },
542
+ );
543
+ assert.equal(calls.length, 1);
544
+ });
545
+
546
+ test("generate translates a backend cap synonym to canonical length", async () => {
315
547
  // gemini shouts MAX_TOKENS, anthropic says max_tokens -- both must reach core as
316
548
  // "length" so its truncation check (=== "length") is a cross-backend invariant.
317
549
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
@@ -320,7 +552,7 @@ test("generate translates a backend cap synonym to canonical length (#425)", asy
320
552
  assert.equal(assistant.finishReason, "length");
321
553
  });
322
554
 
323
- test("generate translates end_turn to canonical stop (#425)", async () => {
555
+ test("generate translates end_turn to canonical stop", async () => {
324
556
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
325
557
  installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "end_turn" }] }]);
326
558
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
@@ -339,10 +571,172 @@ test("generate aggregates reasoning deltas under multiple field names", async ()
339
571
  installFetch([{ choices: [{ delta: { reasoning_content: "be", thinking: "cause" } }] }]);
340
572
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
341
573
  assert.equal(assistant.reasoning, "because");
342
- assert.equal("reasoningEncrypted" in assistant, false); // open reasoning only -> field absent (#482)
574
+ assert.equal("reasoningEncrypted" in assistant, false); // open reasoning only -> field absent
575
+ });
576
+
577
+ test("{§provider-tagged-reasoning} explicit think-tags projects one streamed leading envelope and reclassifies usage", async () => {
578
+ const config = { ...injectedBase, reasoningResponseStyle: "think-tags" as const, rawBody: true };
579
+ const p = new AiSdkProvider(config);
580
+ installFetch([
581
+ { choices: [{ delta: { content: "<thi" } }] },
582
+ { choices: [{ delta: { content: "nk>12345</th" } }] },
583
+ { choices: [{ delta: { content: "ink>abcde" }, finish_reason: "stop" }] },
584
+ { usage: { prompt_tokens: 3, completion_tokens: 10, total_tokens: 13 } },
585
+ ]);
586
+
587
+ const response = await p.generate({ workerId: "tagged-stream", messages: [] });
588
+
589
+ assert.equal(response.assistant.reasoning, "12345");
590
+ assert.equal(response.assistant.content, "abcde");
591
+ assert.deepEqual(response.assistant.usage, {
592
+ prompt: 3,
593
+ completion: 5,
594
+ reasoning: 5,
595
+ cached: 0,
596
+ total: 13,
597
+ });
598
+ assert.deepEqual(
599
+ ((response.assistantRaw as { content: string; reasoning: string }).content),
600
+ "abcde",
601
+ );
602
+ assert.equal((response.assistantRaw as { reasoning: string }).reasoning, "12345");
603
+ assert.match(JSON.stringify(response.rawBody), /<thi/);
604
+ assert.match(JSON.stringify(response.rawBody), /nk>12345/);
605
+ });
606
+
607
+ test("{§provider-tagged-reasoning} explicit think-tags projects one buffered leading envelope", async () => {
608
+ installFetchJson({
609
+ model: "m",
610
+ choices: [{ message: { content: "<think>12345</think>abcde" }, finish_reason: "stop" }],
611
+ usage: { prompt_tokens: 3, completion_tokens: 10, total_tokens: 13 },
612
+ });
613
+ const config = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
614
+ const response = await new AiSdkProvider(config).generate({ workerId: "tagged-buffer", messages: [] });
615
+
616
+ assert.equal(response.assistant.reasoning, "12345");
617
+ assert.equal(response.assistant.content, "abcde");
618
+ assert.equal(response.assistant.usage.completion, 5);
619
+ assert.equal(response.assistant.usage.reasoning, 5);
620
+ });
621
+
622
+ test("{§provider-tagged-reasoning} tagged text does not overwrite itemized reasoning usage", async () => {
623
+ installFetchJson({
624
+ model: "m",
625
+ choices: [{ message: { content: "<think>12345</think>abcde" }, finish_reason: "stop" }],
626
+ usage: {
627
+ prompt_tokens: 3,
628
+ completion_tokens: 10,
629
+ total_tokens: 13,
630
+ completion_tokens_details: { reasoning_tokens: 3 },
631
+ },
632
+ });
633
+ const config = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
634
+ const response = await new AiSdkProvider(config).generate({ workerId: "tagged-itemized", messages: [] });
635
+
636
+ assert.equal(response.assistant.reasoning, "12345");
637
+ assert.equal(response.assistant.content, "abcde");
638
+ assert.equal(response.assistant.usage.completion, 7);
639
+ assert.equal(response.assistant.usage.reasoning, 3);
640
+ });
641
+
642
+ test("{§provider-tagged-reasoning} an unclosed capped envelope is wholly reasoning in streamed and buffered responses", async () => {
643
+ const config = { ...injectedBase, reasoningResponseStyle: "think-tags" as const };
644
+ installFetch([
645
+ { choices: [{ delta: { content: "<think>unfinished" }, finish_reason: "length" }] },
646
+ { usage: { prompt_tokens: 3, completion_tokens: 8, total_tokens: 11 } },
647
+ ]);
648
+ const streamed = await new AiSdkProvider(config).generate({ workerId: "tagged-capped-stream", messages: [] });
649
+ assert.equal(streamed.assistant.reasoning, "unfinished");
650
+ assert.equal(streamed.assistant.content, "");
651
+ assert.deepEqual(streamed.assistant.usage, {
652
+ prompt: 3,
653
+ completion: 0,
654
+ reasoning: 8,
655
+ cached: 0,
656
+ total: 11,
657
+ });
658
+
659
+ mock.restoreAll();
660
+ installFetchJson({
661
+ model: "m",
662
+ choices: [{ message: { content: "<think>unfinished" }, finish_reason: "length" }],
663
+ usage: { prompt_tokens: 3, completion_tokens: 8, total_tokens: 11 },
664
+ });
665
+ const bufferedConfig = { ...config, streaming: false };
666
+ const buffered = await new AiSdkProvider(bufferedConfig).generate({ workerId: "tagged-capped-buffer", messages: [] });
667
+ assert.equal(buffered.assistant.reasoning, "unfinished");
668
+ assert.equal(buffered.assistant.content, "");
669
+ assert.equal(buffered.assistant.usage.completion, 0);
670
+ assert.equal(buffered.assistant.usage.reasoning, 8);
343
671
  });
344
672
 
345
- test("#482 sealed relay reasoning (non-streamed): encrypted reasoning_details surface verbatim, text entries do not", async () => {
673
+ test("{§provider-tagged-reasoning} verbatim, non-leading, and structured-reasoning controls preserve literal tags", async () => {
674
+ installFetchJson({
675
+ model: "m",
676
+ choices: [{ message: { content: "<think>literal</think>answer" }, finish_reason: "stop" }],
677
+ usage: { prompt_tokens: 1, completion_tokens: 4, total_tokens: 5 },
678
+ });
679
+ const verbatim = await new AiSdkProvider({ ...injectedBase, streaming: false })
680
+ .generate({ workerId: "verbatim", messages: [] });
681
+ assert.equal(verbatim.assistant.content, "<think>literal</think>answer");
682
+ assert.equal(verbatim.assistant.reasoning, null);
683
+ assert.equal(verbatim.assistant.usage.completion, 4);
684
+
685
+ mock.restoreAll();
686
+ installFetchJson({
687
+ model: "m",
688
+ choices: [{ message: { content: "show <think>literal</think> exactly" }, finish_reason: "stop" }],
689
+ usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 },
690
+ });
691
+ const taggedConfig = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
692
+ const nonLeading = await new AiSdkProvider(taggedConfig)
693
+ .generate({ workerId: "non-leading", messages: [] });
694
+ assert.equal(nonLeading.assistant.content, "show <think>literal</think> exactly");
695
+ assert.equal(nonLeading.assistant.reasoning, null);
696
+
697
+ mock.restoreAll();
698
+ installFetchJson({
699
+ model: "m",
700
+ choices: [{ message: {
701
+ content: "<think>literal visible bytes</think>",
702
+ reasoning_content: "structured reasoning",
703
+ }, finish_reason: "stop" }],
704
+ usage: { prompt_tokens: 1, completion_tokens: 7, total_tokens: 8 },
705
+ });
706
+ const structured = await new AiSdkProvider(taggedConfig)
707
+ .generate({ workerId: "structured", messages: [] });
708
+ assert.equal(structured.assistant.content, "<think>literal visible bytes</think>");
709
+ assert.equal(structured.assistant.reasoning, "structured reasoning");
710
+ });
711
+
712
+ test("{§provider-tagged-reasoning} grammar evidence retains the exact pre-projection tagged sentence", async () => {
713
+ const content = "<think>🧠reason</think><<PLAN::PLAN\n<<SEND[200]:done:SEND";
714
+ const config = {
715
+ ...injectedBase,
716
+ contextWindow: 640,
717
+ reasoning: { mode: "adaptive" as const, budget: null },
718
+ reasoningResponseStyle: "think-tags" as const,
719
+ reasoningStyle: "think" as const,
720
+ grammarStyle: "llamacpp" as const,
721
+ };
722
+ installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
723
+
724
+ const response = await new AiSdkProvider(config).generate({
725
+ workerId: "tagged-grammar",
726
+ messages: [],
727
+ grammar: `root ::= ${JSON.stringify(content)}`,
728
+ });
729
+
730
+ assert.equal(response.assistant.reasoning, "🧠reason");
731
+ assert.equal(response.assistant.content, "<<PLAN::PLAN\n<<SEND[200]:done:SEND");
732
+ assert.deepEqual(response.grammarEvidence, {
733
+ input: content,
734
+ contentStart: [..."<think>🧠reason</think>"].length,
735
+ transported: true,
736
+ });
737
+ });
738
+
739
+ test("encrypted reasoning (non-streamed): encrypted entries normalize and text entries stay separate", async () => {
346
740
  // The live o4-mini-via-OpenRouter shape: reasoning null, one encrypted entry.
347
741
  installFetchJson({ model: "m", choices: [{ message: {
348
742
  content: "4", reasoning: null,
@@ -353,13 +747,14 @@ test("#482 sealed relay reasoning (non-streamed): encrypted reasoning_details su
353
747
  }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
354
748
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
355
749
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
356
- // item shape: wire `id` preserved, subtype from position (#482 widening)
750
+ // Wire detail ID is preserved; the assistant-message location supports the
751
+ // derived classification but supplies no downstream client entity ID.
357
752
  assert.deepEqual(assistant.reasoningEncrypted, [{ id: "rs_1", subtype: "message", encrypted: [{ data: "gAAAAABqBLOB", format: "openai-responses-v1" }] }]);
358
- assert.equal(assistant.reasoning, null); // sealed turn: nothing readable
753
+ assert.equal(assistant.reasoning, null); // Encrypted turn: nothing readable.
359
754
  assert.equal(assistant.content, "4");
360
755
  });
361
756
 
362
- test("#482 widening: distinct wire ids stay distinct items (a single-object shape would collide them)", async () => {
757
+ test("distinct encrypted-reasoning wire ids stay distinct items", async () => {
363
758
  installFetchJson({ model: "m", choices: [{ message: { content: "ok", reasoning: null, reasoning_details: [
364
759
  { type: "reasoning.encrypted", data: "AAA", format: "openai-responses-v1", id: "rs_1" },
365
760
  { type: "reasoning.encrypted", data: "BBB", format: "openai-responses-v1", id: "rs_2" },
@@ -370,7 +765,20 @@ test("#482 widening: distinct wire ids stay distinct items (a single-object shap
370
765
  assert.deepEqual(assistant.reasoningEncrypted?.map((i) => i.id), ["rs_1", "rs_2"]);
371
766
  });
372
767
 
373
- test("#482 sealed relay reasoning (streamed): chunked blob concatenates per entry index", async () => {
768
+ test("assistant-message location classifies encrypted reasoning without inventing a missing detail id", async () => {
769
+ installFetchJson({ model: "m", choices: [{ message: { content: "ok", reasoning_details: [
770
+ { type: "reasoning.encrypted", data: "OPAQUE", format: "openai-responses-v1", id: null, index: 0 },
771
+ ] }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
772
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
773
+ const { assistant } = await p.generate({ workerId: "r", messages: [] });
774
+ assert.deepEqual(assistant.reasoningEncrypted, [{
775
+ id: null,
776
+ subtype: "message",
777
+ encrypted: [{ data: "OPAQUE", format: "openai-responses-v1" }],
778
+ }]);
779
+ });
780
+
781
+ test("encrypted reasoning (streamed): chunked blob concatenates per entry index", async () => {
374
782
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
375
783
  installFetch([
376
784
  { choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "gAAAA", format: "openai-responses-v1", id: "rs_1", index: 0 }] } }] },
@@ -402,10 +810,10 @@ test("reasoningStyle 'effort' sends a reasoning_effort tier from the budget", as
402
810
  assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, "high");
403
811
  });
404
812
 
405
- test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS (#403 — literal is MiniMax-only), on sends the tier", async () => {
813
+ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS, on sends the tier", async () => {
406
814
  // expected === null → the field must be ABSENT from the wire body. Fireworks
407
815
  // 400s reasoning_effort='adaptive' for non-MiniMax models (wire-verified,
408
- // #403): adaptive = the backend's own default posture = omission.
816
+ // Adaptive = the backend's own default posture = omission.
409
817
  for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "on", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "on"; budget: number | null }, string | null]>) {
410
818
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
411
819
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
@@ -417,7 +825,39 @@ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS (#403 —
417
825
  }
418
826
  });
419
827
 
420
- test("the family temperature default rides every request; caller sampling overrides it (#30)", async () => {
828
+ test("{§deepseek-reasoning-request} #157: thinking_effort maps the complete DeepSeek reasoning contract", async () => {
829
+ const cases = [
830
+ [{ mode: "off", budget: null }, { thinking: { type: "disabled" } }],
831
+ [{ mode: "adaptive", budget: null }, {}],
832
+ [{ mode: "on", budget: 5000 }, { thinking: { type: "enabled" }, reasoning_effort: "high" }],
833
+ ] as const;
834
+ for (const [reasoning, expected] of cases) {
835
+ const p = new AiSdkProvider({
836
+ model: "m",
837
+ url: "http://x/v1/chat/completions",
838
+ fetchTimeoutMs: 5000,
839
+ temperature: 0.2,
840
+ repeatPenalty: 1.15,
841
+ reasoning,
842
+ retryAttempts: 0,
843
+ reasoningStyle: "thinking_effort",
844
+ });
845
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
846
+ await p.generate({
847
+ workerId: "r",
848
+ messages: [],
849
+ sampling: { thinking: { type: "disabled" }, reasoning_effort: "max" },
850
+ });
851
+ const body = JSON.parse(calls[0].init.body as string);
852
+ assert.deepEqual(
853
+ Object.fromEntries(Object.entries(body).filter(([key]) => key === "thinking" || key === "reasoning_effort")),
854
+ expected,
855
+ );
856
+ mock.restoreAll();
857
+ }
858
+ });
859
+
860
+ test("the family temperature default rides every request; caller sampling overrides it", async () => {
421
861
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
422
862
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
423
863
  await p.generate({ workerId: "r", messages: [] });
@@ -434,7 +874,7 @@ test("the family temperature default rides every request; caller sampling overri
434
874
  assert.equal(JSON.parse(calls[0].init.body as string).temperature, 0.2);
435
875
  });
436
876
 
437
- test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box default; never on cloud", async () => {
877
+ test("DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box default; never on cloud", async () => {
438
878
  const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off" as const, budget: null }, retryAttempts: 0 };
439
879
  // set + llamacpp -> the loop-breakers ride the wire
440
880
  const p = new AiSdkProvider({ ...base, grammarStyle: "llamacpp", dryMultiplier: 0.8, dryBase: 1.75, dryAllowedLength: 2, repeatLastN: 512 });
@@ -474,14 +914,14 @@ test("llamacpp grammar path: temperature default + the managed repeat-penalty fl
474
914
  assert.equal(body.repeat_penalty, 1.15);
475
915
  });
476
916
 
477
- test("#426: the repeat penalty rides EVERY request rail-off, keyed per backend (cloud degeneration guard)", async () => {
917
+ test("the repeat penalty rides every request rail-off, keyed per backend", async () => {
478
918
  // llama.cpp with NO grammar carries its key too (unconstrained local is guarded)
479
919
  const llama = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
480
920
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
481
921
  await llama.generate({ workerId: "r", messages: [] });
482
922
  assert.equal(JSON.parse(calls[0].init.body as string).repeat_penalty, 1.15);
483
923
  mock.restoreAll();
484
- // a `none`-style cloud backend WITH a frequency penalty gets frequency_penalty (OpenAI-standard, #426)
924
+ // A `none`-style cloud backend with a frequency penalty gets frequency_penalty.
485
925
  const cloud = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
486
926
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
487
927
  await cloud.generate({ workerId: "r", messages: [] });
@@ -520,7 +960,7 @@ test("sampling passthrough forwards caller params; managed + reserved keys win",
520
960
  assert.equal("id_slot" in body, false); // reserved slot key stripped
521
961
  });
522
962
 
523
- test("#477 sampling passthrough guards contract invariants: n/tools/caps stripped, platform knobs pass", async () => {
963
+ test("sampling passthrough guards contract invariants: n/tools/caps stripped, platform knobs pass", async () => {
524
964
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
525
965
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
526
966
  await p.generate({
@@ -531,7 +971,7 @@ test("#477 sampling passthrough guards contract invariants: n/tools/caps strippe
531
971
  n: 3, // breaks choices[0] atomicity -> stripped
532
972
  tools: [{ type: "function" }], tool_choice: "auto", // tools-in-body doctrine -> stripped
533
973
  modalities: ["text", "audio"], prediction: { type: "content" }, // text-only / decode semantics -> stripped
534
- max_tokens: 999999, max_completion_tokens: 999999, // envelope bypass (#425 cap) -> stripped
974
+ max_tokens: 999999, max_completion_tokens: 999999, // envelope bypass -> stripped
535
975
  seed: 42, user: "acct-7", service_tier: "flex", // platform/sampling intent -> pass
536
976
  },
537
977
  });
@@ -545,58 +985,134 @@ test("#477 sampling passthrough guards contract invariants: n/tools/caps strippe
545
985
  assert.equal(body.service_tier, "flex");
546
986
  });
547
987
 
548
- test("#488 postmortem: intent maps IDENTICALLY under a transported grammar — sanctioned channel coexists with rails", async () => {
549
- // The brief rails-win-the-channel clamp is REVERTED: closing the channel starved a
550
- // reasoning-tuned model into escaping mid-content (unconstrained, discarded, billed).
551
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
988
+ test("template reasoning returns the exact pre-projection grammar sentence ({§gbnf-response-observation})", async () => {
989
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
990
+ const grammarInput = "<|channel>thought\ncon🙂sider<channel|>x";
991
+ const calls = installFetch([{ choices: [{ delta: { content: grammarInput } }] }]);
992
+ const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(grammarInput)}` });
993
+ const body = JSON.parse(calls[0].init.body as string);
994
+ assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true });
995
+ assert.equal(body.reasoning_format, "none");
996
+ assert.equal(body.thinking_budget_tokens, 64);
997
+ assert.equal(body.grammar, `root ::= ${JSON.stringify(grammarInput)}`);
998
+ assert.equal(res.assistant.reasoning, "con🙂sider");
999
+ assert.equal(res.assistant.content, "x");
1000
+ assert.deepEqual(res.grammarEvidence, {
1001
+ input: grammarInput,
1002
+ contentStart: [..."<|channel>thought\ncon🙂sider<channel|>"].length,
1003
+ transported: true,
1004
+ });
1005
+ assert.equal(res.meta?.railsVerdict, undefined, "the provider represents evidence but does not grade itself");
1006
+ });
1007
+
1008
+ test("a verbatim template response remains exact evidence when it has no channel envelope", async () => {
1009
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1010
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1011
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
1012
+ const body = JSON.parse(calls[0].init.body as string);
1013
+ assert.equal(body.reasoning_format, "none");
1014
+ assert.deepEqual(res.grammarEvidence, { input: "x", contentStart: 0, transported: true });
1015
+ });
1016
+
1017
+ test("a template grammar preserves exact evidence when reasoning is disabled", async () => {
1018
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
552
1019
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
553
1020
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
554
1021
  const body = JSON.parse(calls[0].init.body as string);
555
- assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true }); // channel stays sanctioned under the grammar
556
- assert.equal(typeof body.grammar, "string"); // rails ride beside it
557
- // #488 per-request loud state: rail attachment + verdict on meta, drill-readable per turn
558
- assert.equal(res.meta?.railsAttached, true);
559
- assert.equal(res.meta?.railsVerdict, "accept");
1022
+ assert.deepEqual(body.chat_template_kwargs, { enable_thinking: false });
1023
+ assert.equal(body.reasoning_format, "none");
1024
+ assert.deepEqual(res.grammarEvidence, { input: "x", contentStart: 0, transported: true });
1025
+ });
1026
+
1027
+ test("an unexpectedly projected template response cannot claim pre-projection evidence", async () => {
1028
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1029
+ installFetch([{ choices: [{ delta: { reasoning_content: "reason", content: "x" } }] }]);
1030
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
1031
+ assert.equal(res.grammarEvidence, undefined);
1032
+ });
1033
+
1034
+ test("template reasoning preserves an empty grammar-required channel as exact evidence", async () => {
1035
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1036
+ const input = "<|channel>thought\n<channel|>x";
1037
+ const calls = installFetch([{ choices: [{ delta: { content: input } }] }]);
1038
+ const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(input)}` });
1039
+ const body = JSON.parse(calls[0].init.body as string);
1040
+ assert.equal(body.reasoning_format, "none");
1041
+ assert.equal(res.assistant.reasoning, null);
1042
+ assert.equal(res.assistant.content, "x");
1043
+ assert.deepEqual(res.grammarEvidence, {
1044
+ input,
1045
+ contentStart: [..."<|channel>thought\n<channel|>"].length,
1046
+ transported: true,
1047
+ });
560
1048
  });
561
1049
 
562
- test("#488 channel-escape detector: billed completion tokens vastly beyond visible channels attach grammar_unenforced", async () => {
1050
+ test("channel-escape detector: billed completion tokens vastly beyond visible channels attach grammar_unenforced", async () => {
563
1051
  // The run105 shape: tiny visible content, no reasoning, thousands billed — the decode
564
1052
  // escaped into a discarded reasoning block, unconstrained.
565
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
566
- installFetch([
1053
+ const chunks = [
567
1054
  { choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
568
1055
  { usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
569
- ]);
1056
+ ];
1057
+ const fetch: typeof globalThis.fetch = async (input, init) => {
1058
+ if (String(input).endsWith("/tokenize")) {
1059
+ const body = JSON.parse(String(init?.body)) as { content: string };
1060
+ return new Response(JSON.stringify({
1061
+ tokens: body.content.length === 0 ? [] : [1],
1062
+ }), { headers: { "content-type": "application/json" } });
1063
+ }
1064
+ return new Response(sseStream(chunks), { status: 200 });
1065
+ };
1066
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetch, tokenizeUrl: "http://x/tokenize", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
570
1067
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
571
- assert.equal(res.meta?.railsVerdict, "accept"); // the visible fragment conforms...
572
- const escape = res.telemetry?.find((e) => e.message?.includes("escaped the grammar") === true);
573
- assert.ok(escape, "escape telemetry attached");
1068
+ const escape = res.notices?.find((e) => e.message.includes("escaped the grammar"));
1069
+ assert.ok(escape, "escape notice attached");
574
1070
  assert.equal(escape!.kind, "grammar_unenforced");
575
1071
  assert.match(escape!.message ?? "", /5000 completion tokens billed/);
576
1072
  });
577
1073
 
578
- test("#488 loud state absent on grammarless calls; no escape event without a transported grammar", async () => {
1074
+ test("channel-escape state is absent without a transported grammar", async () => {
579
1075
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
580
1076
  installFetch([
581
1077
  { choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
582
1078
  { usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
583
1079
  ]);
584
1080
  const res = await p.generate({ workerId: "r", messages: [] }); // no grammar arg
585
- assert.equal(res.meta?.railsAttached, undefined);
586
- assert.equal(res.telemetry, undefined);
1081
+ assert.equal(res.grammarEvidence, undefined);
1082
+ assert.equal(res.notices, undefined);
587
1083
  });
588
1084
 
589
- test("reasoningStyle 'template' always emits enable_thinking mirroring budget != 0 — explicit false, never omitted", async () => {
590
- const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
1085
+ test("reasoningStyle 'template' sends llama-server activation, parser, and response-wide allowance", async () => {
1086
+ const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
591
1087
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
592
1088
  await on.generate({ workerId: "r", messages: [] });
593
- assert.deepEqual(JSON.parse(calls[0].init.body as string).chat_template_kwargs, { enable_thinking: true });
1089
+ let body = JSON.parse(calls[0].init.body as string);
1090
+ assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true });
1091
+ assert.equal(body.reasoning_format, "auto");
1092
+ assert.equal(body.thinking_budget_tokens, 64);
594
1093
 
595
1094
  mock.restoreAll();
596
- const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
1095
+ const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
597
1096
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
598
1097
  await off.generate({ workerId: "r", messages: [] });
599
- assert.deepEqual(JSON.parse(calls[0].init.body as string).chat_template_kwargs, { enable_thinking: false });
1098
+ body = JSON.parse(calls[0].init.body as string);
1099
+ assert.deepEqual(body.chat_template_kwargs, { enable_thinking: false });
1100
+ assert.equal(body.reasoning_format, "auto");
1101
+ assert.equal(body.thinking_budget_tokens, 0);
1102
+ });
1103
+
1104
+ test("reasoningStyle 'template' explicit budget tightens the reserve and cannot exceed it", async () => {
1105
+ const base = { model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 } as const, completionReserve: { tokens: 160 } as const, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryAttempts: 0, reasoningStyle: "template" as const };
1106
+ const p = new AiSdkProvider({ ...base, reasoning: { mode: "on", budget: 32 } });
1107
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1108
+ await p.generate({ workerId: "r", messages: [], sampling: { thinking_budget_tokens: 999, reasoning_format: "none" } });
1109
+ const body = JSON.parse(calls[0].init.body as string);
1110
+ assert.equal(body.thinking_budget_tokens, 32);
1111
+ assert.equal(body.reasoning_format, "auto");
1112
+ assert.throws(
1113
+ () => new AiSdkProvider({ ...base, reasoning: { mode: "on", budget: 65 } }),
1114
+ /REASONING_BUDGET \(65\) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE \(64\)/,
1115
+ );
600
1116
  });
601
1117
 
602
1118
  test("budget 0 suppresses effort and include_reasoning", async () => {
@@ -619,7 +1135,7 @@ test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", asy
619
1135
  assert.equal(JSON.parse(calls[0].init.body as string).include_reasoning, true);
620
1136
  });
621
1137
 
622
- // — grammar-constrained sampling (SPEC §13, issues #8/#9) —
1138
+ // — grammar-constrained sampling —
623
1139
 
624
1140
  test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor", async () => {
625
1141
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
@@ -640,106 +1156,81 @@ test("grammar transport 'none' (default): the grammar is never sent — no silen
640
1156
  assert.equal("response_format" in body, false);
641
1157
  });
642
1158
 
643
- // — grammar conformance OBSERVATION (SPEC §10.14, §13): a completed exchange always
644
- // returns; bytes flow; a non-accept verdict rides response.telemetry —
1159
+ // — exact pre-projection grammar evidence ({§gbnf-response-observation}) —
645
1160
 
646
1161
  const grammarProvider = () => new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", source: "provider:test" });
647
1162
  const streamingContent = (content: string) => installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
648
1163
 
649
- test("enforcement: conforming output passes through unchanged", async () => {
1164
+ test("an unsplit grammar response carries the exact observed sentence", async () => {
650
1165
  const p = grammarProvider();
651
1166
  streamingContent("ok");
652
- const { assistant } = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
653
- assert.equal(assistant.content, "ok");
654
- });
655
-
656
- test("observation: REJECTED output still returns — bytes present, verdict attached with position", async () => {
657
- const p = grammarProvider();
658
- streamingContent("no");
659
1167
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
660
- assert.equal(res.assistant.content, "no"); // bytes ALWAYS flow
661
- assert.equal(res.telemetry?.length, 1);
662
- const ev = res.telemetry![0];
663
- assert.equal(ev.kind, "grammar_unenforced");
664
- assert.equal(ev.source, "provider:test");
665
- assert.match(String(ev.message), /grammar not enforced: output rejected .* at code point 0/);
666
- assert.equal(ev.position, 0); // divergence offset for consumer policy
667
- });
668
-
669
- test("observation: an incomplete (valid prefix, never terminated) also returns with the verdict", async () => {
670
- const p = grammarProvider();
671
- streamingContent("ok");
672
- const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok" "!"' });
673
1168
  assert.equal(res.assistant.content, "ok");
674
- assert.equal(res.telemetry?.length, 1);
675
- assert.match(String(res.telemetry![0].message), /incomplete match .* never terminated/);
676
- assert.equal(res.telemetry![0].position, 2);
1169
+ assert.deepEqual(res.grammarEvidence, {
1170
+ input: "ok",
1171
+ contentStart: 0,
1172
+ transported: true,
1173
+ });
677
1174
  });
678
1175
 
679
- test("observation: conforming output attaches NO telemetry", async () => {
1176
+ test("the provider returns rejected or incomplete bytes as evidence without grading them", async () => {
680
1177
  const p = grammarProvider();
681
- streamingContent("ok");
1178
+ streamingContent("no");
682
1179
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
683
- assert.equal(res.telemetry, undefined);
1180
+ assert.equal(res.assistant.content, "no");
1181
+ assert.deepEqual(res.grammarEvidence, { input: "no", contentStart: 0, transported: true });
1182
+ assert.equal(res.notices, undefined);
1183
+ assert.equal(res.meta?.railsVerdict, undefined);
684
1184
  });
685
1185
 
686
- test("observation: empty content under a non-empty grammar returns with the verdict (the 'content never arrives' leak, observed)", async () => {
1186
+ test("empty unsplit content remains exact grammar evidence", async () => {
687
1187
  const p = grammarProvider();
688
- installFetch([{ choices: [{ delta: {}, finish_reason: "stop" }] }]); // no content delta → ""
1188
+ installFetch([{ choices: [{ delta: {}, finish_reason: "stop" }] }]);
689
1189
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
690
1190
  assert.equal(res.assistant.content, "");
691
- assert.equal(res.telemetry?.[0].kind, "grammar_unenforced");
1191
+ assert.deepEqual(res.grammarEvidence, { input: "", contentStart: 0, transported: true });
692
1192
  });
693
1193
 
694
- test("enforcement: when no grammar is sent (grammarStyle 'none'), output is NOT validated — no wire fields, no error (SPEC )", async () => {
1194
+ test("grammarStyle 'none' produces no grammar observation", async () => {
695
1195
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // grammarStyle defaults to "none"
696
1196
  streamingContent("anything goes");
697
- const { assistant } = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' }); // grammar passed but never transported
698
- assert.equal(assistant.content, "anything goes"); // no enforcement check
1197
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
1198
+ assert.equal(res.assistant.content, "anything goes");
1199
+ assert.equal(res.grammarEvidence, undefined);
699
1200
  });
700
1201
 
701
- test("enforcement: a grammar our validator can't parse is a NON-FATAL verify gap — warn, return content", async () => {
1202
+ test("provider evidence does not depend on the local validator understanding the grammar", async () => {
702
1203
  const p = grammarProvider();
703
1204
  streamingContent("whatever");
704
- const warnings: Error[] = [];
705
- const onWarn = (w: Error) => warnings.push(w);
706
- process.on("warning", onWarn);
707
- const { assistant } = await p.generate({ workerId: "r", messages: [], grammar: 'foo ::= "a"' }); // no `root` rule → validateGbnf throws
708
- await flush();
709
- process.off("warning", onWarn);
710
- assert.equal(assistant.content, "whatever"); // transport not failed
711
- assert.ok(warnings.some((w) => (w as Error & { code?: string }).code === "PLURNK_GRAMMAR_UNVERIFIABLE"), "emitted the verify-gap warning");
1205
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'foo ::= "a"' });
1206
+ assert.equal(res.assistant.content, "whatever");
1207
+ assert.deepEqual(res.grammarEvidence, { input: "whatever", contentStart: 0, transported: true });
1208
+ assert.equal(res.notices, undefined);
712
1209
  });
713
1210
 
714
- // — PLURNK_PROVIDERS_GBNF_DEBUG: run unconstrained, then verify the free output against the grammar —
1211
+ // — PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar, withhold it, and preserve the observation —
715
1212
 
716
- test("gbnfDebug: the grammar is NOT transported; conforming free output passes through with NO telemetry", async () => {
1213
+ test("gbnfDebug marks an unconstrained observation as not transported", async () => {
717
1214
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
718
1215
  const calls = installFetch([{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }]);
719
1216
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
720
1217
  const body = JSON.parse(calls[0].init.body as string);
721
- assert.equal("grammar" in body, false); // grammar never sent — model ran unconstrained
722
- assert.equal(body.repeat_penalty, 1.15); // #426: penalty rides even rail-off - unconstrained decode needs it MORE
723
- assert.equal(res.assistant.content, "ok"); // free output happens to conform → returned
724
- assert.equal("telemetry" in res, false); // conforming → no event
1218
+ assert.equal("grammar" in body, false);
1219
+ assert.equal(body.repeat_penalty, 1.15);
1220
+ assert.equal(res.assistant.content, "ok");
1221
+ assert.deepEqual(res.grammarEvidence, { input: "ok", contentStart: 0, transported: false });
1222
+ assert.equal(res.notices, undefined);
725
1223
  });
726
1224
 
727
- test("gbnfDebug: a conflict does NOT throw — it returns the bytes plus a grammar_unenforced telemetry event with the divergence position (#24)", async () => {
1225
+ test("gbnfDebug preserves conflicting bytes without a provider verdict", async () => {
728
1226
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
729
- const calls = installFetch([{ choices: [{ delta: { reasoning_content: "let me think about ok", content: "xon-conforming output" }, finish_reason: "stop" }] }]);
1227
+ const calls = installFetch([{ choices: [{ delta: { content: "xon-conforming output" }, finish_reason: "stop" }] }]);
730
1228
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
731
- // The model's bytes survive — not discarded by a throw (the empty-turn cascade root cause).
732
1229
  assert.equal(res.assistant.content, "xon-conforming output");
733
- assert.equal(res.assistant.reasoning, "let me think about ok");
734
- // Non-fatal telemetry carries the divergence so the consumer can self-correct.
735
- assert.equal(res.telemetry?.length, 1);
736
- const [event] = res.telemetry ?? [];
737
- assert.equal(event.source, "provider:test");
738
- assert.equal(event.kind, "grammar_unenforced");
739
- assert.equal(event.position, 0); // 'x' rejected at code point 0
740
- assert.match(event.message ?? "", /output rejected by the transported grammar at code point 0 \("x"\)/);
1230
+ assert.deepEqual(res.grammarEvidence, { input: "xon-conforming output", contentStart: 0, transported: false });
1231
+ assert.equal(res.notices, undefined);
741
1232
  const body = JSON.parse(calls[0].init.body as string);
742
- assert.equal("grammar" in body, false); // still never sent — diagnosed, not enforced
1233
+ assert.equal("grammar" in body, false);
743
1234
  });
744
1235
 
745
1236
  test("gbnfDebug: an INVALID grammar throws before any wire call — it never reaches the model", async () => {
@@ -752,7 +1243,7 @@ test("gbnfDebug: an INVALID grammar throws before any wire call — it never rea
752
1243
  assert.equal(calls.length, 0); // fail-hard before the fetch — grammar never transported
753
1244
  });
754
1245
 
755
- // — meta bag: verbatim provider metadata (#23) —
1246
+ // — meta bag: verbatim provider metadata —
756
1247
 
757
1248
  test("meta: passes backend fields through without reinterpreting monetary values", async () => {
758
1249
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
@@ -763,7 +1254,7 @@ test("meta: passes backend fields through without reinterpreting monetary values
763
1254
  assert.equal(res.meta?.system_fingerprint, "fp_abc");
764
1255
  });
765
1256
 
766
- // — first-party telemetry headers (attribution + client, SPEC §5) —
1257
+ // — first-party telemetry headers ({§provider-request-authority}) —
767
1258
 
768
1259
  const headerVal = (init: RequestInit, name: string): string | undefined =>
769
1260
  new Headers(init.headers).get(name) ?? undefined;
@@ -776,7 +1267,7 @@ test("firstPartyMetadata: attributions + client ride as Plurnk-* headers", async
776
1267
  assert.equal(headerVal(calls[0].init, "Plurnk-Client"), "plurnk.nvim/1.4.0");
777
1268
  });
778
1269
 
779
- test("#522 Plurnk-Worker-Primary: the lineage root rides under the gate; emitted even when it equals workerId", async () => {
1270
+ test("Plurnk-Worker-Primary: the lineage root rides under the gate; emitted even when it equals workerId", async () => {
780
1271
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
781
1272
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
782
1273
  await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
@@ -795,7 +1286,7 @@ test("#522 Plurnk-Worker-Primary: the lineage root rides under the gate; emitted
795
1286
  assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), undefined);
796
1287
  });
797
1288
 
798
- test("#522 Plurnk-Worker-Primary is structurally dropped when firstPartyMetadata is off", async () => {
1289
+ test("Plurnk-Worker-Primary is structurally dropped when firstPartyMetadata is off", async () => {
799
1290
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
800
1291
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
801
1292
  await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
@@ -818,13 +1309,13 @@ test("firstPartyMetadata on but empty values: no header emitted", async () => {
818
1309
  assert.equal(headerVal(calls[0].init, "Plurnk-Client"), undefined);
819
1310
  });
820
1311
 
821
- test("grammar transport: no grammar passed sends no grammar field, but the penalty rides (#426)", async () => {
1312
+ test("grammar transport: no grammar passed sends no grammar field, but the penalty rides", async () => {
822
1313
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
823
1314
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
824
1315
  await p.generate({ workerId: "r", messages: [] });
825
1316
  const body = JSON.parse(calls[0].init.body as string);
826
1317
  assert.equal("grammar" in body, false);
827
- assert.equal(body.repeat_penalty, 1.15); // #426: penalty is no longer grammar-gated - it rides rail-off
1318
+ assert.equal(body.repeat_penalty, 1.15); // penalty is not grammar-gated
828
1319
  });
829
1320
 
830
1321
  test("maxTokens transports as max_tokens; absent → no wire field (server default)", async () => {
@@ -839,7 +1330,7 @@ test("maxTokens transports as max_tokens; absent → no wire field (server defau
839
1330
  assert.equal("max_tokens" in JSON.parse(calls[0].init.body as string), false);
840
1331
  });
841
1332
 
842
- test("slot affinity is internal: sticky per workerId, distinct runs spread across slots (#11)", async () => {
1333
+ test("slot affinity is internal: sticky per workerId, distinct workers spread across slots", async () => {
843
1334
  const pinning = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
844
1335
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
845
1336
  await pinning.generate({ workerId: "run-A", messages: [] });
@@ -863,7 +1354,7 @@ test("slot affinity: no pinning backend or unknown slotCount → no id_slot ever
863
1354
  assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
864
1355
  });
865
1356
 
866
- test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; recent runs stay sticky (#11)", async () => {
1357
+ test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; recent workers stay sticky", async () => {
867
1358
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
868
1359
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
869
1360
  const slotOf = (i: number) => JSON.parse(calls[i].init.body as string).id_slot;
@@ -877,11 +1368,11 @@ test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; r
877
1368
  });
878
1369
 
879
1370
  test("streaming:false: a non-ok response rejects as a classified ProviderError (covers the non-streamed transport)", async () => {
880
- const { ProviderError } = await import("./telemetry.ts");
1371
+ const { ProviderError } = await import("./errors.ts");
881
1372
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, source: "provider:test" });
882
1373
  mock.method(globalThis, "fetch", async () => new Response("boom", { status: 500 }));
883
1374
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
884
- assert.ok(err instanceof ProviderError);
1375
+ assert.ok(err instanceof ProviderError, `expected ProviderError, got ${String(err)}`);
885
1376
  assert.equal(err.kind, "network_failure"); // ≥500 → network_failure
886
1377
  assert.equal(err.status, 500);
887
1378
  return true;
@@ -904,15 +1395,17 @@ test("messages pass through verbatim — the provider injects no turn (PLAN live
904
1395
  assert.equal(res.assistant.content, "out"); // content returned verbatim
905
1396
  });
906
1397
 
907
- test("generate wraps an HTTP failure as a ProviderError carrying a TelemetryEvent", async () => {
908
- const { ProviderError } = await import("./telemetry.ts");
1398
+ test("generate wraps an HTTP failure as a ProviderError carrying Problem Details", async () => {
1399
+ const { ProviderError } = await import("./errors.ts");
909
1400
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, source: "provider:test" });
910
1401
  mock.method(globalThis, "fetch", async () => new Response("rate limited", { status: 429 }));
911
1402
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
912
- assert.ok(err instanceof ProviderError);
1403
+ assert.ok(err instanceof ProviderError, `expected ProviderError, got ${String(err)}`);
913
1404
  assert.equal(err.kind, "rate_limit");
914
1405
  assert.equal(err.status, 429);
915
- assert.deepEqual(err.toTelemetryEvent(), { source: "provider:test", kind: "rate_limit", message: err.message, position: null });
1406
+ assert.equal(err.problem.status, 429);
1407
+ assert.equal(err.problem.detail, err.message);
1408
+ assert.equal(err.problem.type, "https://problems.plurnk.dev/provider/test/rate-limit");
916
1409
  return true;
917
1410
  });
918
1411
  });
@@ -937,10 +1430,19 @@ test("configured headers and url are sent verbatim", async () => {
937
1430
  assert.equal(headers.get("x-title"), "plurnk");
938
1431
  });
939
1432
 
940
- // — transient-failure retry (#18) —
1433
+ // — transient-failure retry —
941
1434
 
942
1435
  const retryCfg = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const };
943
1436
 
1437
+ const stalledStreamResponse = (): Response => new Response(new ReadableStream({
1438
+ start(controller) {
1439
+ controller.enqueue(new TextEncoder().encode(
1440
+ 'data: {"id":"stalled","object":"chat.completion.chunk","created":1,"model":"m","choices":[{"index":0,"delta":{"content":"partial"},"finish_reason":null}]}\n\n',
1441
+ ));
1442
+ setTimeout(() => controller.close(), 100);
1443
+ },
1444
+ }), { status: 200 });
1445
+
944
1446
  test("retry: a transient failure retries and a later success resolves", async () => {
945
1447
  const calls = installFetchScript([
946
1448
  { status: 429, retryAfter: 0 },
@@ -953,20 +1455,11 @@ test("retry: a transient failure retries and a later success resolves", async ()
953
1455
  assert.equal(calls.length, 3); // 429 → 503 → 200
954
1456
  });
955
1457
 
956
- test("#559: streamed-body silence fails the exchange without replaying partial output", async () => {
1458
+ test("streamed-body silence retries and returns the retry's complete output", async () => {
957
1459
  let calls = 0;
958
1460
  mock.method(globalThis, "fetch", async () => {
959
1461
  calls++;
960
- if (calls === 1) {
961
- return new Response(new ReadableStream({
962
- start(controller) {
963
- controller.enqueue(new TextEncoder().encode(
964
- 'data: {"id":"first","object":"chat.completion.chunk","created":1,"model":"m","choices":[{"index":0,"delta":{"content":"partial"},"finish_reason":null}]}\n\n',
965
- ));
966
- setTimeout(() => controller.close(), 100);
967
- },
968
- }), { status: 200 });
969
- }
1462
+ if (calls === 1) return stalledStreamResponse();
970
1463
  return new Response(new ReadableStream({
971
1464
  start(controller) {
972
1465
  controller.enqueue(new TextEncoder().encode(
@@ -979,7 +1472,7 @@ test("#559: streamed-body silence fails the exchange without replaying partial o
979
1472
  const p = new AiSdkProvider({
980
1473
  model: "m",
981
1474
  url: "http://x/v1/chat/completions",
982
- fetchTimeoutMs: 1000,
1475
+ fetchTimeoutMs: 5000,
983
1476
  streamIdleTimeoutMs: 10,
984
1477
  temperature: 0.2,
985
1478
  repeatPenalty: 1.15,
@@ -987,16 +1480,104 @@ test("#559: streamed-body silence fails the exchange without replaying partial o
987
1480
  retryAttempts: 1,
988
1481
  source: "provider:test",
989
1482
  });
1483
+ const result = await p.generate({ workerId: "r", messages: [] });
1484
+ assert.equal(result.assistant.content, "recovered", "the retry's complete output, not the stalled partial");
1485
+ assert.equal(calls, 2, "the stall retried once and the retry succeeded");
1486
+ mock.restoreAll();
1487
+ });
1488
+
1489
+ test("streamed-body silence does not replay when retries are disabled", async () => {
1490
+ let calls = 0;
1491
+ mock.method(globalThis, "fetch", async () => {
1492
+ calls++;
1493
+ return stalledStreamResponse();
1494
+ });
1495
+ const p = new AiSdkProvider({
1496
+ model: "m",
1497
+ url: "http://x/v1/chat/completions",
1498
+ fetchTimeoutMs: 1000,
1499
+ streamIdleTimeoutMs: 10,
1500
+ temperature: 0.2,
1501
+ repeatPenalty: 1.15,
1502
+ reasoning: { mode: "off", budget: null },
1503
+ retryAttempts: 0,
1504
+ source: "provider:test",
1505
+ });
990
1506
  await assert.rejects(
991
1507
  p.generate({ workerId: "r", messages: [] }),
992
1508
  (error: ProviderError) => error.kind === "network_failure"
993
1509
  && /chunk timeout/i.test(error.message),
994
1510
  );
995
- assert.equal(calls, 1);
1511
+ assert.equal(calls, 1, "zero retries permits exactly one provider request");
1512
+ mock.restoreAll();
1513
+ });
1514
+
1515
+ test("streamed-body silence exhausts the configured retry budget once", async () => {
1516
+ let calls = 0;
1517
+ mock.method(globalThis, "fetch", async () => {
1518
+ calls++;
1519
+ return stalledStreamResponse();
1520
+ });
1521
+ const p = new AiSdkProvider({
1522
+ model: "m",
1523
+ url: "http://x/v1/chat/completions",
1524
+ fetchTimeoutMs: 5000,
1525
+ streamIdleTimeoutMs: 10,
1526
+ temperature: 0.2,
1527
+ repeatPenalty: 1.15,
1528
+ reasoning: { mode: "off", budget: null },
1529
+ retryAttempts: 1,
1530
+ source: "provider:test",
1531
+ });
1532
+ await assert.rejects(
1533
+ p.generate({ workerId: "r", messages: [] }),
1534
+ (error: ProviderError) => error.kind === "network_failure"
1535
+ && error.problem.attempts === 2
1536
+ && error.problem.retryExhausted === true
1537
+ && error.problem.retryable === false,
1538
+ );
1539
+ assert.equal(calls, 2, "one configured retry permits exactly two provider requests");
1540
+ mock.restoreAll();
1541
+ });
1542
+
1543
+ test("the total generation deadline spans stalled-stream retry scheduling", async () => {
1544
+ let calls = 0;
1545
+ mock.method(globalThis, "fetch", async () => {
1546
+ calls++;
1547
+ if (calls > 1) {
1548
+ return new Response(new ReadableStream({
1549
+ start(controller) {
1550
+ controller.enqueue(new TextEncoder().encode(
1551
+ 'data: {"id":"second","object":"chat.completion.chunk","created":2,"model":"m","choices":[{"index":0,"delta":{"content":"late"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n',
1552
+ ));
1553
+ controller.close();
1554
+ },
1555
+ }), { status: 200 });
1556
+ }
1557
+ return stalledStreamResponse();
1558
+ });
1559
+ const p = new AiSdkProvider({
1560
+ model: "m",
1561
+ url: "http://x/v1/chat/completions",
1562
+ fetchTimeoutMs: 50,
1563
+ streamIdleTimeoutMs: 10,
1564
+ temperature: 0.2,
1565
+ repeatPenalty: 1.15,
1566
+ reasoning: { mode: "off", budget: null },
1567
+ retryAttempts: 3,
1568
+ source: "provider:test",
1569
+ });
1570
+ const started = Date.now();
1571
+ await assert.rejects(
1572
+ p.generate({ workerId: "r", messages: [] }),
1573
+ (error: ProviderError) => error.kind === "network_failure",
1574
+ );
1575
+ assert.ok(Date.now() - started < 500, "the configured total deadline ends retry scheduling");
1576
+ assert.equal(calls, 1, "the total deadline expires before another request begins");
996
1577
  mock.restoreAll();
997
1578
  });
998
1579
 
999
- test("#559: a zero stream-idle timeout permits a slow inter-chunk pause", async () => {
1580
+ test("a zero stream-idle timeout permits a slow inter-chunk pause", async () => {
1000
1581
  mock.method(globalThis, "fetch", async () => new Response(new ReadableStream({
1001
1582
  async start(controller) {
1002
1583
  controller.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"slow "}}]}\n\n'));
@@ -1021,7 +1602,7 @@ test("#559: a zero stream-idle timeout permits a slow inter-chunk pause", async
1021
1602
  });
1022
1603
 
1023
1604
  test("retry: exhausting the budget surfaces the classified ProviderError", async () => {
1024
- const { ProviderError } = await import("./telemetry.ts");
1605
+ const { ProviderError } = await import("./errors.ts");
1025
1606
  const calls = installFetchScript([{ status: 429, retryAfter: 0 }]); // always rate-limited
1026
1607
  const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 2 });
1027
1608
  await assert.rejects(
@@ -1056,7 +1637,7 @@ test("retry: retryAttempts 0 surfaces the first transient failure immediately",
1056
1637
  assert.equal(calls.length, 1); // no retry budget
1057
1638
  });
1058
1639
 
1059
- test("retry: a caller abort during backoff rejects promptly with no further attempt (mid-flight abort, SPEC )", async () => {
1640
+ test("retry: a caller abort during backoff rejects promptly with no further attempt", async () => {
1060
1641
  const ac = new AbortController();
1061
1642
  const calls = installFetchScript([{ status: 503, retryAfter: 5 }]); // 5s backoff we never wait out
1062
1643
  const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 3 });
@@ -1068,7 +1649,7 @@ test("retry: a caller abort during backoff rejects promptly with no further atte
1068
1649
  assert.equal(calls.length, 1); // never retried after cancellation
1069
1650
  });
1070
1651
 
1071
- // — anthropic reasoning style (thinking param, #18) —
1652
+ // — Anthropic reasoning style (wire `thinking` parameter) —
1072
1653
 
1073
1654
  test("reasoningStyle 'anthropic' maps the budget to the thinking param", async () => {
1074
1655
  // N>0 → enabled with budget_tokens
@@ -1115,10 +1696,10 @@ test("streaming:false posts without stream and parses the single JSON response",
1115
1696
  mock.restoreAll();
1116
1697
  });
1117
1698
 
1118
- // ── Data capture (#36): logprobs + verbatim rawBody, opt-in, off by default ──
1699
+ // ── Data capture ({§provider-evidence}): logprobs + verbatim rawBody, opt-in, off by default ──
1119
1700
  const captureBase = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
1120
1701
 
1121
- test("#36 logprobs OFF by default: no wire request, no assistant.logprobs, no rawBody", async () => {
1702
+ test("logprobs OFF by default: no wire request, no assistant.logprobs, no rawBody", async () => {
1122
1703
  const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
1123
1704
  const p = new AiSdkProvider({ ...captureBase });
1124
1705
  const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
@@ -1131,7 +1712,7 @@ test("#36 logprobs OFF by default: no wire request, no assistant.logprobs, no ra
1131
1712
  mock.restoreAll();
1132
1713
  });
1133
1714
 
1134
- test("#36 logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw logprob + meanLogprob", async () => {
1715
+ test("logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw logprob + meanLogprob", async () => {
1135
1716
  const chunk = { model: "m", usage: { prompt_tokens: 1, completion_tokens: 2, total_tokens: 3 }, choices: [{ delta: { content: "yesno" }, finish_reason: "stop", logprobs: { content: [
1136
1717
  { token: "yes", logprob: -0.5, sampling_logprob: -0.5, top_logprobs: [{ token: "yes", logprob: -0.5 }, { token: "no", logprob: -1.0 }] },
1137
1718
  { token: "no", logprob: -0.1, sampling_logprob: -0.1, top_logprobs: [{ token: "no", logprob: -0.1 }] },
@@ -1148,7 +1729,7 @@ test("#36 logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw l
1148
1729
  mock.restoreAll();
1149
1730
  });
1150
1731
 
1151
- test("#36 rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob preserved", async () => {
1732
+ test("rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob preserved", async () => {
1152
1733
  const wire = { model: "m", extra_top_level: "kept", choices: [{ message: { content: "no" }, finish_reason: "stop", logprobs: { content: [{ token: "no", logprob: -0.1, sampling_logprob: -0.1, token_id: 42 }] } }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } };
1153
1734
  installFetchJson(wire);
1154
1735
  const p = new AiSdkProvider({ ...captureBase, streaming: false, topLogprobs: 0, rawBody: true });
@@ -1160,7 +1741,7 @@ test("#36 rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob p
1160
1741
  mock.restoreAll();
1161
1742
  });
1162
1743
 
1163
- test("#36 caller sampling cannot forge logprobs (reserved keys): the env flag is the only control", async () => {
1744
+ test("caller sampling cannot forge logprobs (reserved keys): the env flag is the only control", async () => {
1164
1745
  const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
1165
1746
  const p = new AiSdkProvider({ ...captureBase }); // logprobs OFF
1166
1747
  await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }], sampling: { logprobs: true, top_logprobs: 5 } });
@@ -1170,9 +1751,9 @@ test("#36 caller sampling cannot forge logprobs (reserved keys): the env flag is
1170
1751
  mock.restoreAll();
1171
1752
  });
1172
1753
 
1173
- // — turn coordinate headers (#404, per #391): same gate as every first-party signal —
1754
+ // — turn coordinate headers ({§lifecycle-terms}): same gate as every first-party signal —
1174
1755
 
1175
- test("#404: workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under the first-party gate", async () => {
1756
+ test("workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under the first-party gate", async () => {
1176
1757
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1177
1758
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1178
1759
  await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
@@ -1182,7 +1763,7 @@ test("#404: workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under th
1182
1763
  assert.equal(headers.get("plurnk-turn"), "41");
1183
1764
  });
1184
1765
 
1185
- test("#404: third-party providers structurally DROP the coordinate (gate off by default)", async () => {
1766
+ test("third-party providers structurally DROP the coordinate (gate off by default)", async () => {
1186
1767
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1187
1768
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1188
1769
  await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
@@ -1192,7 +1773,7 @@ test("#404: third-party providers structurally DROP the coordinate (gate off by
1192
1773
  assert.equal(headers.has("plurnk-turn"), false);
1193
1774
  });
1194
1775
 
1195
- test("#404: coordinates are 1-based — 0/absent/empty emit no header (no strikes-style zero exception)", async () => {
1776
+ test("coordinates are 1-based — 0/absent/empty emit no header", async () => {
1196
1777
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1197
1778
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1198
1779
  await p.generate({ workerId: "r", messages: [], workspaceId: "", loop: 0, turn: 0 });
@@ -1203,9 +1784,9 @@ test("#404: coordinates are 1-based — 0/absent/empty emit no header (no strike
1203
1784
  assert.equal(headers.has("plurnk-strikes"), false);
1204
1785
  });
1205
1786
 
1206
- // -- #507: envelope surface + router-owned tuning --
1787
+ // -- {§provider-generation-envelope} --
1207
1788
 
1208
- test("#507 reserves derive from the detected window; absolutes stand alone; null window + percent = no claim", () => {
1789
+ test("reserves derive from the detected window; absolutes stand alone; null window + percent = no claim", () => {
1209
1790
  const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
1210
1791
  const derived = new AiSdkProvider({ ...base, contextWindow: 49152, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 } });
1211
1792
  assert.equal(derived.reasoningReserve, 4915); // jennifer/turboderp: 10% of 49152
@@ -1217,32 +1798,32 @@ test("#507 reserves derive from the detected window; absolutes stand alone; null
1217
1798
  assert.equal(legacy.reasoningReserve, null); // out-of-date sibling: no claim
1218
1799
  });
1219
1800
 
1220
- test("#507 router-owned tuning: tuningFloors:false drops the temperature/penalty floors, caller sampling still rides", async () => {
1801
+ test("router-owned tuning: tuningFloors:false drops the temperature/penalty floors, caller sampling still rides", async () => {
1221
1802
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0, tuningFloors: false });
1222
1803
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1223
1804
  await p.generate({ workerId: "r", messages: [], sampling: { temperature: 0.9 } });
1224
1805
  const body = JSON.parse(calls[0].init.body as string);
1225
1806
  assert.equal(body.temperature, 0.9); // caller intent passes verbatim
1226
- assert.equal("frequency_penalty" in body, false); // the floor is suppressed (router owns tuning, SPEC §5)
1807
+ assert.equal("frequency_penalty" in body, false); // the floor is suppressed; the router owns tuning
1227
1808
  });
1228
1809
 
1229
- // -- #518: prompt-cache affinity (workerId -> prompt_cache_key) --
1810
+ // -- prompt-cache affinity (workerId -> prompt_cache_key) --
1230
1811
 
1231
- test("#518 promptCacheKey on: body sends prompt_cache_key = workerId (serverless replica affinity)", async () => {
1812
+ test("promptCacheKey on: body sends prompt_cache_key = workerId (serverless replica affinity)", async () => {
1232
1813
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
1233
1814
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1234
1815
  await p.generate({ workerId: "worker-abc", messages: [] });
1235
1816
  assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc");
1236
1817
  });
1237
1818
 
1238
- test("#518 promptCacheKey off (default): no prompt_cache_key on the wire", async () => {
1819
+ test("promptCacheKey off (default): no prompt_cache_key on the wire", async () => {
1239
1820
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1240
1821
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1241
1822
  await p.generate({ workerId: "worker-abc", messages: [] });
1242
1823
  assert.equal("prompt_cache_key" in JSON.parse(calls[0].init.body as string), false);
1243
1824
  });
1244
1825
 
1245
- test("#518 prompt_cache_key is managed: caller sampling cannot forge/override the affinity key", async () => {
1826
+ test("prompt_cache_key is managed: caller sampling cannot forge/override the affinity key", async () => {
1246
1827
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
1247
1828
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1248
1829
  await p.generate({ workerId: "worker-abc", messages: [], sampling: { prompt_cache_key: "hijack" } });