@plurnk/plurnk-providers 1.3.12 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/.env.defaults +39 -22
  2. package/README.md +65 -4
  3. package/SPEC.md +222 -56
  4. package/dist/AiSdkProvider.d.ts +18 -5
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +240 -153
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +14 -15
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +26 -10
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +8 -2
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +41 -8
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/ProviderRegistry.d.ts +4 -1
  17. package/dist/ProviderRegistry.d.ts.map +1 -1
  18. package/dist/ProviderRegistry.js +7 -3
  19. package/dist/ProviderRegistry.js.map +1 -1
  20. package/dist/aiSdkTransport.d.ts +4 -2
  21. package/dist/aiSdkTransport.d.ts.map +1 -1
  22. package/dist/aiSdkTransport.js +18 -3
  23. package/dist/aiSdkTransport.js.map +1 -1
  24. package/dist/catalogProvider.d.ts +3 -1
  25. package/dist/catalogProvider.d.ts.map +1 -1
  26. package/dist/catalogProvider.js +20 -7
  27. package/dist/catalogProvider.js.map +1 -1
  28. package/dist/compatibleProvider.d.ts.map +1 -1
  29. package/dist/compatibleProvider.js +11 -4
  30. package/dist/compatibleProvider.js.map +1 -1
  31. package/dist/cost.d.ts +11 -0
  32. package/dist/cost.d.ts.map +1 -0
  33. package/dist/cost.js +64 -0
  34. package/dist/cost.js.map +1 -0
  35. package/dist/discover.d.ts +2 -0
  36. package/dist/discover.d.ts.map +1 -1
  37. package/dist/discover.js +15 -9
  38. package/dist/discover.js.map +1 -1
  39. package/dist/env.d.ts +4 -0
  40. package/dist/env.d.ts.map +1 -1
  41. package/dist/env.js +29 -9
  42. package/dist/env.js.map +1 -1
  43. package/dist/errors.d.ts +27 -0
  44. package/dist/errors.d.ts.map +1 -0
  45. package/dist/errors.js +150 -0
  46. package/dist/errors.js.map +1 -0
  47. package/dist/index.d.ts +11 -5
  48. package/dist/index.d.ts.map +1 -1
  49. package/dist/index.js +7 -4
  50. package/dist/index.js.map +1 -1
  51. package/dist/notices.d.ts +10 -0
  52. package/dist/notices.d.ts.map +1 -0
  53. package/dist/notices.js +11 -0
  54. package/dist/notices.js.map +1 -0
  55. package/dist/ollama.d.ts.map +1 -1
  56. package/dist/ollama.js +3 -3
  57. package/dist/ollama.js.map +1 -1
  58. package/dist/openai.d.ts +1 -1
  59. package/dist/openai.d.ts.map +1 -1
  60. package/dist/promptTokens.d.ts +4 -0
  61. package/dist/promptTokens.d.ts.map +1 -0
  62. package/dist/promptTokens.js +32 -0
  63. package/dist/promptTokens.js.map +1 -0
  64. package/dist/sdkModels.d.ts.map +1 -1
  65. package/dist/sdkModels.js +4 -3
  66. package/dist/sdkModels.js.map +1 -1
  67. package/dist/types.d.ts +43 -16
  68. package/dist/types.d.ts.map +1 -1
  69. package/dist/types.js +1 -1
  70. package/dist/types.js.map +1 -1
  71. package/dist/usage.d.ts +3 -0
  72. package/dist/usage.d.ts.map +1 -1
  73. package/dist/usage.js +26 -14
  74. package/dist/usage.js.map +1 -1
  75. package/dist/warnings.js +0 -0
  76. package/dist/warnings.js.map +1 -1
  77. package/package.json +13 -9
  78. package/src/AiSdkProvider.test.ts +480 -159
  79. package/src/AiSdkProvider.ts +320 -196
  80. package/src/Mock.test.ts +29 -14
  81. package/src/Mock.ts +33 -15
  82. package/src/Pool.test.ts +43 -6
  83. package/src/Pool.ts +56 -10
  84. package/src/ProviderRegistry.test.ts +158 -9
  85. package/src/ProviderRegistry.ts +19 -6
  86. package/src/aiSdkTransport.ts +25 -6
  87. package/src/boundaries.test.ts +8 -3
  88. package/src/catalogProvider.test.ts +17 -0
  89. package/src/catalogProvider.ts +25 -10
  90. package/src/compatibleProvider.test.ts +96 -0
  91. package/src/compatibleProvider.ts +15 -6
  92. package/src/cost.test.ts +63 -0
  93. package/src/cost.ts +83 -0
  94. package/src/defaults.test.ts +1 -0
  95. package/src/discover.test.ts +48 -7
  96. package/src/discover.ts +31 -21
  97. package/src/env.test.ts +38 -23
  98. package/src/env.ts +45 -18
  99. package/src/errors.test.ts +148 -0
  100. package/src/errors.ts +207 -0
  101. package/src/index.ts +29 -7
  102. package/src/lexicon-guard.test.ts +6 -6
  103. package/src/notices.ts +22 -0
  104. package/src/ollama.test.ts +64 -0
  105. package/src/ollama.ts +6 -3
  106. package/src/openai.ts +3 -0
  107. package/src/promptTokens.ts +41 -0
  108. package/src/sdkModels.test.ts +7 -0
  109. package/src/sdkModels.ts +4 -8
  110. package/src/types.ts +106 -64
  111. package/src/usage.test.ts +15 -4
  112. package/src/usage.ts +32 -14
  113. package/src/warnings.test.ts +10 -10
  114. package/src/warnings.ts +0 -0
  115. package/dist/OpenAICompat.d.ts +0 -76
  116. package/dist/OpenAICompat.d.ts.map +0 -1
  117. package/dist/OpenAICompat.js +0 -555
  118. package/dist/OpenAICompat.js.map +0 -1
  119. package/dist/openaiStream.d.ts +0 -47
  120. package/dist/openaiStream.d.ts.map +0 -1
  121. package/dist/openaiStream.js +0 -280
  122. package/dist/openaiStream.js.map +0 -1
  123. package/dist/standardProviders.d.ts +0 -31
  124. package/dist/standardProviders.d.ts.map +0 -1
  125. package/dist/standardProviders.js +0 -518
  126. package/dist/standardProviders.js.map +0 -1
  127. package/dist/telemetry.d.ts +0 -24
  128. package/dist/telemetry.d.ts.map +0 -1
  129. package/dist/telemetry.js +0 -85
  130. package/dist/telemetry.js.map +0 -1
  131. package/src/telemetry.test.ts +0 -69
  132. package/src/telemetry.ts +0 -116
@@ -1,7 +1,7 @@
1
1
  import test, { mock } from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
3
  import AiSdkProvider, { effortFromBudget } from "./AiSdkProvider.ts";
4
- import { ProviderError } from "./telemetry.ts";
4
+ import { ProviderError } from "./errors.ts";
5
5
 
6
6
  // Build a fake fetch returning a one-chunk SSE stream, capturing the request
7
7
  // so tests can assert what the spine sent on the wire.
@@ -71,7 +71,7 @@ const injectedBase = {
71
71
  reasoning: { mode: "off" as const, budget: null },
72
72
  };
73
73
 
74
- test("#608: per-instance fetch owns streaming and buffered requests", async () => {
74
+ test("per-instance fetch owns streaming and buffered requests", async () => {
75
75
  const calls: Array<{ input: string | URL | Request; init?: RequestInit }> = [];
76
76
  const streamingFetch: typeof globalThis.fetch = async (input, init) => {
77
77
  calls.push({ input, init });
@@ -105,7 +105,7 @@ test("#608: per-instance fetch owns streaming and buffered requests", async () =
105
105
  assert.equal(JSON.parse(String(calls[1].init?.body)).stream, undefined);
106
106
  });
107
107
 
108
- test("#608: caller cancellation and provider timeout reach an injected fetch", async () => {
108
+ test("caller cancellation and provider timeout reach an injected fetch", async () => {
109
109
  const pendingFetch: typeof globalThis.fetch = async (_input, init) => {
110
110
  init?.signal?.throwIfAborted();
111
111
  return new Promise((_resolve, reject) => {
@@ -126,7 +126,7 @@ test("#608: caller cancellation and provider timeout reach an injected fetch", a
126
126
  );
127
127
  });
128
128
 
129
- test("#608: per-instance fetch owns tokenization and retry attempts", async () => {
129
+ test("per-instance fetch owns tokenization and retry attempts", async () => {
130
130
  const calls: string[] = [];
131
131
  let generationAttempts = 0;
132
132
  const providerFetch: typeof globalThis.fetch = async (input) => {
@@ -192,7 +192,7 @@ const installFetchScript = (responses: ScriptedResponse[]) => {
192
192
  const flush = () => new Promise<void>((r) => setImmediate(r));
193
193
 
194
194
  import { resetEmittedWarnings } from "./warnings.ts";
195
- test.afterEach(() => { mock.restoreAll(); resetEmittedWarnings(); }); // #40: warning-asserting tests stay order-independent
195
+ test.afterEach(() => { mock.restoreAll(); resetEmittedWarnings(); });
196
196
 
197
197
  test("effortFromBudget: maps budget to tiers", () => {
198
198
  assert.equal(effortFromBudget(1), "low");
@@ -202,7 +202,7 @@ test("effortFromBudget: maps budget to tiers", () => {
202
202
  assert.equal(effortFromBudget(4001), "high");
203
203
  });
204
204
 
205
- test("#543: a 524 Cloudflare edge timeout fails fast - not retried despite retryAttempts", async () => {
205
+ test("a 524 Cloudflare edge timeout fails fast - not retried despite retryAttempts", async () => {
206
206
  const calls = installFetchScript([{ status: 524, retryAfter: 120 }]);
207
207
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
208
208
  await assert.rejects(p.generate({ workerId: "r", messages: [] }));
@@ -211,7 +211,7 @@ test("#543: a 524 Cloudflare edge timeout fails fast - not retried despite retry
211
211
  mock.restoreAll();
212
212
  });
213
213
 
214
- test("#548: a 422 grammar_invalid is a failed exchange, not transport replay policy", async () => {
214
+ test("a 422 grammar_invalid is a failed exchange, not transport replay policy", async () => {
215
215
  const body = JSON.stringify({ error: { message: "non-conforming emission rejected: ...", type: "grammar_invalid" } });
216
216
  const calls = installFetchScript([{ status: 422, body }]);
217
217
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 2 });
@@ -237,43 +237,59 @@ test("an SSE error frame is a failed exchange, not an empty completion", async (
237
237
  assert.equal(calls.length, 1);
238
238
  });
239
239
 
240
- test("#539: a trailing eos_token (--special EOG leak) is stripped from content", async () => {
240
+ test("a trailing eos_token (--special EOG leak) is stripped from content", async () => {
241
241
  installFetchJson({ model: "m", choices: [{ message: { content: "the answer<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
242
242
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
243
243
  const res = await p.generate({ workerId: "r", messages: [] });
244
244
  assert.equal(res.assistant.content, "the answer"); // trailing <eos> gone; packet + verdict see clean bytes
245
245
  });
246
246
 
247
- test("#539: without a probed eos_token the content passes through untouched", async () => {
247
+ test("without a probed eos_token the content passes through untouched", async () => {
248
248
  installFetchJson({ model: "m", choices: [{ message: { content: "keeps <eos> literally" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
249
249
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
250
250
  const res = await p.generate({ workerId: "r", messages: [] });
251
251
  assert.equal(res.assistant.content, "keeps <eos> literally"); // no eosText (a cloud backend) -> no strip
252
252
  });
253
253
 
254
- test("#539: only the TRAILING eos_token is stripped; a quoted one mid-body survives", async () => {
254
+ test("only the trailing eos_token is stripped; a quoted one mid-body survives", async () => {
255
255
  installFetchJson({ model: "m", choices: [{ message: { content: "quotes <eos> in the body<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 } });
256
256
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
257
257
  const res = await p.generate({ workerId: "r", messages: [] });
258
258
  assert.equal(res.assistant.content, "quotes <eos> in the body"); // only the tail goes
259
259
  });
260
260
 
261
- test("identity getters and defaults", () => {
261
+ test("identity getters and default prompt estimate", async () => {
262
262
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
263
263
  assert.equal(p.model, "m");
264
264
  assert.equal(p.contextWindow, null); // default
265
- assert.equal(p.countTokens(""), 0);
266
- assert.equal(p.countTokens("four"), 2); // default heuristic ceil(4/2) upper bound
267
- assert.equal(p.calculateCost({ prompt: 9, completion: 9, reasoning: 0, cached: 0, total: 18 }), 0); // default free
265
+ assert.deepEqual(
266
+ await p.countPromptTokens([{ role: "user", content: "漢漢漢" }]),
267
+ {
268
+ kind: "estimate",
269
+ tokens: 2,
270
+ source: "heuristic:chars2",
271
+ detail: "chars/2 over message content; provider request framing is unknown",
272
+ },
273
+ "chars/2 is explicitly an estimate; high-token-density Unicode prevents an upper-bound claim",
274
+ );
275
+ assert.equal(p.calculateCost({ prompt: 9, completion: 9, reasoning: 0, cached: 0, total: 18 }), 0); // current unknown-rate sentinel
268
276
  });
269
277
 
270
- test("injected countTokens and calculateCost are used", () => {
278
+ test("injected prompt measurement preserves provenance and calculateCost is used", async () => {
279
+ const seen: string[] = [];
271
280
  const p = new AiSdkProvider({
272
281
  model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
273
- countTokens: (t) => t.length,
282
+ countPromptTokens: (messages) => {
283
+ seen.push(...messages.map(({ content }) => content));
284
+ return { kind: "upper_bound", tokens: 7, source: "test:proven-bound" };
285
+ },
274
286
  calculateCost: (u) => u.total * 2,
275
287
  });
276
- assert.equal(p.countTokens("abc"), 3);
288
+ assert.deepEqual(
289
+ await p.countPromptTokens([{ role: "system", content: "system" }, { role: "user", content: "user" }]),
290
+ { kind: "upper_bound", tokens: 7, source: "test:proven-bound" },
291
+ );
292
+ assert.deepEqual(seen, ["system", "user"]);
277
293
  assert.equal(p.calculateCost({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 5 }), 10);
278
294
  });
279
295
 
@@ -311,7 +327,88 @@ test("generate surfaces and normalizes an out-of-set finish_reason", async () =>
311
327
  }]);
312
328
  });
313
329
 
314
- test("generate translates a backend cap synonym to canonical length (#425)", async () => {
330
+ test("#161: a streamed resource interruption is a failed exchange with complete attempt evidence", async () => {
331
+ const calls = installFetch([
332
+ { model: "served-model", choices: [{ delta: { reasoning_content: "partial thought", content: "partial answer" } }] },
333
+ {
334
+ choices: [{ delta: {}, finish_reason: "insufficient_system_resource" }],
335
+ usage: { prompt_tokens: 7, completion_tokens: 5, total_tokens: 12 },
336
+ },
337
+ ]);
338
+ const provider = new AiSdkProvider({
339
+ ...injectedBase,
340
+ retryAttempts: 2,
341
+ rawBody: true,
342
+ });
343
+
344
+ await assert.rejects(
345
+ provider.generate({ workerId: "interrupted", messages: [{ role: "user", content: "hello" }] }),
346
+ (error: unknown) => {
347
+ assert.ok(error instanceof ProviderError);
348
+ assert.equal(error.kind, "resource_interrupted");
349
+ assert.equal(error.status, 503);
350
+ assert.equal(error.problem.stage, "provider-response");
351
+ assert.equal(error.problem.retryable, false);
352
+ assert.equal(error.problem.finishReason, "resource_interrupted");
353
+ assert.equal(error.problem.rawFinishReason, "insufficient_system_resource");
354
+ assert.equal(error.attempt?.assistant.content, "partial answer");
355
+ assert.equal(error.attempt?.assistant.reasoning, "partial thought");
356
+ assert.equal(error.attempt?.assistant.finishReason, "resource_interrupted");
357
+ assert.deepEqual(error.attempt?.assistant.usage, {
358
+ prompt: 7,
359
+ completion: 2,
360
+ reasoning: 3,
361
+ cached: 0,
362
+ total: 12,
363
+ });
364
+ assert.equal(
365
+ (error.attempt?.assistantRaw as { rawFinishReason?: string }).rawFinishReason,
366
+ "insufficient_system_resource",
367
+ );
368
+ assert.ok(Array.isArray(error.attempt?.rawBody));
369
+ return true;
370
+ },
371
+ );
372
+ assert.equal(calls.length, 1, "a semantic interruption is not replayed as an HTTP failure");
373
+ });
374
+
375
+ test("#161: a buffered resource interruption preserves the successful wire response as failed-attempt evidence", async () => {
376
+ const wire = {
377
+ model: "served-model",
378
+ choices: [{
379
+ message: { content: "partial answer", reasoning_content: "partial thought" },
380
+ finish_reason: "insufficient_system_resource",
381
+ }],
382
+ usage: { prompt_tokens: 7, completion_tokens: 5, total_tokens: 12 },
383
+ };
384
+ const calls = installFetchJson(wire);
385
+ const provider = new AiSdkProvider({
386
+ ...injectedBase,
387
+ streaming: false,
388
+ retryAttempts: 2,
389
+ rawBody: true,
390
+ });
391
+
392
+ await assert.rejects(
393
+ provider.generate({ workerId: "interrupted", messages: [{ role: "user", content: "hello" }] }),
394
+ (error: unknown) => {
395
+ assert.ok(error instanceof ProviderError);
396
+ assert.equal(error.kind, "resource_interrupted");
397
+ assert.equal(error.attempt?.assistant.content, "partial answer");
398
+ assert.equal(error.attempt?.assistant.reasoning, "partial thought");
399
+ assert.equal(error.attempt?.assistant.finishReason, "resource_interrupted");
400
+ assert.deepEqual(error.attempt?.rawBody, wire);
401
+ assert.equal(
402
+ (error.attempt?.assistantRaw as { rawFinishReason?: string }).rawFinishReason,
403
+ "insufficient_system_resource",
404
+ );
405
+ return true;
406
+ },
407
+ );
408
+ assert.equal(calls.length, 1);
409
+ });
410
+
411
+ test("generate translates a backend cap synonym to canonical length", async () => {
315
412
  // gemini shouts MAX_TOKENS, anthropic says max_tokens -- both must reach core as
316
413
  // "length" so its truncation check (=== "length") is a cross-backend invariant.
317
414
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
@@ -320,7 +417,7 @@ test("generate translates a backend cap synonym to canonical length (#425)", asy
320
417
  assert.equal(assistant.finishReason, "length");
321
418
  });
322
419
 
323
- test("generate translates end_turn to canonical stop (#425)", async () => {
420
+ test("generate translates end_turn to canonical stop", async () => {
324
421
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
325
422
  installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "end_turn" }] }]);
326
423
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
@@ -339,10 +436,172 @@ test("generate aggregates reasoning deltas under multiple field names", async ()
339
436
  installFetch([{ choices: [{ delta: { reasoning_content: "be", thinking: "cause" } }] }]);
340
437
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
341
438
  assert.equal(assistant.reasoning, "because");
342
- assert.equal("reasoningEncrypted" in assistant, false); // open reasoning only -> field absent (#482)
439
+ assert.equal("reasoningEncrypted" in assistant, false); // open reasoning only -> field absent
440
+ });
441
+
442
+ test("{§provider-tagged-reasoning} explicit think-tags projects one streamed leading envelope and reclassifies usage", async () => {
443
+ const config = { ...injectedBase, reasoningResponseStyle: "think-tags" as const, rawBody: true };
444
+ const p = new AiSdkProvider(config);
445
+ installFetch([
446
+ { choices: [{ delta: { content: "<thi" } }] },
447
+ { choices: [{ delta: { content: "nk>12345</th" } }] },
448
+ { choices: [{ delta: { content: "ink>abcde" }, finish_reason: "stop" }] },
449
+ { usage: { prompt_tokens: 3, completion_tokens: 10, total_tokens: 13 } },
450
+ ]);
451
+
452
+ const response = await p.generate({ workerId: "tagged-stream", messages: [] });
453
+
454
+ assert.equal(response.assistant.reasoning, "12345");
455
+ assert.equal(response.assistant.content, "abcde");
456
+ assert.deepEqual(response.assistant.usage, {
457
+ prompt: 3,
458
+ completion: 5,
459
+ reasoning: 5,
460
+ cached: 0,
461
+ total: 13,
462
+ });
463
+ assert.deepEqual(
464
+ ((response.assistantRaw as { content: string; reasoning: string }).content),
465
+ "abcde",
466
+ );
467
+ assert.equal((response.assistantRaw as { reasoning: string }).reasoning, "12345");
468
+ assert.match(JSON.stringify(response.rawBody), /<thi/);
469
+ assert.match(JSON.stringify(response.rawBody), /nk>12345/);
470
+ });
471
+
472
+ test("{§provider-tagged-reasoning} explicit think-tags projects one buffered leading envelope", async () => {
473
+ installFetchJson({
474
+ model: "m",
475
+ choices: [{ message: { content: "<think>12345</think>abcde" }, finish_reason: "stop" }],
476
+ usage: { prompt_tokens: 3, completion_tokens: 10, total_tokens: 13 },
477
+ });
478
+ const config = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
479
+ const response = await new AiSdkProvider(config).generate({ workerId: "tagged-buffer", messages: [] });
480
+
481
+ assert.equal(response.assistant.reasoning, "12345");
482
+ assert.equal(response.assistant.content, "abcde");
483
+ assert.equal(response.assistant.usage.completion, 5);
484
+ assert.equal(response.assistant.usage.reasoning, 5);
485
+ });
486
+
487
+ test("{§provider-tagged-reasoning} tagged text does not overwrite itemized reasoning usage", async () => {
488
+ installFetchJson({
489
+ model: "m",
490
+ choices: [{ message: { content: "<think>12345</think>abcde" }, finish_reason: "stop" }],
491
+ usage: {
492
+ prompt_tokens: 3,
493
+ completion_tokens: 10,
494
+ total_tokens: 13,
495
+ completion_tokens_details: { reasoning_tokens: 3 },
496
+ },
497
+ });
498
+ const config = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
499
+ const response = await new AiSdkProvider(config).generate({ workerId: "tagged-itemized", messages: [] });
500
+
501
+ assert.equal(response.assistant.reasoning, "12345");
502
+ assert.equal(response.assistant.content, "abcde");
503
+ assert.equal(response.assistant.usage.completion, 7);
504
+ assert.equal(response.assistant.usage.reasoning, 3);
505
+ });
506
+
507
+ test("{§provider-tagged-reasoning} an unclosed capped envelope is wholly reasoning in streamed and buffered responses", async () => {
508
+ const config = { ...injectedBase, reasoningResponseStyle: "think-tags" as const };
509
+ installFetch([
510
+ { choices: [{ delta: { content: "<think>unfinished" }, finish_reason: "length" }] },
511
+ { usage: { prompt_tokens: 3, completion_tokens: 8, total_tokens: 11 } },
512
+ ]);
513
+ const streamed = await new AiSdkProvider(config).generate({ workerId: "tagged-capped-stream", messages: [] });
514
+ assert.equal(streamed.assistant.reasoning, "unfinished");
515
+ assert.equal(streamed.assistant.content, "");
516
+ assert.deepEqual(streamed.assistant.usage, {
517
+ prompt: 3,
518
+ completion: 0,
519
+ reasoning: 8,
520
+ cached: 0,
521
+ total: 11,
522
+ });
523
+
524
+ mock.restoreAll();
525
+ installFetchJson({
526
+ model: "m",
527
+ choices: [{ message: { content: "<think>unfinished" }, finish_reason: "length" }],
528
+ usage: { prompt_tokens: 3, completion_tokens: 8, total_tokens: 11 },
529
+ });
530
+ const bufferedConfig = { ...config, streaming: false };
531
+ const buffered = await new AiSdkProvider(bufferedConfig).generate({ workerId: "tagged-capped-buffer", messages: [] });
532
+ assert.equal(buffered.assistant.reasoning, "unfinished");
533
+ assert.equal(buffered.assistant.content, "");
534
+ assert.equal(buffered.assistant.usage.completion, 0);
535
+ assert.equal(buffered.assistant.usage.reasoning, 8);
536
+ });
537
+
538
+ test("{§provider-tagged-reasoning} verbatim, non-leading, and structured-reasoning controls preserve literal tags", async () => {
539
+ installFetchJson({
540
+ model: "m",
541
+ choices: [{ message: { content: "<think>literal</think>answer" }, finish_reason: "stop" }],
542
+ usage: { prompt_tokens: 1, completion_tokens: 4, total_tokens: 5 },
543
+ });
544
+ const verbatim = await new AiSdkProvider({ ...injectedBase, streaming: false })
545
+ .generate({ workerId: "verbatim", messages: [] });
546
+ assert.equal(verbatim.assistant.content, "<think>literal</think>answer");
547
+ assert.equal(verbatim.assistant.reasoning, null);
548
+ assert.equal(verbatim.assistant.usage.completion, 4);
549
+
550
+ mock.restoreAll();
551
+ installFetchJson({
552
+ model: "m",
553
+ choices: [{ message: { content: "show <think>literal</think> exactly" }, finish_reason: "stop" }],
554
+ usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 },
555
+ });
556
+ const taggedConfig = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
557
+ const nonLeading = await new AiSdkProvider(taggedConfig)
558
+ .generate({ workerId: "non-leading", messages: [] });
559
+ assert.equal(nonLeading.assistant.content, "show <think>literal</think> exactly");
560
+ assert.equal(nonLeading.assistant.reasoning, null);
561
+
562
+ mock.restoreAll();
563
+ installFetchJson({
564
+ model: "m",
565
+ choices: [{ message: {
566
+ content: "<think>literal visible bytes</think>",
567
+ reasoning_content: "structured reasoning",
568
+ }, finish_reason: "stop" }],
569
+ usage: { prompt_tokens: 1, completion_tokens: 7, total_tokens: 8 },
570
+ });
571
+ const structured = await new AiSdkProvider(taggedConfig)
572
+ .generate({ workerId: "structured", messages: [] });
573
+ assert.equal(structured.assistant.content, "<think>literal visible bytes</think>");
574
+ assert.equal(structured.assistant.reasoning, "structured reasoning");
575
+ });
576
+
577
+ test("{§provider-tagged-reasoning} grammar evidence retains the exact pre-projection tagged sentence", async () => {
578
+ const content = "<think>🧠reason</think><<PLAN::PLAN\n<<SEND[200]:done:SEND";
579
+ const config = {
580
+ ...injectedBase,
581
+ contextWindow: 640,
582
+ reasoning: { mode: "adaptive" as const, budget: null },
583
+ reasoningResponseStyle: "think-tags" as const,
584
+ reasoningStyle: "think" as const,
585
+ grammarStyle: "llamacpp" as const,
586
+ };
587
+ installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
588
+
589
+ const response = await new AiSdkProvider(config).generate({
590
+ workerId: "tagged-grammar",
591
+ messages: [],
592
+ grammar: `root ::= ${JSON.stringify(content)}`,
593
+ });
594
+
595
+ assert.equal(response.assistant.reasoning, "🧠reason");
596
+ assert.equal(response.assistant.content, "<<PLAN::PLAN\n<<SEND[200]:done:SEND");
597
+ assert.deepEqual(response.grammarEvidence, {
598
+ input: content,
599
+ contentStart: [..."<think>🧠reason</think>"].length,
600
+ transported: true,
601
+ });
343
602
  });
344
603
 
345
- test("#482 sealed relay reasoning (non-streamed): encrypted reasoning_details surface verbatim, text entries do not", async () => {
604
+ test("encrypted reasoning (non-streamed): encrypted entries normalize and text entries stay separate", async () => {
346
605
  // The live o4-mini-via-OpenRouter shape: reasoning null, one encrypted entry.
347
606
  installFetchJson({ model: "m", choices: [{ message: {
348
607
  content: "4", reasoning: null,
@@ -353,13 +612,14 @@ test("#482 sealed relay reasoning (non-streamed): encrypted reasoning_details su
353
612
  }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
354
613
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
355
614
  const { assistant } = await p.generate({ workerId: "r", messages: [] });
356
- // item shape: wire `id` preserved, subtype from position (#482 widening)
615
+ // Wire detail ID is preserved; the assistant-message location supports the
616
+ // derived classification but supplies no downstream client entity ID.
357
617
  assert.deepEqual(assistant.reasoningEncrypted, [{ id: "rs_1", subtype: "message", encrypted: [{ data: "gAAAAABqBLOB", format: "openai-responses-v1" }] }]);
358
- assert.equal(assistant.reasoning, null); // sealed turn: nothing readable
618
+ assert.equal(assistant.reasoning, null); // Encrypted turn: nothing readable.
359
619
  assert.equal(assistant.content, "4");
360
620
  });
361
621
 
362
- test("#482 widening: distinct wire ids stay distinct items (a single-object shape would collide them)", async () => {
622
+ test("distinct encrypted-reasoning wire ids stay distinct items", async () => {
363
623
  installFetchJson({ model: "m", choices: [{ message: { content: "ok", reasoning: null, reasoning_details: [
364
624
  { type: "reasoning.encrypted", data: "AAA", format: "openai-responses-v1", id: "rs_1" },
365
625
  { type: "reasoning.encrypted", data: "BBB", format: "openai-responses-v1", id: "rs_2" },
@@ -370,7 +630,20 @@ test("#482 widening: distinct wire ids stay distinct items (a single-object shap
370
630
  assert.deepEqual(assistant.reasoningEncrypted?.map((i) => i.id), ["rs_1", "rs_2"]);
371
631
  });
372
632
 
373
- test("#482 sealed relay reasoning (streamed): chunked blob concatenates per entry index", async () => {
633
+ test("assistant-message location classifies encrypted reasoning without inventing a missing detail id", async () => {
634
+ installFetchJson({ model: "m", choices: [{ message: { content: "ok", reasoning_details: [
635
+ { type: "reasoning.encrypted", data: "OPAQUE", format: "openai-responses-v1", id: null, index: 0 },
636
+ ] }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
637
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
638
+ const { assistant } = await p.generate({ workerId: "r", messages: [] });
639
+ assert.deepEqual(assistant.reasoningEncrypted, [{
640
+ id: null,
641
+ subtype: "message",
642
+ encrypted: [{ data: "OPAQUE", format: "openai-responses-v1" }],
643
+ }]);
644
+ });
645
+
646
+ test("encrypted reasoning (streamed): chunked blob concatenates per entry index", async () => {
374
647
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
375
648
  installFetch([
376
649
  { choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "gAAAA", format: "openai-responses-v1", id: "rs_1", index: 0 }] } }] },
@@ -402,10 +675,10 @@ test("reasoningStyle 'effort' sends a reasoning_effort tier from the budget", as
402
675
  assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, "high");
403
676
  });
404
677
 
405
- test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS (#403 — literal is MiniMax-only), on sends the tier", async () => {
678
+ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS, on sends the tier", async () => {
406
679
  // expected === null → the field must be ABSENT from the wire body. Fireworks
407
680
  // 400s reasoning_effort='adaptive' for non-MiniMax models (wire-verified,
408
- // #403): adaptive = the backend's own default posture = omission.
681
+ // Adaptive = the backend's own default posture = omission.
409
682
  for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "on", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "on"; budget: number | null }, string | null]>) {
410
683
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
411
684
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
@@ -417,7 +690,39 @@ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS (#403 —
417
690
  }
418
691
  });
419
692
 
420
- test("the family temperature default rides every request; caller sampling overrides it (#30)", async () => {
693
+ test("{§deepseek-reasoning-request} #157: thinking_effort maps the complete DeepSeek reasoning contract", async () => {
694
+ const cases = [
695
+ [{ mode: "off", budget: null }, { thinking: { type: "disabled" } }],
696
+ [{ mode: "adaptive", budget: null }, {}],
697
+ [{ mode: "on", budget: 5000 }, { thinking: { type: "enabled" }, reasoning_effort: "high" }],
698
+ ] as const;
699
+ for (const [reasoning, expected] of cases) {
700
+ const p = new AiSdkProvider({
701
+ model: "m",
702
+ url: "http://x/v1/chat/completions",
703
+ fetchTimeoutMs: 5000,
704
+ temperature: 0.2,
705
+ repeatPenalty: 1.15,
706
+ reasoning,
707
+ retryAttempts: 0,
708
+ reasoningStyle: "thinking_effort",
709
+ });
710
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
711
+ await p.generate({
712
+ workerId: "r",
713
+ messages: [],
714
+ sampling: { thinking: { type: "disabled" }, reasoning_effort: "max" },
715
+ });
716
+ const body = JSON.parse(calls[0].init.body as string);
717
+ assert.deepEqual(
718
+ Object.fromEntries(Object.entries(body).filter(([key]) => key === "thinking" || key === "reasoning_effort")),
719
+ expected,
720
+ );
721
+ mock.restoreAll();
722
+ }
723
+ });
724
+
725
+ test("the family temperature default rides every request; caller sampling overrides it", async () => {
421
726
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
422
727
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
423
728
  await p.generate({ workerId: "r", messages: [] });
@@ -434,7 +739,7 @@ test("the family temperature default rides every request; caller sampling overri
434
739
  assert.equal(JSON.parse(calls[0].init.body as string).temperature, 0.2);
435
740
  });
436
741
 
437
- test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box default; never on cloud", async () => {
742
+ test("DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box default; never on cloud", async () => {
438
743
  const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off" as const, budget: null }, retryAttempts: 0 };
439
744
  // set + llamacpp -> the loop-breakers ride the wire
440
745
  const p = new AiSdkProvider({ ...base, grammarStyle: "llamacpp", dryMultiplier: 0.8, dryBase: 1.75, dryAllowedLength: 2, repeatLastN: 512 });
@@ -474,14 +779,14 @@ test("llamacpp grammar path: temperature default + the managed repeat-penalty fl
474
779
  assert.equal(body.repeat_penalty, 1.15);
475
780
  });
476
781
 
477
- test("#426: the repeat penalty rides EVERY request rail-off, keyed per backend (cloud degeneration guard)", async () => {
782
+ test("the repeat penalty rides every request rail-off, keyed per backend", async () => {
478
783
  // llama.cpp with NO grammar carries its key too (unconstrained local is guarded)
479
784
  const llama = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
480
785
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
481
786
  await llama.generate({ workerId: "r", messages: [] });
482
787
  assert.equal(JSON.parse(calls[0].init.body as string).repeat_penalty, 1.15);
483
788
  mock.restoreAll();
484
- // a `none`-style cloud backend WITH a frequency penalty gets frequency_penalty (OpenAI-standard, #426)
789
+ // A `none`-style cloud backend with a frequency penalty gets frequency_penalty.
485
790
  const cloud = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
486
791
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
487
792
  await cloud.generate({ workerId: "r", messages: [] });
@@ -520,7 +825,7 @@ test("sampling passthrough forwards caller params; managed + reserved keys win",
520
825
  assert.equal("id_slot" in body, false); // reserved slot key stripped
521
826
  });
522
827
 
523
- test("#477 sampling passthrough guards contract invariants: n/tools/caps stripped, platform knobs pass", async () => {
828
+ test("sampling passthrough guards contract invariants: n/tools/caps stripped, platform knobs pass", async () => {
524
829
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
525
830
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
526
831
  await p.generate({
@@ -531,7 +836,7 @@ test("#477 sampling passthrough guards contract invariants: n/tools/caps strippe
531
836
  n: 3, // breaks choices[0] atomicity -> stripped
532
837
  tools: [{ type: "function" }], tool_choice: "auto", // tools-in-body doctrine -> stripped
533
838
  modalities: ["text", "audio"], prediction: { type: "content" }, // text-only / decode semantics -> stripped
534
- max_tokens: 999999, max_completion_tokens: 999999, // envelope bypass (#425 cap) -> stripped
839
+ max_tokens: 999999, max_completion_tokens: 999999, // envelope bypass -> stripped
535
840
  seed: 42, user: "acct-7", service_tier: "flex", // platform/sampling intent -> pass
536
841
  },
537
842
  });
@@ -545,58 +850,97 @@ test("#477 sampling passthrough guards contract invariants: n/tools/caps strippe
545
850
  assert.equal(body.service_tier, "flex");
546
851
  });
547
852
 
548
- test("#488 postmortem: intent maps IDENTICALLY under a transported grammar — sanctioned channel coexists with rails", async () => {
549
- // The brief rails-win-the-channel clamp is REVERTED: closing the channel starved a
550
- // reasoning-tuned model into escaping mid-content (unconstrained, discarded, billed).
551
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
552
- const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
553
- const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
853
+ test("template reasoning returns the exact pre-projection grammar sentence ({§gbnf-response-observation})", async () => {
854
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
855
+ const calls = installFetch([{ choices: [{ delta: { reasoning_content: "con🙂sider", content: "x" } }] }]);
856
+ const grammarInput = "<|channel>thought\ncon🙂sider<channel|>x";
857
+ const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(grammarInput)}` });
554
858
  const body = JSON.parse(calls[0].init.body as string);
555
- assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true }); // channel stays sanctioned under the grammar
556
- assert.equal(typeof body.grammar, "string"); // rails ride beside it
557
- // #488 per-request loud state: rail attachment + verdict on meta, drill-readable per turn
558
- assert.equal(res.meta?.railsAttached, true);
559
- assert.equal(res.meta?.railsVerdict, "accept");
859
+ assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true });
860
+ assert.equal(body.reasoning_format, "auto");
861
+ assert.equal(body.thinking_budget_tokens, 64);
862
+ assert.equal(body.grammar, `root ::= ${JSON.stringify(grammarInput)}`);
863
+ assert.deepEqual(res.grammarEvidence, {
864
+ input: grammarInput,
865
+ contentStart: [..."<|channel>thought\ncon🙂sider<channel|>"].length,
866
+ transported: true,
867
+ });
868
+ assert.equal(res.meta?.railsVerdict, undefined, "the provider represents evidence but does not grade itself");
869
+ });
870
+
871
+ test("template reasoning does not invent pre-projection evidence when the wire omits its reasoning field", async () => {
872
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
873
+ installFetch([{ choices: [{ delta: { content: "x" } }] }]);
874
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
875
+ assert.equal(res.grammarEvidence, undefined);
560
876
  });
561
877
 
562
- test("#488 channel-escape detector: billed completion tokens vastly beyond visible channels attach grammar_unenforced", async () => {
878
+ test("channel-escape detector: billed completion tokens vastly beyond visible channels attach grammar_unenforced", async () => {
563
879
  // The run105 shape: tiny visible content, no reasoning, thousands billed — the decode
564
880
  // escaped into a discarded reasoning block, unconstrained.
565
- const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
566
- installFetch([
881
+ const chunks = [
567
882
  { choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
568
883
  { usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
569
- ]);
884
+ ];
885
+ const fetch: typeof globalThis.fetch = async (input, init) => {
886
+ if (String(input).endsWith("/tokenize")) {
887
+ const body = JSON.parse(String(init?.body)) as { content: string };
888
+ return new Response(JSON.stringify({
889
+ tokens: body.content.length === 0 ? [] : [1],
890
+ }), { headers: { "content-type": "application/json" } });
891
+ }
892
+ return new Response(sseStream(chunks), { status: 200 });
893
+ };
894
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetch, tokenizeUrl: "http://x/tokenize", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
570
895
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
571
- assert.equal(res.meta?.railsVerdict, "accept"); // the visible fragment conforms...
572
- const escape = res.telemetry?.find((e) => e.message?.includes("escaped the grammar") === true);
573
- assert.ok(escape, "escape telemetry attached");
896
+ const escape = res.notices?.find((e) => e.message.includes("escaped the grammar"));
897
+ assert.ok(escape, "escape notice attached");
574
898
  assert.equal(escape!.kind, "grammar_unenforced");
575
899
  assert.match(escape!.message ?? "", /5000 completion tokens billed/);
576
900
  });
577
901
 
578
- test("#488 loud state absent on grammarless calls; no escape event without a transported grammar", async () => {
902
+ test("channel-escape state is absent without a transported grammar", async () => {
579
903
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
580
904
  installFetch([
581
905
  { choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
582
906
  { usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
583
907
  ]);
584
908
  const res = await p.generate({ workerId: "r", messages: [] }); // no grammar arg
585
- assert.equal(res.meta?.railsAttached, undefined);
586
- assert.equal(res.telemetry, undefined);
909
+ assert.equal(res.grammarEvidence, undefined);
910
+ assert.equal(res.notices, undefined);
587
911
  });
588
912
 
589
- test("reasoningStyle 'template' always emits enable_thinking mirroring budget != 0 — explicit false, never omitted", async () => {
590
- const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
913
+ test("reasoningStyle 'template' sends llama-server activation, parser, and response-wide allowance", async () => {
914
+ const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
591
915
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
592
916
  await on.generate({ workerId: "r", messages: [] });
593
- assert.deepEqual(JSON.parse(calls[0].init.body as string).chat_template_kwargs, { enable_thinking: true });
917
+ let body = JSON.parse(calls[0].init.body as string);
918
+ assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true });
919
+ assert.equal(body.reasoning_format, "auto");
920
+ assert.equal(body.thinking_budget_tokens, 64);
594
921
 
595
922
  mock.restoreAll();
596
- const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
923
+ const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
597
924
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
598
925
  await off.generate({ workerId: "r", messages: [] });
599
- assert.deepEqual(JSON.parse(calls[0].init.body as string).chat_template_kwargs, { enable_thinking: false });
926
+ body = JSON.parse(calls[0].init.body as string);
927
+ assert.deepEqual(body.chat_template_kwargs, { enable_thinking: false });
928
+ assert.equal(body.reasoning_format, "auto");
929
+ assert.equal(body.thinking_budget_tokens, 0);
930
+ });
931
+
932
+ test("reasoningStyle 'template' explicit budget tightens the reserve and cannot exceed it", async () => {
933
+ const base = { model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 } as const, completionReserve: { tokens: 160 } as const, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryAttempts: 0, reasoningStyle: "template" as const };
934
+ const p = new AiSdkProvider({ ...base, reasoning: { mode: "on", budget: 32 } });
935
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
936
+ await p.generate({ workerId: "r", messages: [], sampling: { thinking_budget_tokens: 999, reasoning_format: "none" } });
937
+ const body = JSON.parse(calls[0].init.body as string);
938
+ assert.equal(body.thinking_budget_tokens, 32);
939
+ assert.equal(body.reasoning_format, "auto");
940
+ assert.throws(
941
+ () => new AiSdkProvider({ ...base, reasoning: { mode: "on", budget: 65 } }),
942
+ /REASONING_BUDGET \(65\) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE \(64\)/,
943
+ );
600
944
  });
601
945
 
602
946
  test("budget 0 suppresses effort and include_reasoning", async () => {
@@ -619,7 +963,7 @@ test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", asy
619
963
  assert.equal(JSON.parse(calls[0].init.body as string).include_reasoning, true);
620
964
  });
621
965
 
622
- // — grammar-constrained sampling (SPEC §13, issues #8/#9) —
966
+ // — grammar-constrained sampling —
623
967
 
624
968
  test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor", async () => {
625
969
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
@@ -640,106 +984,81 @@ test("grammar transport 'none' (default): the grammar is never sent — no silen
640
984
  assert.equal("response_format" in body, false);
641
985
  });
642
986
 
643
- // — grammar conformance OBSERVATION (SPEC §10.14, §13): a completed exchange always
644
- // returns; bytes flow; a non-accept verdict rides response.telemetry —
987
+ // — exact pre-projection grammar evidence ({§gbnf-response-observation}) —
645
988
 
646
989
  const grammarProvider = () => new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", source: "provider:test" });
647
990
  const streamingContent = (content: string) => installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
648
991
 
649
- test("enforcement: conforming output passes through unchanged", async () => {
992
+ test("an unsplit grammar response carries the exact observed sentence", async () => {
650
993
  const p = grammarProvider();
651
994
  streamingContent("ok");
652
- const { assistant } = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
653
- assert.equal(assistant.content, "ok");
654
- });
655
-
656
- test("observation: REJECTED output still returns — bytes present, verdict attached with position", async () => {
657
- const p = grammarProvider();
658
- streamingContent("no");
659
995
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
660
- assert.equal(res.assistant.content, "no"); // bytes ALWAYS flow
661
- assert.equal(res.telemetry?.length, 1);
662
- const ev = res.telemetry![0];
663
- assert.equal(ev.kind, "grammar_unenforced");
664
- assert.equal(ev.source, "provider:test");
665
- assert.match(String(ev.message), /grammar not enforced: output rejected .* at code point 0/);
666
- assert.equal(ev.position, 0); // divergence offset for consumer policy
667
- });
668
-
669
- test("observation: an incomplete (valid prefix, never terminated) also returns with the verdict", async () => {
670
- const p = grammarProvider();
671
- streamingContent("ok");
672
- const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok" "!"' });
673
996
  assert.equal(res.assistant.content, "ok");
674
- assert.equal(res.telemetry?.length, 1);
675
- assert.match(String(res.telemetry![0].message), /incomplete match .* never terminated/);
676
- assert.equal(res.telemetry![0].position, 2);
997
+ assert.deepEqual(res.grammarEvidence, {
998
+ input: "ok",
999
+ contentStart: 0,
1000
+ transported: true,
1001
+ });
677
1002
  });
678
1003
 
679
- test("observation: conforming output attaches NO telemetry", async () => {
1004
+ test("the provider returns rejected or incomplete bytes as evidence without grading them", async () => {
680
1005
  const p = grammarProvider();
681
- streamingContent("ok");
1006
+ streamingContent("no");
682
1007
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
683
- assert.equal(res.telemetry, undefined);
1008
+ assert.equal(res.assistant.content, "no");
1009
+ assert.deepEqual(res.grammarEvidence, { input: "no", contentStart: 0, transported: true });
1010
+ assert.equal(res.notices, undefined);
1011
+ assert.equal(res.meta?.railsVerdict, undefined);
684
1012
  });
685
1013
 
686
- test("observation: empty content under a non-empty grammar returns with the verdict (the 'content never arrives' leak, observed)", async () => {
1014
+ test("empty unsplit content remains exact grammar evidence", async () => {
687
1015
  const p = grammarProvider();
688
- installFetch([{ choices: [{ delta: {}, finish_reason: "stop" }] }]); // no content delta → ""
1016
+ installFetch([{ choices: [{ delta: {}, finish_reason: "stop" }] }]);
689
1017
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
690
1018
  assert.equal(res.assistant.content, "");
691
- assert.equal(res.telemetry?.[0].kind, "grammar_unenforced");
1019
+ assert.deepEqual(res.grammarEvidence, { input: "", contentStart: 0, transported: true });
692
1020
  });
693
1021
 
694
- test("enforcement: when no grammar is sent (grammarStyle 'none'), output is NOT validated — no wire fields, no error (SPEC )", async () => {
1022
+ test("grammarStyle 'none' produces no grammar observation", async () => {
695
1023
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // grammarStyle defaults to "none"
696
1024
  streamingContent("anything goes");
697
- const { assistant } = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' }); // grammar passed but never transported
698
- assert.equal(assistant.content, "anything goes"); // no enforcement check
1025
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
1026
+ assert.equal(res.assistant.content, "anything goes");
1027
+ assert.equal(res.grammarEvidence, undefined);
699
1028
  });
700
1029
 
701
- test("enforcement: a grammar our validator can't parse is a NON-FATAL verify gap — warn, return content", async () => {
1030
+ test("provider evidence does not depend on the local validator understanding the grammar", async () => {
702
1031
  const p = grammarProvider();
703
1032
  streamingContent("whatever");
704
- const warnings: Error[] = [];
705
- const onWarn = (w: Error) => warnings.push(w);
706
- process.on("warning", onWarn);
707
- const { assistant } = await p.generate({ workerId: "r", messages: [], grammar: 'foo ::= "a"' }); // no `root` rule → validateGbnf throws
708
- await flush();
709
- process.off("warning", onWarn);
710
- assert.equal(assistant.content, "whatever"); // transport not failed
711
- assert.ok(warnings.some((w) => (w as Error & { code?: string }).code === "PLURNK_GRAMMAR_UNVERIFIABLE"), "emitted the verify-gap warning");
1033
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'foo ::= "a"' });
1034
+ assert.equal(res.assistant.content, "whatever");
1035
+ assert.deepEqual(res.grammarEvidence, { input: "whatever", contentStart: 0, transported: true });
1036
+ assert.equal(res.notices, undefined);
712
1037
  });
713
1038
 
714
- // — PLURNK_PROVIDERS_GBNF_DEBUG: run unconstrained, then verify the free output against the grammar —
1039
+ // — PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar, withhold it, and preserve the observation —
715
1040
 
716
- test("gbnfDebug: the grammar is NOT transported; conforming free output passes through with NO telemetry", async () => {
1041
+ test("gbnfDebug marks an unconstrained observation as not transported", async () => {
717
1042
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
718
1043
  const calls = installFetch([{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }]);
719
1044
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
720
1045
  const body = JSON.parse(calls[0].init.body as string);
721
- assert.equal("grammar" in body, false); // grammar never sent — model ran unconstrained
722
- assert.equal(body.repeat_penalty, 1.15); // #426: penalty rides even rail-off - unconstrained decode needs it MORE
723
- assert.equal(res.assistant.content, "ok"); // free output happens to conform → returned
724
- assert.equal("telemetry" in res, false); // conforming → no event
1046
+ assert.equal("grammar" in body, false);
1047
+ assert.equal(body.repeat_penalty, 1.15);
1048
+ assert.equal(res.assistant.content, "ok");
1049
+ assert.deepEqual(res.grammarEvidence, { input: "ok", contentStart: 0, transported: false });
1050
+ assert.equal(res.notices, undefined);
725
1051
  });
726
1052
 
727
- test("gbnfDebug: a conflict does NOT throw — it returns the bytes plus a grammar_unenforced telemetry event with the divergence position (#24)", async () => {
1053
+ test("gbnfDebug preserves conflicting bytes without a provider verdict", async () => {
728
1054
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
729
- const calls = installFetch([{ choices: [{ delta: { reasoning_content: "let me think about ok", content: "xon-conforming output" }, finish_reason: "stop" }] }]);
1055
+ const calls = installFetch([{ choices: [{ delta: { content: "xon-conforming output" }, finish_reason: "stop" }] }]);
730
1056
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
731
- // The model's bytes survive — not discarded by a throw (the empty-turn cascade root cause).
732
1057
  assert.equal(res.assistant.content, "xon-conforming output");
733
- assert.equal(res.assistant.reasoning, "let me think about ok");
734
- // Non-fatal telemetry carries the divergence so the consumer can self-correct.
735
- assert.equal(res.telemetry?.length, 1);
736
- const [event] = res.telemetry ?? [];
737
- assert.equal(event.source, "provider:test");
738
- assert.equal(event.kind, "grammar_unenforced");
739
- assert.equal(event.position, 0); // 'x' rejected at code point 0
740
- assert.match(event.message ?? "", /output rejected by the transported grammar at code point 0 \("x"\)/);
1058
+ assert.deepEqual(res.grammarEvidence, { input: "xon-conforming output", contentStart: 0, transported: false });
1059
+ assert.equal(res.notices, undefined);
741
1060
  const body = JSON.parse(calls[0].init.body as string);
742
- assert.equal("grammar" in body, false); // still never sent — diagnosed, not enforced
1061
+ assert.equal("grammar" in body, false);
743
1062
  });
744
1063
 
745
1064
  test("gbnfDebug: an INVALID grammar throws before any wire call — it never reaches the model", async () => {
@@ -752,7 +1071,7 @@ test("gbnfDebug: an INVALID grammar throws before any wire call — it never rea
752
1071
  assert.equal(calls.length, 0); // fail-hard before the fetch — grammar never transported
753
1072
  });
754
1073
 
755
- // — meta bag: verbatim provider metadata (#23) —
1074
+ // — meta bag: verbatim provider metadata —
756
1075
 
757
1076
  test("meta: passes backend fields through without reinterpreting monetary values", async () => {
758
1077
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
@@ -763,7 +1082,7 @@ test("meta: passes backend fields through without reinterpreting monetary values
763
1082
  assert.equal(res.meta?.system_fingerprint, "fp_abc");
764
1083
  });
765
1084
 
766
- // — first-party telemetry headers (attribution + client, SPEC §5) —
1085
+ // — first-party telemetry headers ({§provider-request-authority}) —
767
1086
 
768
1087
  const headerVal = (init: RequestInit, name: string): string | undefined =>
769
1088
  new Headers(init.headers).get(name) ?? undefined;
@@ -776,7 +1095,7 @@ test("firstPartyMetadata: attributions + client ride as Plurnk-* headers", async
776
1095
  assert.equal(headerVal(calls[0].init, "Plurnk-Client"), "plurnk.nvim/1.4.0");
777
1096
  });
778
1097
 
779
- test("#522 Plurnk-Worker-Primary: the lineage root rides under the gate; emitted even when it equals workerId", async () => {
1098
+ test("Plurnk-Worker-Primary: the lineage root rides under the gate; emitted even when it equals workerId", async () => {
780
1099
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
781
1100
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
782
1101
  await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
@@ -795,7 +1114,7 @@ test("#522 Plurnk-Worker-Primary: the lineage root rides under the gate; emitted
795
1114
  assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), undefined);
796
1115
  });
797
1116
 
798
- test("#522 Plurnk-Worker-Primary is structurally dropped when firstPartyMetadata is off", async () => {
1117
+ test("Plurnk-Worker-Primary is structurally dropped when firstPartyMetadata is off", async () => {
799
1118
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
800
1119
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
801
1120
  await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
@@ -818,13 +1137,13 @@ test("firstPartyMetadata on but empty values: no header emitted", async () => {
818
1137
  assert.equal(headerVal(calls[0].init, "Plurnk-Client"), undefined);
819
1138
  });
820
1139
 
821
- test("grammar transport: no grammar passed sends no grammar field, but the penalty rides (#426)", async () => {
1140
+ test("grammar transport: no grammar passed sends no grammar field, but the penalty rides", async () => {
822
1141
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
823
1142
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
824
1143
  await p.generate({ workerId: "r", messages: [] });
825
1144
  const body = JSON.parse(calls[0].init.body as string);
826
1145
  assert.equal("grammar" in body, false);
827
- assert.equal(body.repeat_penalty, 1.15); // #426: penalty is no longer grammar-gated - it rides rail-off
1146
+ assert.equal(body.repeat_penalty, 1.15); // penalty is not grammar-gated
828
1147
  });
829
1148
 
830
1149
  test("maxTokens transports as max_tokens; absent → no wire field (server default)", async () => {
@@ -839,7 +1158,7 @@ test("maxTokens transports as max_tokens; absent → no wire field (server defau
839
1158
  assert.equal("max_tokens" in JSON.parse(calls[0].init.body as string), false);
840
1159
  });
841
1160
 
842
- test("slot affinity is internal: sticky per workerId, distinct runs spread across slots (#11)", async () => {
1161
+ test("slot affinity is internal: sticky per workerId, distinct workers spread across slots", async () => {
843
1162
  const pinning = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
844
1163
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
845
1164
  await pinning.generate({ workerId: "run-A", messages: [] });
@@ -863,7 +1182,7 @@ test("slot affinity: no pinning backend or unknown slotCount → no id_slot ever
863
1182
  assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
864
1183
  });
865
1184
 
866
- test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; recent runs stay sticky (#11)", async () => {
1185
+ test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; recent workers stay sticky", async () => {
867
1186
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
868
1187
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
869
1188
  const slotOf = (i: number) => JSON.parse(calls[i].init.body as string).id_slot;
@@ -877,11 +1196,11 @@ test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; r
877
1196
  });
878
1197
 
879
1198
  test("streaming:false: a non-ok response rejects as a classified ProviderError (covers the non-streamed transport)", async () => {
880
- const { ProviderError } = await import("./telemetry.ts");
1199
+ const { ProviderError } = await import("./errors.ts");
881
1200
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, source: "provider:test" });
882
1201
  mock.method(globalThis, "fetch", async () => new Response("boom", { status: 500 }));
883
1202
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
884
- assert.ok(err instanceof ProviderError);
1203
+ assert.ok(err instanceof ProviderError, `expected ProviderError, got ${String(err)}`);
885
1204
  assert.equal(err.kind, "network_failure"); // ≥500 → network_failure
886
1205
  assert.equal(err.status, 500);
887
1206
  return true;
@@ -904,15 +1223,17 @@ test("messages pass through verbatim — the provider injects no turn (PLAN live
904
1223
  assert.equal(res.assistant.content, "out"); // content returned verbatim
905
1224
  });
906
1225
 
907
- test("generate wraps an HTTP failure as a ProviderError carrying a TelemetryEvent", async () => {
908
- const { ProviderError } = await import("./telemetry.ts");
1226
+ test("generate wraps an HTTP failure as a ProviderError carrying Problem Details", async () => {
1227
+ const { ProviderError } = await import("./errors.ts");
909
1228
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, source: "provider:test" });
910
1229
  mock.method(globalThis, "fetch", async () => new Response("rate limited", { status: 429 }));
911
1230
  await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
912
- assert.ok(err instanceof ProviderError);
1231
+ assert.ok(err instanceof ProviderError, `expected ProviderError, got ${String(err)}`);
913
1232
  assert.equal(err.kind, "rate_limit");
914
1233
  assert.equal(err.status, 429);
915
- assert.deepEqual(err.toTelemetryEvent(), { source: "provider:test", kind: "rate_limit", message: err.message, position: null });
1234
+ assert.equal(err.problem.status, 429);
1235
+ assert.equal(err.problem.detail, err.message);
1236
+ assert.equal(err.problem.type, "https://problems.plurnk.dev/provider/test/rate-limit");
916
1237
  return true;
917
1238
  });
918
1239
  });
@@ -937,7 +1258,7 @@ test("configured headers and url are sent verbatim", async () => {
937
1258
  assert.equal(headers.get("x-title"), "plurnk");
938
1259
  });
939
1260
 
940
- // — transient-failure retry (#18) —
1261
+ // — transient-failure retry —
941
1262
 
942
1263
  const retryCfg = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const };
943
1264
 
@@ -953,7 +1274,7 @@ test("retry: a transient failure retries and a later success resolves", async ()
953
1274
  assert.equal(calls.length, 3); // 429 → 503 → 200
954
1275
  });
955
1276
 
956
- test("#559: streamed-body silence fails the exchange without replaying partial output", async () => {
1277
+ test("streamed-body silence fails the exchange without replaying partial output", async () => {
957
1278
  let calls = 0;
958
1279
  mock.method(globalThis, "fetch", async () => {
959
1280
  calls++;
@@ -996,7 +1317,7 @@ test("#559: streamed-body silence fails the exchange without replaying partial o
996
1317
  mock.restoreAll();
997
1318
  });
998
1319
 
999
- test("#559: a zero stream-idle timeout permits a slow inter-chunk pause", async () => {
1320
+ test("a zero stream-idle timeout permits a slow inter-chunk pause", async () => {
1000
1321
  mock.method(globalThis, "fetch", async () => new Response(new ReadableStream({
1001
1322
  async start(controller) {
1002
1323
  controller.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"slow "}}]}\n\n'));
@@ -1021,7 +1342,7 @@ test("#559: a zero stream-idle timeout permits a slow inter-chunk pause", async
1021
1342
  });
1022
1343
 
1023
1344
  test("retry: exhausting the budget surfaces the classified ProviderError", async () => {
1024
- const { ProviderError } = await import("./telemetry.ts");
1345
+ const { ProviderError } = await import("./errors.ts");
1025
1346
  const calls = installFetchScript([{ status: 429, retryAfter: 0 }]); // always rate-limited
1026
1347
  const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 2 });
1027
1348
  await assert.rejects(
@@ -1056,7 +1377,7 @@ test("retry: retryAttempts 0 surfaces the first transient failure immediately",
1056
1377
  assert.equal(calls.length, 1); // no retry budget
1057
1378
  });
1058
1379
 
1059
- test("retry: a caller abort during backoff rejects promptly with no further attempt (mid-flight abort, SPEC )", async () => {
1380
+ test("retry: a caller abort during backoff rejects promptly with no further attempt", async () => {
1060
1381
  const ac = new AbortController();
1061
1382
  const calls = installFetchScript([{ status: 503, retryAfter: 5 }]); // 5s backoff we never wait out
1062
1383
  const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 3 });
@@ -1068,7 +1389,7 @@ test("retry: a caller abort during backoff rejects promptly with no further atte
1068
1389
  assert.equal(calls.length, 1); // never retried after cancellation
1069
1390
  });
1070
1391
 
1071
- // — anthropic reasoning style (thinking param, #18) —
1392
+ // — Anthropic reasoning style (wire `thinking` parameter) —
1072
1393
 
1073
1394
  test("reasoningStyle 'anthropic' maps the budget to the thinking param", async () => {
1074
1395
  // N>0 → enabled with budget_tokens
@@ -1115,10 +1436,10 @@ test("streaming:false posts without stream and parses the single JSON response",
1115
1436
  mock.restoreAll();
1116
1437
  });
1117
1438
 
1118
- // ── Data capture (#36): logprobs + verbatim rawBody, opt-in, off by default ──
1439
+ // ── Data capture ({§provider-evidence}): logprobs + verbatim rawBody, opt-in, off by default ──
1119
1440
  const captureBase = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
1120
1441
 
1121
- test("#36 logprobs OFF by default: no wire request, no assistant.logprobs, no rawBody", async () => {
1442
+ test("logprobs OFF by default: no wire request, no assistant.logprobs, no rawBody", async () => {
1122
1443
  const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
1123
1444
  const p = new AiSdkProvider({ ...captureBase });
1124
1445
  const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
@@ -1131,7 +1452,7 @@ test("#36 logprobs OFF by default: no wire request, no assistant.logprobs, no ra
1131
1452
  mock.restoreAll();
1132
1453
  });
1133
1454
 
1134
- test("#36 logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw logprob + meanLogprob", async () => {
1455
+ test("logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw logprob + meanLogprob", async () => {
1135
1456
  const chunk = { model: "m", usage: { prompt_tokens: 1, completion_tokens: 2, total_tokens: 3 }, choices: [{ delta: { content: "yesno" }, finish_reason: "stop", logprobs: { content: [
1136
1457
  { token: "yes", logprob: -0.5, sampling_logprob: -0.5, top_logprobs: [{ token: "yes", logprob: -0.5 }, { token: "no", logprob: -1.0 }] },
1137
1458
  { token: "no", logprob: -0.1, sampling_logprob: -0.1, top_logprobs: [{ token: "no", logprob: -0.1 }] },
@@ -1148,7 +1469,7 @@ test("#36 logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw l
1148
1469
  mock.restoreAll();
1149
1470
  });
1150
1471
 
1151
- test("#36 rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob preserved", async () => {
1472
+ test("rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob preserved", async () => {
1152
1473
  const wire = { model: "m", extra_top_level: "kept", choices: [{ message: { content: "no" }, finish_reason: "stop", logprobs: { content: [{ token: "no", logprob: -0.1, sampling_logprob: -0.1, token_id: 42 }] } }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } };
1153
1474
  installFetchJson(wire);
1154
1475
  const p = new AiSdkProvider({ ...captureBase, streaming: false, topLogprobs: 0, rawBody: true });
@@ -1160,7 +1481,7 @@ test("#36 rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob p
1160
1481
  mock.restoreAll();
1161
1482
  });
1162
1483
 
1163
- test("#36 caller sampling cannot forge logprobs (reserved keys): the env flag is the only control", async () => {
1484
+ test("caller sampling cannot forge logprobs (reserved keys): the env flag is the only control", async () => {
1164
1485
  const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
1165
1486
  const p = new AiSdkProvider({ ...captureBase }); // logprobs OFF
1166
1487
  await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }], sampling: { logprobs: true, top_logprobs: 5 } });
@@ -1170,9 +1491,9 @@ test("#36 caller sampling cannot forge logprobs (reserved keys): the env flag is
1170
1491
  mock.restoreAll();
1171
1492
  });
1172
1493
 
1173
- // — turn coordinate headers (#404, per #391): same gate as every first-party signal —
1494
+ // — turn coordinate headers ({§lifecycle-terms}): same gate as every first-party signal —
1174
1495
 
1175
- test("#404: workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under the first-party gate", async () => {
1496
+ test("workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under the first-party gate", async () => {
1176
1497
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1177
1498
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1178
1499
  await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
@@ -1182,7 +1503,7 @@ test("#404: workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under th
1182
1503
  assert.equal(headers.get("plurnk-turn"), "41");
1183
1504
  });
1184
1505
 
1185
- test("#404: third-party providers structurally DROP the coordinate (gate off by default)", async () => {
1506
+ test("third-party providers structurally DROP the coordinate (gate off by default)", async () => {
1186
1507
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1187
1508
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1188
1509
  await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
@@ -1192,7 +1513,7 @@ test("#404: third-party providers structurally DROP the coordinate (gate off by
1192
1513
  assert.equal(headers.has("plurnk-turn"), false);
1193
1514
  });
1194
1515
 
1195
- test("#404: coordinates are 1-based — 0/absent/empty emit no header (no strikes-style zero exception)", async () => {
1516
+ test("coordinates are 1-based — 0/absent/empty emit no header", async () => {
1196
1517
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1197
1518
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1198
1519
  await p.generate({ workerId: "r", messages: [], workspaceId: "", loop: 0, turn: 0 });
@@ -1203,9 +1524,9 @@ test("#404: coordinates are 1-based — 0/absent/empty emit no header (no strike
1203
1524
  assert.equal(headers.has("plurnk-strikes"), false);
1204
1525
  });
1205
1526
 
1206
- // -- #507: envelope surface + router-owned tuning --
1527
+ // -- {§provider-generation-envelope} --
1207
1528
 
1208
- test("#507 reserves derive from the detected window; absolutes stand alone; null window + percent = no claim", () => {
1529
+ test("reserves derive from the detected window; absolutes stand alone; null window + percent = no claim", () => {
1209
1530
  const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
1210
1531
  const derived = new AiSdkProvider({ ...base, contextWindow: 49152, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 } });
1211
1532
  assert.equal(derived.reasoningReserve, 4915); // jennifer/turboderp: 10% of 49152
@@ -1217,32 +1538,32 @@ test("#507 reserves derive from the detected window; absolutes stand alone; null
1217
1538
  assert.equal(legacy.reasoningReserve, null); // out-of-date sibling: no claim
1218
1539
  });
1219
1540
 
1220
- test("#507 router-owned tuning: tuningFloors:false drops the temperature/penalty floors, caller sampling still rides", async () => {
1541
+ test("router-owned tuning: tuningFloors:false drops the temperature/penalty floors, caller sampling still rides", async () => {
1221
1542
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0, tuningFloors: false });
1222
1543
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1223
1544
  await p.generate({ workerId: "r", messages: [], sampling: { temperature: 0.9 } });
1224
1545
  const body = JSON.parse(calls[0].init.body as string);
1225
1546
  assert.equal(body.temperature, 0.9); // caller intent passes verbatim
1226
- assert.equal("frequency_penalty" in body, false); // the floor is suppressed (router owns tuning, SPEC §5)
1547
+ assert.equal("frequency_penalty" in body, false); // the floor is suppressed; the router owns tuning
1227
1548
  });
1228
1549
 
1229
- // -- #518: prompt-cache affinity (workerId -> prompt_cache_key) --
1550
+ // -- prompt-cache affinity (workerId -> prompt_cache_key) --
1230
1551
 
1231
- test("#518 promptCacheKey on: body sends prompt_cache_key = workerId (serverless replica affinity)", async () => {
1552
+ test("promptCacheKey on: body sends prompt_cache_key = workerId (serverless replica affinity)", async () => {
1232
1553
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
1233
1554
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1234
1555
  await p.generate({ workerId: "worker-abc", messages: [] });
1235
1556
  assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc");
1236
1557
  });
1237
1558
 
1238
- test("#518 promptCacheKey off (default): no prompt_cache_key on the wire", async () => {
1559
+ test("promptCacheKey off (default): no prompt_cache_key on the wire", async () => {
1239
1560
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1240
1561
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1241
1562
  await p.generate({ workerId: "worker-abc", messages: [] });
1242
1563
  assert.equal("prompt_cache_key" in JSON.parse(calls[0].init.body as string), false);
1243
1564
  });
1244
1565
 
1245
- test("#518 prompt_cache_key is managed: caller sampling cannot forge/override the affinity key", async () => {
1566
+ test("prompt_cache_key is managed: caller sampling cannot forge/override the affinity key", async () => {
1246
1567
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
1247
1568
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1248
1569
  await p.generate({ workerId: "worker-abc", messages: [], sampling: { prompt_cache_key: "hijack" } });