@plurnk/plurnk-providers 1.16.4 → 1.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/.env.defaults +87 -200
  2. package/README.md +2 -1
  3. package/SPEC.md +73 -66
  4. package/dist/AiSdkProvider.d.ts +0 -1
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +25 -55
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/AiSdkRequestBody.d.ts +0 -1
  9. package/dist/AiSdkRequestBody.d.ts.map +1 -1
  10. package/dist/AiSdkRequestBody.js +1 -17
  11. package/dist/AiSdkRequestBody.js.map +1 -1
  12. package/dist/LeadingReasoning.d.ts +16 -0
  13. package/dist/LeadingReasoning.d.ts.map +1 -0
  14. package/dist/LeadingReasoning.js +81 -0
  15. package/dist/LeadingReasoning.js.map +1 -0
  16. package/dist/aiSdkTransport.d.ts +7 -0
  17. package/dist/aiSdkTransport.d.ts.map +1 -1
  18. package/dist/aiSdkTransport.js +19 -5
  19. package/dist/aiSdkTransport.js.map +1 -1
  20. package/dist/catalogProvider.d.ts +3 -1
  21. package/dist/catalogProvider.d.ts.map +1 -1
  22. package/dist/catalogProvider.js +24 -18
  23. package/dist/catalogProvider.js.map +1 -1
  24. package/dist/compatibleProvider.d.ts.map +1 -1
  25. package/dist/compatibleProvider.js +0 -3
  26. package/dist/compatibleProvider.js.map +1 -1
  27. package/dist/env.d.ts.map +1 -1
  28. package/dist/env.js +0 -1
  29. package/dist/env.js.map +1 -1
  30. package/dist/errors.d.ts +6 -0
  31. package/dist/errors.d.ts.map +1 -1
  32. package/dist/errors.js +46 -15
  33. package/dist/errors.js.map +1 -1
  34. package/dist/index.d.ts +2 -2
  35. package/dist/index.d.ts.map +1 -1
  36. package/dist/index.js +2 -1
  37. package/dist/index.js.map +1 -1
  38. package/dist/providerError.d.ts +1 -1
  39. package/dist/providerError.d.ts.map +1 -1
  40. package/dist/providerError.js +3 -0
  41. package/dist/providerError.js.map +1 -1
  42. package/dist/sdkModels.d.ts +1 -13
  43. package/dist/sdkModels.d.ts.map +1 -1
  44. package/dist/sdkModels.js +2 -94
  45. package/dist/sdkModels.js.map +1 -1
  46. package/dist/types.d.ts.map +1 -1
  47. package/docs/models.md +90 -0
  48. package/package.json +6 -6
  49. package/src/AiSdkProvider.test.ts +99 -38
  50. package/src/AiSdkProvider.ts +24 -78
  51. package/src/AiSdkRequestBody.ts +1 -16
  52. package/src/LeadingReasoning.test.ts +34 -0
  53. package/src/LeadingReasoning.ts +87 -0
  54. package/src/ProviderRegistry.test.ts +19 -1
  55. package/src/aiSdkTransport.test.ts +56 -0
  56. package/src/aiSdkTransport.ts +31 -5
  57. package/src/boundaries.test.ts +1 -0
  58. package/src/catalogProvider.test.ts +110 -20
  59. package/src/catalogProvider.ts +40 -18
  60. package/src/compatibleProvider.ts +0 -3
  61. package/src/env.test.ts +1 -0
  62. package/src/env.ts +1 -2
  63. package/src/errors.test.ts +58 -7
  64. package/src/errors.ts +55 -18
  65. package/src/index.ts +2 -2
  66. package/src/providerError.ts +4 -0
  67. package/src/sdkModels.test.ts +1 -42
  68. package/src/sdkModels.ts +1 -123
  69. package/src/types.ts +8 -10
@@ -1061,6 +1061,45 @@ test("generate aggregates reasoning deltas under multiple field names", async ()
1061
1061
  assert.equal("reasoningEncrypted" in assistant, false); // open reasoning only -> field absent
1062
1062
  });
1063
1063
 
1064
+ for (const style of ["think-tags", "template-think", "template-channel"] as const) test(`{§provider-reasoning-observer} ${style} reasoning arrives while the response is still open`, async () => {
1065
+ const template = style !== "think-tags";
1066
+ const opening = style === "template-channel" ? "<|channel>thought\n" : template ? "<think>\n" : "<think>";
1067
+ const closing = style === "template-channel" ? "<channel|>" : "</think>";
1068
+ const observed: string[] = [];
1069
+ const ready = Promise.withResolvers<void>();
1070
+ let controller!: ReadableStreamDefaultController<Uint8Array>;
1071
+ const send = (text: string, finish: string | null = null): void => controller.enqueue(new TextEncoder().encode(
1072
+ `data: ${JSON.stringify({ id: "paced", object: "chat.completion.chunk", created: 1, model: "m", choices: [{ index: 0, delta: { content: text }, finish_reason: finish }] })}\n\n`,
1073
+ ));
1074
+ const fetch: typeof globalThis.fetch = async () => new Response(new ReadableStream<Uint8Array>({ start(stream) {
1075
+ controller = stream;
1076
+ send(opening.slice(0, 3));
1077
+ send(`${opening.slice(3)}Thinking 🧠${closing.slice(0, 3)}`);
1078
+ } }), { headers: { "Content-Type": "text/event-stream" } });
1079
+ const provider = testProvider({ ...injectedBase, fetch, rawBody: true, ...(template
1080
+ ? { reasoningStyle: "template" as const, grammarStyle: "llamacpp" as const, reasoning: { mode: "adaptive" as const, budget: null } }
1081
+ : { reasoningResponseStyle: "think-tags" as const }) });
1082
+ const result = provider.generate({ workerId: "paced", messages: [], ...(template ? { grammar: 'root ::= "x"' } : {}), observeReasoning(delta) {
1083
+ observed.push(delta);
1084
+ if (observed.join("") === "Thinking 🧠") ready.resolve();
1085
+ } });
1086
+ const timer = setTimeout(() => ready.reject(new Error("reasoning was buffered until response completion")), 1500);
1087
+ try {
1088
+ await ready.promise;
1089
+ assert.deepEqual(observed, ["Thinking 🧠"], "an incomplete closing delimiter is not reasoning text");
1090
+ } finally {
1091
+ clearTimeout(timer);
1092
+ send(`${closing.slice(3)}Answer. <think>literal suffix</think>`, "stop");
1093
+ controller.enqueue(new TextEncoder().encode("data: [DONE]\n\n"));
1094
+ controller.close();
1095
+ const response = await result;
1096
+ assert.equal(response.assistant.reasoning, "Thinking 🧠");
1097
+ assert.equal(response.assistant.content, "Answer. <think>literal suffix</think>");
1098
+ assert.equal(observed.join(""), "Thinking 🧠", "completion does not replay streamed reasoning");
1099
+ if (template) assert.equal(response.grammarEvidence?.input, `${opening}Thinking 🧠${closing}Answer. <think>literal suffix</think>`);
1100
+ }
1101
+ });
1102
+
1064
1103
  test("{§provider-tagged-reasoning} explicit think-tags project content without estimating token attribution", async () => {
1065
1104
  const config = { ...injectedBase, reasoningResponseStyle: "think-tags" as const, rawBody: true };
1066
1105
  const p = testProvider(config);
@@ -1089,6 +1128,17 @@ test("{§provider-tagged-reasoning} explicit think-tags project content without
1089
1128
  assert.match(JSON.stringify(response.rawBody), /nk>12345/);
1090
1129
  });
1091
1130
 
1131
+ test("{§provider-tagged-reasoning} structured streamed reasoning keeps tagged content literal", async () => {
1132
+ installFetch([{ choices: [{ delta: { content: "<think>literal</think>", reasoning_content: "Native thought." }, finish_reason: "stop" }] }]);
1133
+ const observed: string[] = [];
1134
+ const response = await testProvider({ ...injectedBase, reasoningResponseStyle: "think-tags" }).generate({
1135
+ workerId: "native-control", messages: [], observeReasoning: (delta) => observed.push(delta),
1136
+ });
1137
+ assert.equal(response.assistant.reasoning, "Native thought.");
1138
+ assert.equal(response.assistant.content, "<think>literal</think>");
1139
+ assert.deepEqual(observed, ["Native thought."]);
1140
+ });
1141
+
1092
1142
  test("{§provider-tagged-reasoning} explicit think-tags projects one buffered leading envelope", async () => {
1093
1143
  installFetchJson({
1094
1144
  model: "m",
@@ -1201,7 +1251,7 @@ test("{§provider-tagged-reasoning} verbatim, non-leading, and structured-reason
1201
1251
  });
1202
1252
 
1203
1253
  test("{§provider-tagged-reasoning} grammar evidence retains the exact pre-projection tagged sentence", async () => {
1204
- const content = "<think>🧠reason</think>## PLAN0\n\n### SEND0 (TERM)\ndone";
1254
+ const content = "<think>🧠reason</think>```SEND\ndone\n```\n```TASK\n[{\"content\":\"Task completed.\",\"status\":\"completed\"}]\n```";
1205
1255
  const config = {
1206
1256
  ...injectedBase,
1207
1257
  contextWindow: 640,
@@ -1219,7 +1269,7 @@ test("{§provider-tagged-reasoning} grammar evidence retains the exact pre-proje
1219
1269
  });
1220
1270
 
1221
1271
  assert.equal(response.assistant.reasoning, "🧠reason");
1222
- assert.equal(response.assistant.content, "## PLAN0\n\n### SEND0 (TERM)\ndone");
1272
+ assert.equal(response.assistant.content, "```SEND\ndone\n```\n```TASK\n[{\"content\":\"Task completed.\",\"status\":\"completed\"}]\n```");
1223
1273
  assert.deepEqual(response.grammarEvidence, {
1224
1274
  input: content,
1225
1275
  contentStart: [..."<think>🧠reason</think>"].length,
@@ -1516,7 +1566,7 @@ test("sampling passthrough guards contract invariants: n/tools/caps stripped, pl
1516
1566
  assert.equal(body.service_tier, "flex");
1517
1567
  });
1518
1568
 
1519
- test("template reasoning returns the exact pre-projection grammar sentence ({§gbnf-response-observation})", async () => {
1569
+ test("template reasoning returns the exact pre-projection grammar sentence ({§provider-grammar-evidence})", async () => {
1520
1570
  const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, outputBudget: 224, reasoningBudget: 64, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: 64 }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1521
1571
  const grammarInput = "<|channel>thought\ncon🙂sider<channel|>x";
1522
1572
  const calls = installFetch([{ choices: [{ delta: { content: grammarInput } }] }]);
@@ -1724,7 +1774,7 @@ test("grammar transport 'none' (default): the grammar is never sent — no silen
1724
1774
  assert.equal("response_format" in body, false);
1725
1775
  });
1726
1776
 
1727
- // — exact pre-projection grammar evidence ({§gbnf-response-observation}) —
1777
+ // — exact pre-projection grammar evidence ({§provider-grammar-evidence}) —
1728
1778
 
1729
1779
  const grammarProvider = () => testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", source: "provider:test" });
1730
1780
  const streamingContent = (content: string) => installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
@@ -1776,39 +1826,24 @@ test("provider evidence does not depend on the local validator understanding the
1776
1826
  assert.equal(res.notices, undefined);
1777
1827
  });
1778
1828
 
1779
- // — PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar, withhold it, and preserve the observation —
1780
-
1781
- test("gbnfDebug marks an unconstrained observation as not transported", async () => {
1782
- const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
1783
- const calls = installFetch([{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }]);
1784
- const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
1785
- const body = JSON.parse(calls[0].init.body as string);
1786
- assert.equal("grammar" in body, false);
1787
- assert.equal(body.repeat_penalty, 1.15);
1788
- assert.equal(res.assistant.content, "ok");
1789
- assert.deepEqual(res.grammarEvidence, { input: "ok", contentStart: 0, transported: false });
1790
- assert.equal(res.notices, undefined);
1791
- });
1792
-
1793
- test("gbnfDebug preserves conflicting bytes without a provider verdict", async () => {
1794
- const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
1795
- const calls = installFetch([{ choices: [{ delta: { content: "xon-conforming output" }, finish_reason: "stop" }] }]);
1796
- const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
1797
- assert.equal(res.assistant.content, "xon-conforming output");
1798
- assert.deepEqual(res.grammarEvidence, { input: "xon-conforming output", contentStart: 0, transported: false });
1799
- assert.equal(res.notices, undefined);
1800
- const body = JSON.parse(calls[0].init.body as string);
1801
- assert.equal("grammar" in body, false);
1802
- });
1803
-
1804
- test("gbnfDebug: an INVALID grammar throws before any wire call — it never reaches the model", async () => {
1805
- const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true });
1806
- const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1807
- await assert.rejects(
1808
- () => p.generate({ workerId: "r", messages: [], grammar: 'foo ::= "a"' }), // no `root` rule → invalid GBNF
1809
- /grammar validation \(PLURNK_PROVIDERS_GBNF_DEBUG\): invalid GBNF/,
1810
- );
1811
- assert.equal(calls.length, 0); // fail-hard before the fetch — grammar never transported
1829
+ // — {§provider-grammar-transport}: the operator's grammar rides a llama-style route verbatim, never another —
1830
+
1831
+ test("an operator grammar reaches the request body verbatim under llama style and is absent under every other style", async () => {
1832
+ const grammar = 'root ::= "ok" | "fine"\n# an operator-written rail\n';
1833
+ const styles = ["none", "llamacpp"] as const;
1834
+ for (const grammarStyle of styles) {
1835
+ const calls = installFetchJson({ ...jsonChoice, choices: [{ message: { role: "assistant", content: "ok" }, finish_reason: "stop" }] });
1836
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, grammarStyle });
1837
+ const res = await p.generate({ workerId: "r", messages: [], grammar });
1838
+ const body = JSON.parse(String(calls[0]!.init.body)) as { grammar?: unknown };
1839
+ if (grammarStyle === "llamacpp") {
1840
+ assert.equal(body.grammar, grammar, "the file's text, byte for byte");
1841
+ assert.deepEqual(res.grammarEvidence, { input: "ok", contentStart: 0, transported: true });
1842
+ } else {
1843
+ assert.equal("grammar" in body, false, "no grammar field on a route without transport");
1844
+ assert.equal(res.grammarEvidence, undefined);
1845
+ }
1846
+ }
1812
1847
  });
1813
1848
 
1814
1849
  // — meta bag: verbatim provider metadata —
@@ -2020,7 +2055,7 @@ test("generate fail-hards on a missing or empty workerId", async () => {
2020
2055
  await assert.rejects(() => (p.generate as (a: object) => Promise<unknown>)({ messages: [] }), /workerId is required/);
2021
2056
  });
2022
2057
 
2023
- test("messages pass through verbatim — the provider injects no turn (PLAN lives in the grammar, never a provider prefill)", async () => {
2058
+ test("messages pass through verbatim — the provider injects no turn (turn structure belongs to the grammar, never provider prefill)", async () => {
2024
2059
  const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
2025
2060
  const calls = installFetch([{ choices: [{ delta: { content: "out" } }] }]);
2026
2061
  const input = [{ role: "user" as const, content: "hi" }];
@@ -2927,3 +2962,29 @@ test("a manual-reasoning model rejects an envelope below its provider minimum be
2927
2962
  );
2928
2963
  assert.equal(calls, 0);
2929
2964
  });
2965
+
2966
+ // {§provider-usage-refusal} (#580) — through the provider: the response is delivered, its
2967
+ // accounting is an honest unknown, and the refused counters are durable evidence, not a 503.
2968
+ test("a response whose usage counters contradict each other is delivered with unknown cost, never retried as a network failure", async () => {
2969
+ const usage = { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6, completion_tokens_details: { reasoning_tokens: 7 } };
2970
+ installFetchJson({
2971
+ id: "response-1", object: "chat.completion", created: 1, model: "served-model",
2972
+ choices: [{ index: 0, message: { role: "assistant", content: "the answer" }, finish_reason: "stop" }],
2973
+ usage,
2974
+ });
2975
+ const p = testProvider({
2976
+ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: null, repeatPenalty: null, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
2977
+ streaming: false,
2978
+ estimateCost: (known) => known === undefined
2979
+ ? { kind: "unknown", reason: "the provider response reported no normalized usage" }
2980
+ : { kind: "estimated", amount: { amount: "1", currency: "USD" }, source: "test estimator" },
2981
+ });
2982
+ const response = await p.generate({ workerId: "accounted", messages: [{ role: "user", content: "q" }] });
2983
+ assert.equal(response.assistant.content, "the answer");
2984
+ assert.equal(response.accounting.length, 1);
2985
+ assert.equal(response.accounting[0]?.outcome, "response");
2986
+ assert.equal(response.accounting[0]?.usage, undefined, "no counters are reported as known");
2987
+ assert.deepEqual(response.accounting[0]?.cost, { kind: "unknown", reason: "the provider response reported no normalized usage" });
2988
+ const raw = response.assistantRaw as { usageRefusal?: { reason: string; usage: unknown } };
2989
+ assert.deepEqual(raw.usageRefusal, { reason: "provider usage.outputTokenDetails.textTokens must be a non-negative safe integer", usage }, "the durable response carries the refused counters");
2990
+ });
@@ -27,12 +27,12 @@ import { validateProviderUsage } from "./usage.ts";
27
27
  import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.ts";
28
28
  import { nativeFixedEffort } from "./reasoning-effort.ts";
29
29
  import AiSdkRequestBody from "./AiSdkRequestBody.ts";
30
+ import LeadingReasoning from "./LeadingReasoning.ts";
30
31
 
31
32
  export type ProviderFetch = typeof globalThis.fetch;
32
33
 
33
34
  // {§provider-connectivity} — an AbortSignal is advisory: a wedged transport that never observes it can
34
- // hang the await past the deadline (#505). This backstops the deadline the same way an embedding request
35
- // bounds a wedged adapter (#463): once the signal fires, a well-behaved transport unwinds and settles its
35
+ // hang the await past the deadline (#505). Once the signal fires, a well-behaved transport unwinds and settles its
36
36
  // own attempt (and accounting) within a short grace, and that path wins the race untouched; only a
37
37
  // transport still wedged after the grace is force-rejected with the signal's own reason, so the existing
38
38
  // operation/cancellation classification is unchanged. The grace is unwind slack, not a second deadline.
@@ -125,7 +125,6 @@ export type AiSdkProviderConfig = {
125
125
  // Optional provider-configured service tier. Unlike caller sampling, this is
126
126
  // a fixed deployment choice and therefore wins on every request.
127
127
  serviceTier?: string;
128
- gbnfDebug?: boolean; // PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar locally + throw on invalid, but DON'T transport it (run unconstrained); default false
129
128
  streaming?: boolean; // SSE transport (default true); false → one non-streamed JSON
130
129
  firstPartyMetadata?: boolean; // forward per-turn attributions + client as Plurnk-* headers (plurnk only); default false
131
130
  apiKeyRejectedMessage?: string; // friendly hint when a present key is 401/403-rejected (distinct from unset); default undefined
@@ -239,66 +238,6 @@ const stripTrailingSpecial = (content: string, marker: string): string => {
239
238
  return out;
240
239
  };
241
240
 
242
- type TaggedReasoningProjection = {
243
- readonly content: string;
244
- readonly reasoning: string;
245
- readonly projected: boolean;
246
- readonly contentStart: number;
247
- };
248
-
249
- const projectLeadingReasoning = (
250
- content: string,
251
- structuredReasoning: string,
252
- opening: string,
253
- closing: string,
254
- ): TaggedReasoningProjection => {
255
- if (structuredReasoning.length > 0 || !content.startsWith(opening)) {
256
- return { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
257
- }
258
- const closingIndex = content.indexOf(closing, opening.length);
259
- if (closingIndex === -1) {
260
- return {
261
- content: "",
262
- reasoning: content.slice(opening.length),
263
- projected: true,
264
- contentStart: [...content].length,
265
- };
266
- }
267
- const suffixStart = closingIndex + closing.length;
268
- return {
269
- content: content.slice(suffixStart),
270
- reasoning: content.slice(opening.length, closingIndex),
271
- projected: true,
272
- contentStart: [...content.slice(0, suffixStart)].length,
273
- };
274
- };
275
-
276
- // {§provider-tagged-reasoning} Only the model-contract position is structural:
277
- // one exact leading envelope. Parsing after stream assembly keeps SSE and JSON
278
- // on one path and leaves later literal tags in the visible suffix untouched.
279
- const projectTaggedReasoning = (
280
- content: string,
281
- structuredReasoning: string,
282
- style: ReasoningResponseStyle,
283
- ): TaggedReasoningProjection => style === "think-tags"
284
- ? projectLeadingReasoning(content, structuredReasoning, "<think>", "</think>")
285
- : { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
286
-
287
- // llama-server's template reasoning parser can project either supported leading
288
- // reasoning envelope out of the OpenAI-compatible response. Grammar evidence
289
- // needs the sentence before that lossy projection, so constrained template turns
290
- // request it verbatim and split the observed enclosure here.
291
- const projectTemplateReasoning = (content: string): TaggedReasoningProjection => {
292
- for (const [opening, closing] of [
293
- ["<|channel>thought\n", "<channel|>"],
294
- ["<think>\n", "</think>"],
295
- ] as const) {
296
- if (content.startsWith(opening)) return projectLeadingReasoning(content, "", opening, closing);
297
- }
298
- return { content, reasoning: "", projected: false, contentStart: 0 };
299
- };
300
-
301
-
302
241
  const providerWarningMessage = (warning: CallWarning): string => {
303
242
  switch (warning.type) {
304
243
  case "unsupported":
@@ -356,7 +295,6 @@ export default class AiSdkProvider implements Provider {
356
295
  #systemCacheProviderOptions: AiSdkProviderOptions | undefined;
357
296
  #reasoningResponseProviderOptions: AiSdkProviderOptions | undefined;
358
297
  #serviceTier: string | undefined;
359
- #gbnfDebug: boolean;
360
298
  #streaming: boolean;
361
299
  #firstPartyMetadata: boolean;
362
300
  #supportsSlotPinning: boolean;
@@ -479,7 +417,6 @@ export default class AiSdkProvider implements Provider {
479
417
  throw new Error(`${this.#source}: reasoning response options conflict with cache-affinity option ${this.#cacheAffinity.provider}.${this.#cacheAffinity.name}`);
480
418
  }
481
419
  this.#serviceTier = config.serviceTier;
482
- this.#gbnfDebug = config.gbnfDebug ?? false;
483
420
  this.#streaming = config.streaming ?? true;
484
421
  this.#firstPartyMetadata = config.firstPartyMetadata ?? false;
485
422
  this.#apiKeyRejectedMessage = config.apiKeyRejectedMessage;
@@ -686,11 +623,10 @@ export default class AiSdkProvider implements Provider {
686
623
  // ({§provider-failure-normalization}).
687
624
  signal?.throwIfAborted();
688
625
 
689
- // Grammar handling ({§gbnf-response-observation}). Debug validates the
690
- // supplied grammar before the call but withholds it from the backend.
626
+ // Grammar transport ({§provider-grammar-transport}): an operator's grammar rides a
627
+ // llama-style route verbatim and is never sent on any other style.
691
628
  const wantGrammar = grammar !== undefined && this.#grammarStyle !== "none";
692
- if (wantGrammar && this.#gbnfDebug) this.#requestBody.assertGrammarValid(grammar!);
693
- const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
629
+ const sendGrammar = wantGrammar ? grammar : undefined;
694
630
  const preserveGrammarSentence = wantGrammar
695
631
  && this.#reasoningStyle === "template";
696
632
 
@@ -767,12 +703,25 @@ export default class AiSdkProvider implements Provider {
767
703
  let recoveredAfterOutput = false;
768
704
  const executeRequest = async () => {
769
705
  let requestReasoningStream = "";
706
+ let structuredReasoning = false;
707
+ const envelopes = preserveGrammarSentence ? LeadingReasoning.TEMPLATE
708
+ : this.#reasoningResponseStyle === "think-tags" ? LeadingReasoning.THINK : [];
709
+ const liveProjection = emitReasoning !== undefined && envelopes.length > 0 ? new LeadingReasoning(envelopes) : undefined;
770
710
  const observeRequestReasoning = emitReasoning === undefined
771
711
  ? undefined
772
712
  : (delta: string): void => {
773
713
  requestReasoningStream += delta;
774
714
  emitReasoning(delta);
775
715
  };
716
+ const observers = {
717
+ ...(observeRequestReasoning === undefined ? {} : { observeReasoning: (delta: string): void => {
718
+ structuredReasoning = true;
719
+ observeRequestReasoning(delta);
720
+ } }),
721
+ ...(liveProjection === undefined ? {} : { observeText: (delta: string): void => {
722
+ if (!structuredReasoning) observeRequestReasoning!(liveProjection.push(delta));
723
+ } }),
724
+ };
776
725
  let settle: ProviderRequestSettlement | undefined;
777
726
  try {
778
727
  settle = await observeRequest?.({
@@ -839,7 +788,7 @@ export default class AiSdkProvider implements Provider {
839
788
  streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
840
789
  streaming: this.#streaming,
841
790
  captureRawBody: this.#rawBody,
842
- ...(observeRequestReasoning === undefined ? {} : { observeReasoning: observeRequestReasoning }),
791
+ ...observers,
843
792
  })
844
793
  : await executeAiSdkModel({
845
794
  languageModel: this.#languageModel,
@@ -853,7 +802,7 @@ export default class AiSdkProvider implements Provider {
853
802
  streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
854
803
  streaming: this.#streaming,
855
804
  captureRawBody: this.#rawBody,
856
- ...(observeRequestReasoning === undefined ? {} : { observeReasoning: observeRequestReasoning }),
805
+ ...observers,
857
806
  temperature: this.#tuningFloors
858
807
  ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature ?? undefined)
859
808
  : typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
@@ -887,6 +836,7 @@ export default class AiSdkProvider implements Provider {
887
836
  );
888
837
  throw error;
889
838
  }
839
+ if (!structuredReasoning && liveProjection !== undefined) observeRequestReasoning!(liveProjection.finish());
890
840
  successfulReasoningStream = requestReasoningStream;
891
841
  await settleAccounting(
892
842
  "response",
@@ -950,13 +900,9 @@ export default class AiSdkProvider implements Provider {
950
900
  if (this.#eosText !== undefined) raw.content = stripTrailingSpecial(raw.content, this.#eosText);
951
901
 
952
902
  const grammarInput = raw.content;
953
- const projectedReasoning = preserveGrammarSentence && !raw.reasoningProjected
954
- ? projectTemplateReasoning(raw.content)
955
- : projectTaggedReasoning(
956
- raw.content,
957
- raw.reasoning,
958
- this.#reasoningResponseStyle,
959
- );
903
+ const projectedReasoning = LeadingReasoning.project(raw.content, raw.reasoning,
904
+ preserveGrammarSentence && !raw.reasoningProjected ? LeadingReasoning.TEMPLATE
905
+ : this.#reasoningResponseStyle === "think-tags" ? LeadingReasoning.THINK : []);
960
906
 
961
907
  // Preserve the exact pre-projection response. Constrained template turns
962
908
  // request `reasoning_format: "none"`, so even an empty channel and any
@@ -2,7 +2,6 @@
2
2
  import type { ProviderCallKind, ReasoningPolicy } from "./types.ts";
3
3
  import type { JSONValue } from "ai";
4
4
  import { type Reasoning } from "./env.ts";
5
- import { validateGbnf } from "@plurnk/gbnf";
6
5
  import { fixedEffort } from "./reasoning-effort.ts";
7
6
  import type { ReasoningStyle, CompatibleReasoningEffort, GrammarStyle, CacheAffinity, AiSdkProviderOptions } from "./AiSdkProvider.ts";
8
7
 
@@ -224,7 +223,7 @@ export default class AiSdkRequestBody {
224
223
  // bookkeeping so a long-lived daemon never grows the map unboundedly —
225
224
  // an evicted-and-returning run simply re-pins, worst case one cold prefill.
226
225
 
227
- // Optional local llama-server GBNF transport ({§gbnf-response-observation}). Unsupported
226
+ // Optional local llama-server GBNF transport ({§provider-grammar-transport}). Unsupported
228
227
  // backends receive no grammar-related field.
229
228
  grammarBody(grammar: string | undefined): Record<string, unknown> {
230
229
  if (grammar === undefined) return {};
@@ -401,19 +400,5 @@ export default class AiSdkRequestBody {
401
400
  }
402
401
 
403
402
 
404
- // PLURNK_PROVIDERS_GBNF_DEBUG ({§gbnf-response-observation}): validate the supplied GBNF locally and fail
405
- // hard if it's malformed, BEFORE any wire call — and the grammar is NOT
406
- // transported, so the request runs unconstrained. A debug aid to catch invalid
407
- // grammars (e.g. while editing the plurnk grammar) without a model round-trip;
408
- // off in production. `validateGbnf(grammar, "")` parses the grammar + resolves
409
- // its root, throwing iff the grammar itself is invalid (the empty input's
410
- // verdict is irrelevant — we only care that parsing succeeded).
411
- assertGrammarValid(grammar: string): void {
412
- try {
413
- validateGbnf(grammar, "");
414
- } catch (cause) {
415
- throw new Error(`grammar validation (PLURNK_PROVIDERS_GBNF_DEBUG): invalid GBNF — ${(cause as Error).message}`, { cause });
416
- }
417
- }
418
403
 
419
404
  }
@@ -0,0 +1,34 @@
1
+ import assert from "node:assert/strict";
2
+ import test from "node:test";
3
+ import LeadingReasoning from "./LeadingReasoning.ts";
4
+
5
+ test("{§provider-tagged-reasoning}: every delimiter split preserves exact reasoning and the suffix", () => {
6
+ for (const envelopes of [LeadingReasoning.THINK, LeadingReasoning.TEMPLATE]) {
7
+ for (const [opening, closing] of envelopes) {
8
+ const thought = "Decision 🧠\nA < B.\n";
9
+ const suffix = "Answer with literal <think>tags</think>.";
10
+ for (const ended of [true, false]) {
11
+ const input = `${opening}${thought}${ended ? closing + suffix : closing.slice(0, 3)}`;
12
+ const expected = thought + (ended ? "" : closing.slice(0, 3));
13
+ for (let split = 0; split <= input.length; split++) {
14
+ const parser = new LeadingReasoning(envelopes);
15
+ const observed = parser.push(input.slice(0, split)) + parser.push(input.slice(split)) + parser.finish();
16
+ assert.equal(observed, expected, `${opening}: split ${split}`);
17
+ }
18
+ const settled = LeadingReasoning.project(input, "", envelopes);
19
+ assert.equal(settled.reasoning, expected);
20
+ assert.equal(settled.content, ended ? suffix : "");
21
+ assert.equal(settled.contentStart, [...input].length - [...settled.content].length);
22
+ }
23
+ }
24
+ }
25
+ });
26
+
27
+ test("{§provider-tagged-reasoning}: an incomplete opener, later tags, and structured reasoning stay literal", () => {
28
+ for (const input of ["<thi", " <think>text</think>", "prefix <think>text</think>", ""]) {
29
+ assert.deepEqual(LeadingReasoning.project(input, "", LeadingReasoning.THINK), { content: input, reasoning: "", projected: false, contentStart: 0 });
30
+ }
31
+ assert.deepEqual(LeadingReasoning.project("<think>literal</think>", "native", LeadingReasoning.THINK), {
32
+ content: "<think>literal</think>", reasoning: "native", projected: false, contentStart: 0,
33
+ });
34
+ });
@@ -0,0 +1,87 @@
1
+ export type ReasoningEnvelope = readonly [opening: string, closing: string];
2
+
3
+ // {§provider-tagged-reasoning} One leading envelope, shared by incremental
4
+ // observation and settled response projection. Only an undecided delimiter is held.
5
+ export default class LeadingReasoning {
6
+ static readonly THINK: readonly ReasoningEnvelope[] = [["<think>", "</think>"]];
7
+ static readonly TEMPLATE: readonly ReasoningEnvelope[] = [
8
+ ["<|channel>thought\n", "<channel|>"],
9
+ ["<think>\n", "</think>"],
10
+ ];
11
+
12
+ readonly #envelopes: readonly ReasoningEnvelope[];
13
+ #pending = "";
14
+ #closing: string | null = null;
15
+ #closed = false;
16
+ #projected = false;
17
+ #reasoning = "";
18
+ #content = "";
19
+ #contentStart = 0;
20
+
21
+ constructor(envelopes: readonly ReasoningEnvelope[]) {
22
+ this.#envelopes = envelopes;
23
+ }
24
+
25
+ push(text: string): string {
26
+ if (this.#closed) {
27
+ this.#content += text;
28
+ return "";
29
+ }
30
+ this.#pending += text;
31
+ if (this.#closing === null) {
32
+ const envelope = this.#envelopes.find(([opening]) => this.#pending.startsWith(opening));
33
+ if (envelope === undefined) {
34
+ if (!this.#envelopes.some(([opening]) => opening.startsWith(this.#pending))) {
35
+ this.#content = this.#pending;
36
+ this.#pending = "";
37
+ this.#closed = true;
38
+ }
39
+ return "";
40
+ }
41
+ const [opening, closing] = envelope;
42
+ this.#pending = this.#pending.slice(opening.length);
43
+ this.#closing = closing;
44
+ this.#projected = true;
45
+ this.#contentStart = [...opening].length;
46
+ }
47
+ const end = this.#pending.indexOf(this.#closing);
48
+ if (end !== -1) {
49
+ const delta = this.#emit(this.#pending.slice(0, end));
50
+ this.#contentStart += [...this.#closing].length;
51
+ this.#content = this.#pending.slice(end + this.#closing.length);
52
+ this.#pending = "";
53
+ this.#closed = true;
54
+ return delta;
55
+ }
56
+ let held = Math.min(this.#pending.length, this.#closing.length - 1);
57
+ while (held > 0 && !this.#pending.endsWith(this.#closing.slice(0, held))) held--;
58
+ const split = this.#pending.length - held;
59
+ const delta = this.#emit(this.#pending.slice(0, split));
60
+ this.#pending = this.#pending.slice(split);
61
+ return delta;
62
+ }
63
+
64
+ finish(): string {
65
+ const delta = this.#projected ? this.#emit(this.#pending) : "";
66
+ if (!this.#projected) this.#content += this.#pending;
67
+ this.#pending = "";
68
+ this.#closed = true;
69
+ return delta;
70
+ }
71
+
72
+ #emit(delta: string): string {
73
+ this.#reasoning += delta;
74
+ this.#contentStart += [...delta].length;
75
+ return delta;
76
+ }
77
+
78
+ static project(content: string, reasoning: string, envelopes: readonly ReasoningEnvelope[]): {
79
+ content: string; reasoning: string; projected: boolean; contentStart: number;
80
+ } {
81
+ if (reasoning.length > 0) return { content, reasoning, projected: false, contentStart: 0 };
82
+ const parser = new LeadingReasoning(envelopes);
83
+ parser.push(content);
84
+ parser.finish();
85
+ return { content: parser.#content, reasoning: parser.#reasoning, projected: parser.#projected, contentStart: parser.#contentStart };
86
+ }
87
+ }
@@ -1,6 +1,22 @@
1
1
  import test, { mock } from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
3
  import { instantiateProvider, loadActiveProvider, resetDiscoveryCache } from "./ProviderRegistry.ts";
4
+ import { resolveModel } from "@plurnk/plurnk-models";
5
+ import { calculateCostUsdDecimal } from "./usage.ts";
6
+
7
+ // A rate literal breaks at every catalog refresh (the 1.17.0 stamp moved DeepSeek's rates); the
8
+ // claim under test is that the estimate is the catalog's rates applied to the reported usage.
9
+ const catalogRatesOf = (provider: string, model: string) => {
10
+ const cost = resolveModel(provider, model)?.info.cost;
11
+ if (cost === undefined) throw new Error(`${provider}/${model} has no catalog rates`);
12
+ return {
13
+ input: cost.inputPer1M,
14
+ output: cost.outputPer1M,
15
+ ...(cost.reasoningPer1M === undefined ? {} : { reasoning: cost.reasoningPer1M }),
16
+ ...(cost.cacheReadPer1M === undefined ? {} : { cacheRead: cost.cacheReadPer1M }),
17
+ ...(cost.cacheWritePer1M === undefined ? {} : { cacheWrite: cost.cacheWritePer1M }),
18
+ };
19
+ };
4
20
  import type { PluginAttributionContext } from "@plurnk/plurnk-meta";
5
21
 
6
22
  const mapOf = (entries: Record<string, string>, skipped: Record<string, string> = {}) =>
@@ -279,9 +295,11 @@ test("{§deepseek-reasoning-request} #157: direct DeepSeek composes catalog fact
279
295
  totalTokens: 12,
280
296
  inputTokenDetails: { noCacheTokens: 2, cacheReadTokens: 8 },
281
297
  });
298
+ const expected = calculateCostUsdDecimal(response.accounting[0]!.usage!, catalogRatesOf("deepseek", "deepseek-v4-flash"));
299
+ assert.match(String(expected), /^0\.0*[1-9]/, "the catalog prices this usage");
282
300
  assert.deepEqual(response.accounting[0]?.cost, {
283
301
  kind: "estimated",
284
- amount: { amount: "0.0000008624", currency: "USD" },
302
+ amount: { amount: expected, currency: "USD" },
285
303
  source: "Models.dev catalog rates",
286
304
  });
287
305
  mock.restoreAll();
@@ -360,3 +360,59 @@ test("normalizeRetryAttemptError — only provider-directed waits retry: 429, Re
360
360
  });
361
361
  assert.equal((normalizeRetryAttemptError(bareServerError) as APICallError).isRetryable, false, "a bare 5xx surfaces at once for the engine's recovery");
362
362
  });
363
+
364
+ // {§provider-usage-refusal} (#580) — the provider's counters disagree with themselves; the
365
+ // exchange is not the casualty.
366
+ test("inconsistent usage counters refuse normalization without failing the response, in both shapes", async (t) => {
367
+ const usage = { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6, completion_tokens_details: { reasoning_tokens: 7 } };
368
+ await t.test("non-streamed", async () => {
369
+ const result = await executeOpenAICompatible({
370
+ ...request,
371
+ fetch: async () => new Response(JSON.stringify({
372
+ id: "response-1", object: "chat.completion", created: 1, model: "served-model",
373
+ choices: [{ index: 0, message: { role: "assistant", content: "answer" }, finish_reason: "stop" }],
374
+ usage,
375
+ }), { headers: { "content-type": "application/json" } }),
376
+ });
377
+ assert.equal(result.content, "answer", "the model's answer survives the provider's bookkeeping");
378
+ assert.equal(result.usage, undefined, "no counter is invented, clamped, or zeroed");
379
+ assert.deepEqual(result.usageRefusal, {
380
+ reason: "provider usage.outputTokenDetails.textTokens must be a non-negative safe integer",
381
+ usage,
382
+ }, "the counters as reported ride beside the refusal");
383
+ assert.deepEqual(result.chargeEvidence.usage, usage, "charge evidence still carries the wire usage");
384
+ });
385
+ await t.test("streamed", async () => {
386
+ const chunks = [
387
+ { id: "r", object: "chat.completion.chunk", created: 1, model: "served-model", choices: [{ index: 0, delta: { content: "answer" }, finish_reason: null }] },
388
+ { id: "r", object: "chat.completion.chunk", created: 1, model: "served-model", choices: [{ index: 0, delta: {}, finish_reason: "stop" }], usage },
389
+ ];
390
+ const result = await executeOpenAICompatible({
391
+ ...request,
392
+ streaming: true,
393
+ fetch: async () => new Response(new ReadableStream({
394
+ start(controller) {
395
+ for (const chunk of chunks) controller.enqueue(new TextEncoder().encode(`data: ${JSON.stringify(chunk)}\n\n`));
396
+ controller.enqueue(new TextEncoder().encode("data: [DONE]\n\n"));
397
+ controller.close();
398
+ },
399
+ }), { headers: { "content-type": "text/event-stream" } }),
400
+ });
401
+ assert.equal(result.content, "answer");
402
+ assert.equal(result.usage, undefined);
403
+ assert.equal(result.usageRefusal?.reason, "provider usage.outputTokenDetails.textTokens must be a non-negative safe integer");
404
+ assert.deepEqual(result.usageRefusal?.usage, usage);
405
+ });
406
+ await t.test("consistent counters still normalize", async () => {
407
+ const result = await executeOpenAICompatible({
408
+ ...request,
409
+ fetch: async () => new Response(JSON.stringify({
410
+ id: "response-1", object: "chat.completion", created: 1, model: "served-model",
411
+ choices: [{ index: 0, message: { role: "assistant", content: "answer" }, finish_reason: "stop" }],
412
+ usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6, completion_tokens_details: { reasoning_tokens: 2 } },
413
+ }), { headers: { "content-type": "application/json" } }),
414
+ });
415
+ assert.deepEqual(result.usage, { inputTokens: 1, outputTokens: 5, totalTokens: 6, outputTokenDetails: { textTokens: 3, reasoningTokens: 2 } });
416
+ assert.equal(result.usageRefusal, undefined);
417
+ });
418
+ });