@plurnk/plurnk-providers 1.3.12 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (139) hide show
  1. package/.env.defaults +47 -38
  2. package/README.md +68 -4
  3. package/SPEC.md +245 -62
  4. package/dist/AiSdkProvider.d.ts +19 -5
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +264 -156
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +15 -17
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +27 -10
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +8 -2
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +41 -8
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/ProviderRegistry.d.ts +4 -1
  17. package/dist/ProviderRegistry.d.ts.map +1 -1
  18. package/dist/ProviderRegistry.js +7 -3
  19. package/dist/ProviderRegistry.js.map +1 -1
  20. package/dist/accounting.d.ts +3 -0
  21. package/dist/accounting.d.ts.map +1 -0
  22. package/dist/accounting.js +84 -0
  23. package/dist/accounting.js.map +1 -0
  24. package/dist/aiSdkTransport.d.ts +5 -2
  25. package/dist/aiSdkTransport.d.ts.map +1 -1
  26. package/dist/aiSdkTransport.js +91 -5
  27. package/dist/aiSdkTransport.js.map +1 -1
  28. package/dist/catalogProvider.d.ts +5 -2
  29. package/dist/catalogProvider.d.ts.map +1 -1
  30. package/dist/catalogProvider.js +24 -11
  31. package/dist/catalogProvider.js.map +1 -1
  32. package/dist/compatibleProvider.d.ts.map +1 -1
  33. package/dist/compatibleProvider.js +11 -4
  34. package/dist/compatibleProvider.js.map +1 -1
  35. package/dist/cost.d.ts +11 -0
  36. package/dist/cost.d.ts.map +1 -0
  37. package/dist/cost.js +61 -0
  38. package/dist/cost.js.map +1 -0
  39. package/dist/discover.d.ts +2 -0
  40. package/dist/discover.d.ts.map +1 -1
  41. package/dist/discover.js +15 -9
  42. package/dist/discover.js.map +1 -1
  43. package/dist/env.d.ts +4 -6
  44. package/dist/env.d.ts.map +1 -1
  45. package/dist/env.js +27 -29
  46. package/dist/env.js.map +1 -1
  47. package/dist/errors.d.ts +27 -0
  48. package/dist/errors.d.ts.map +1 -0
  49. package/dist/errors.js +152 -0
  50. package/dist/errors.js.map +1 -0
  51. package/dist/index.d.ts +12 -6
  52. package/dist/index.d.ts.map +1 -1
  53. package/dist/index.js +8 -5
  54. package/dist/index.js.map +1 -1
  55. package/dist/notices.d.ts +10 -0
  56. package/dist/notices.d.ts.map +1 -0
  57. package/dist/notices.js +11 -0
  58. package/dist/notices.js.map +1 -0
  59. package/dist/ollama.d.ts.map +1 -1
  60. package/dist/ollama.js +3 -3
  61. package/dist/ollama.js.map +1 -1
  62. package/dist/openai.d.ts +1 -1
  63. package/dist/openai.d.ts.map +1 -1
  64. package/dist/promptTokens.d.ts +4 -0
  65. package/dist/promptTokens.d.ts.map +1 -0
  66. package/dist/promptTokens.js +32 -0
  67. package/dist/promptTokens.js.map +1 -0
  68. package/dist/sdkModels.d.ts +2 -0
  69. package/dist/sdkModels.d.ts.map +1 -1
  70. package/dist/sdkModels.js +17 -6
  71. package/dist/sdkModels.js.map +1 -1
  72. package/dist/types.d.ts +52 -16
  73. package/dist/types.d.ts.map +1 -1
  74. package/dist/types.js +1 -1
  75. package/dist/types.js.map +1 -1
  76. package/dist/usage.d.ts +4 -0
  77. package/dist/usage.d.ts.map +1 -1
  78. package/dist/usage.js +64 -19
  79. package/dist/usage.js.map +1 -1
  80. package/dist/warnings.js +0 -0
  81. package/dist/warnings.js.map +1 -1
  82. package/package.json +15 -10
  83. package/src/AiSdkProvider.test.ts +750 -169
  84. package/src/AiSdkProvider.ts +354 -200
  85. package/src/Mock.test.ts +29 -14
  86. package/src/Mock.ts +36 -15
  87. package/src/Pool.test.ts +43 -6
  88. package/src/Pool.ts +56 -10
  89. package/src/ProviderRegistry.test.ts +158 -9
  90. package/src/ProviderRegistry.ts +19 -6
  91. package/src/accounting.test.ts +58 -0
  92. package/src/accounting.ts +88 -0
  93. package/src/aiSdkTransport.ts +101 -8
  94. package/src/boundaries.test.ts +9 -3
  95. package/src/catalogProvider.test.ts +43 -15
  96. package/src/catalogProvider.ts +32 -16
  97. package/src/compatibleProvider.test.ts +96 -0
  98. package/src/compatibleProvider.ts +15 -6
  99. package/src/cost.test.ts +64 -0
  100. package/src/cost.ts +78 -0
  101. package/src/defaults.test.ts +1 -0
  102. package/src/discover.test.ts +48 -7
  103. package/src/discover.ts +31 -21
  104. package/src/env.test.ts +38 -48
  105. package/src/env.ts +43 -40
  106. package/src/errors.test.ts +148 -0
  107. package/src/errors.ts +208 -0
  108. package/src/index.ts +30 -8
  109. package/src/lexicon-guard.test.ts +6 -6
  110. package/src/notices.ts +22 -0
  111. package/src/ollama.test.ts +64 -0
  112. package/src/ollama.ts +6 -3
  113. package/src/openai.ts +3 -0
  114. package/src/promptTokens.ts +41 -0
  115. package/src/sdkModels.test.ts +29 -3
  116. package/src/sdkModels.ts +19 -11
  117. package/src/types.ts +125 -64
  118. package/src/usage.test.ts +24 -5
  119. package/src/usage.ts +72 -21
  120. package/src/warnings.test.ts +10 -10
  121. package/src/warnings.ts +0 -0
  122. package/dist/OpenAICompat.d.ts +0 -76
  123. package/dist/OpenAICompat.d.ts.map +0 -1
  124. package/dist/OpenAICompat.js +0 -555
  125. package/dist/OpenAICompat.js.map +0 -1
  126. package/dist/openaiStream.d.ts +0 -47
  127. package/dist/openaiStream.d.ts.map +0 -1
  128. package/dist/openaiStream.js +0 -280
  129. package/dist/openaiStream.js.map +0 -1
  130. package/dist/standardProviders.d.ts +0 -31
  131. package/dist/standardProviders.d.ts.map +0 -1
  132. package/dist/standardProviders.js +0 -518
  133. package/dist/standardProviders.js.map +0 -1
  134. package/dist/telemetry.d.ts +0 -24
  135. package/dist/telemetry.d.ts.map +0 -1
  136. package/dist/telemetry.js +0 -85
  137. package/dist/telemetry.js.map +0 -1
  138. package/src/telemetry.test.ts +0 -69
  139. package/src/telemetry.ts +0 -116
@@ -6,10 +6,13 @@
6
6
  // ordinary vendor protocol. The compatible URL path remains only for PLURNK
7
7
  // extensions and local endpoint probes the SDK cannot represent.
8
8
  import { executeAiSdkModel, executeOpenAICompatible } from "./aiSdkTransport.js";
9
- import { toProviderError, ProviderError } from "./telemetry.js";
9
+ import { toProviderError, ProviderError } from "./errors.js";
10
+ import { attributeUnitemizedReasoning } from "./usage.js";
10
11
  import { validateGbnf } from "@plurnk/gbnf";
12
+ import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.js";
11
13
  import { emitWarningOnce } from "./warnings.js";
12
- // #539: drop trailing occurrences of a server-rendered EOG marker. llama-server
14
+ import { validateAuthoritativeCharge } from "./cost.js";
15
+ // Drop trailing occurrences of a server-rendered EOG marker. llama-server
13
16
  // under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
14
17
  // trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
15
18
  // can never eat body content (a body ending in the literal marker isn't producible
@@ -22,6 +25,38 @@ const stripTrailingSpecial = (content, marker) => {
22
25
  out = out.slice(0, -marker.length);
23
26
  return out;
24
27
  };
28
+ const projectLeadingReasoning = (content, structuredReasoning, opening, closing) => {
29
+ if (structuredReasoning.length > 0 || !content.startsWith(opening)) {
30
+ return { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
31
+ }
32
+ const closingIndex = content.indexOf(closing, opening.length);
33
+ if (closingIndex === -1) {
34
+ return {
35
+ content: "",
36
+ reasoning: content.slice(opening.length),
37
+ projected: true,
38
+ contentStart: [...content].length,
39
+ };
40
+ }
41
+ const suffixStart = closingIndex + closing.length;
42
+ return {
43
+ content: content.slice(suffixStart),
44
+ reasoning: content.slice(opening.length, closingIndex),
45
+ projected: true,
46
+ contentStart: [...content.slice(0, suffixStart)].length,
47
+ };
48
+ };
49
+ // {§provider-tagged-reasoning} Only the model-contract position is structural:
50
+ // one exact leading envelope. Parsing after stream assembly keeps SSE and JSON
51
+ // on one path and leaves later literal tags in the visible suffix untouched.
52
+ const projectTaggedReasoning = (content, structuredReasoning, style) => style === "think-tags"
53
+ ? projectLeadingReasoning(content, structuredReasoning, "<think>", "</think>")
54
+ : { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
55
+ // llama-server's template reasoning parser can project this leading channel out
56
+ // of the OpenAI-compatible response. Grammar evidence needs the sentence before
57
+ // that lossy projection, so constrained template turns request it verbatim and
58
+ // split the observed enclosure here.
59
+ const projectTemplateReasoning = (content) => projectLeadingReasoning(content, "", "<|channel>thought\n", "<channel|>");
25
60
  // Shared budget→effort breakpoints (xai and google had identical copies).
26
61
  export const effortFromBudget = (budget) => {
27
62
  if (budget <= 1000)
@@ -30,41 +65,27 @@ export const effortFromBudget = (budget) => {
30
65
  return "medium";
31
66
  return "high";
32
67
  };
33
- // chars/2 upper bound (see ./tokenizers.ts) — overcounts safely, never under.
34
- const heuristicTokens = (text) => (text.length === 0 ? 0 : Math.ceil(text.length / 2));
35
68
  // Body keys the provider owns — a caller's `sampling` passthrough may not set
36
- // these. Two families (#477 audit):
69
+ // these. Two families:
37
70
  // transport/managed — grammar transport, the stream/JSON choice, slot pinning,
38
- // data capture (SPEC §8: backend-specific fields never cross the contract);
71
+ // data capture ({§provider-evidence}: backend-specific fields never cross the contract);
39
72
  // contract invariants — `n` (atomic single completion: choices[0] is the
40
73
  // response; n>1 = paid, dropped output), the tool-calling family (tools-in-
41
74
  // body doctrine, §2: native tool_calls return null content = a broken turn),
42
75
  // modalities/audio (text-only contract), prediction (decode semantics, not
43
76
  // sampling), and the token caps (the envelope is the managed maxTokens —
44
- // sampling must not bypass the consumer's #425 cap).
77
+ // sampling must not bypass the consumer's cap).
45
78
  // Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
46
79
  // platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
47
80
  // metadata, store, verbosity) pass through; the managed floors spread UNDER
48
81
  // sampling stay deliberately caller-overridable.
49
82
  const RESERVED_BODY_KEYS = new Set([
50
83
  "model", "messages", "stream", "stream_options", "grammar", "response_format", "id_slot", "logprobs", "top_logprobs",
84
+ "reasoning_format", "reasoning_effort", "thinking", "think", "include_reasoning", "chat_template_kwargs", "thinking_budget_tokens", // lexicon-allow: backend wire fields
51
85
  "n", "tools", "tool_choice", "functions", "function_call", "parallel_tool_calls",
52
86
  "modalities", "audio", "prediction", "max_tokens", "max_completion_tokens",
53
87
  "prompt_cache_key",
54
88
  ]);
55
- // Render a non-accept verdict into a terse, factual grammar_unenforced message
56
- // (SPEC §12 message policy: no guidance prose). `reject` names the diverging code
57
- // point + what the grammar would have accepted; `incomplete` names the valid-prefix
58
- // length that never reached a terminal state.
59
- const describeUnenforced = (v) => {
60
- if (v.status === "reject") {
61
- const expected = v.expected.length > 0
62
- ? v.expected.map((e) => `${e.rule} accepts ${e.accepts}`).join(", ")
63
- : "end of input";
64
- return `grammar not enforced: output rejected by the transported grammar at code point ${v.pos} (${JSON.stringify(v.char)}); expected ${expected}`;
65
- }
66
- return `grammar not enforced: output is an incomplete match of the transported grammar — a valid prefix of ${v.pos} code points that never terminated`;
67
- };
68
89
  export default class AiSdkProvider {
69
90
  #model;
70
91
  #url;
@@ -86,8 +107,12 @@ export default class AiSdkProvider {
86
107
  #dryAllowedLength;
87
108
  #repeatLastN;
88
109
  #reasoningStyle;
89
- #countTokens;
110
+ #reasoningResponseStyle;
111
+ #countPromptTokens;
112
+ #promptTokensUrl;
90
113
  #calculateCost;
114
+ #calculateCharge;
115
+ #normalizeCharge;
91
116
  #source;
92
117
  #grammarStyle;
93
118
  #promptCacheKey;
@@ -98,6 +123,7 @@ export default class AiSdkProvider {
98
123
  #supportsSlotPinning;
99
124
  #slotCount;
100
125
  #retryAttempts;
126
+ #errorDetailLimit;
101
127
  #topLogprobs;
102
128
  #reasoningReserve;
103
129
  #completionReserve;
@@ -105,7 +131,8 @@ export default class AiSdkProvider {
105
131
  #rawBody;
106
132
  #servedModel;
107
133
  #requiresMaxTokens;
108
- // Optional capability (SPEC §2): exact tokenization served by the backend's
134
+ attributions;
135
+ // Optional capability ({§provider-local-capabilities}): exact tokenization served by the backend's
109
136
  // own vocab. Assigned in the constructor ONLY when the config carries a
110
137
  // tokenizeUrl (llama-server), so `provider.tokenize === undefined` remains
111
138
  // the honest capability signal for every other backend.
@@ -114,6 +141,7 @@ export default class AiSdkProvider {
114
141
  this.#model = config.model;
115
142
  this.#url = config.url;
116
143
  this.#languageModel = config.languageModel;
144
+ this.attributions = config.attributions;
117
145
  if ((this.#url === undefined) === (this.#languageModel === undefined)) {
118
146
  throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
119
147
  }
@@ -137,9 +165,17 @@ export default class AiSdkProvider {
137
165
  this.#dryAllowedLength = config.dryAllowedLength;
138
166
  this.#repeatLastN = config.repeatLastN;
139
167
  this.#retryAttempts = config.retryAttempts;
168
+ this.#errorDetailLimit = config.errorDetailLimit;
140
169
  this.#reasoningStyle = config.reasoningStyle ?? "none";
141
- this.#countTokens = config.countTokens ?? heuristicTokens;
170
+ this.#reasoningResponseStyle = config.reasoningResponseStyle ?? "verbatim";
171
+ if (config.countPromptTokens !== undefined && config.promptTokensUrl !== undefined) {
172
+ throw new Error(`${config.source ?? "provider"}: configure countPromptTokens or promptTokensUrl, not both`);
173
+ }
174
+ this.#countPromptTokens = config.countPromptTokens ?? ((messages) => estimatePromptTokens(messages));
175
+ this.#promptTokensUrl = config.promptTokensUrl;
142
176
  this.#calculateCost = config.calculateCost ?? (() => 0);
177
+ this.#calculateCharge = config.calculateCharge;
178
+ this.#normalizeCharge = config.normalizeCharge;
143
179
  this.#source = config.source ?? "provider";
144
180
  this.#grammarStyle = config.grammarStyle ?? "none";
145
181
  this.#promptCacheKey = config.promptCacheKey ?? false;
@@ -159,6 +195,13 @@ export default class AiSdkProvider {
159
195
  this.#rawBody = config.rawBody ?? false;
160
196
  this.#servedModel = config.servedModel;
161
197
  this.#requiresMaxTokens = config.requiresMaxTokens;
198
+ const reasoningReserve = this.reasoningReserve;
199
+ if (this.#reasoningStyle === "template"
200
+ && this.#reasoning.mode === "on"
201
+ && reasoningReserve !== null
202
+ && this.#reasoning.budget > reasoningReserve) {
203
+ throw new Error(`${this.#source}: PLURNK_PROVIDERS_REASONING_BUDGET (${this.#reasoning.budget}) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE (${reasoningReserve})`);
204
+ }
162
205
  const { tokenizeUrl } = config;
163
206
  if (tokenizeUrl !== undefined) {
164
207
  this.tokenize = async (text) => {
@@ -179,7 +222,7 @@ export default class AiSdkProvider {
179
222
  }
180
223
  }
181
224
  get contextWindow() { return this.#contextWindow; }
182
- // #507: envelope reserves — absolute pins stand alone; percentages need the
225
+ // {§provider-generation-envelope} Envelope reserves — absolute pins stand alone; percentages need the
183
226
  // detected window; null = underivable (no claim for core's no-cap path).
184
227
  #resolveReserve(spec) {
185
228
  if (spec === undefined)
@@ -191,61 +234,93 @@ export default class AiSdkProvider {
191
234
  get reasoningReserve() { return this.#resolveReserve(this.#reasoningReserve); }
192
235
  get completionReserve() { return this.#resolveReserve(this.#completionReserve); }
193
236
  get model() { return this.#model; }
194
- // #37: backend's self-reported served id; undefined when unprobed/unknown.
237
+ // Backend's self-reported served id; undefined when unprobed/unknown.
195
238
  get servedModel() { return this.#servedModel; }
196
- // #43: resolved "decodes unbounded without a cap" fact; undefined = no claim.
239
+ // Resolved "decodes unbounded without a cap" fact; undefined = no claim.
197
240
  get requiresMaxTokens() { return this.#requiresMaxTokens; }
198
- // Resolved capability (#34): will a transported grammar actually constrain
241
+ // Resolved capability: will a transported grammar actually constrain
199
242
  // this backend's decode? Introspectable so a consumer can verify the rails
200
243
  // are LIVE without spending a generation on a forcing-grammar probe.
201
244
  get constrainsOutput() { return this.#grammarStyle !== "none"; }
202
- countTokens(text) { return this.#countTokens(text); }
245
+ async countPromptTokens(messages, signal) {
246
+ if (this.#promptTokensUrl === undefined) {
247
+ return assertPromptTokenMeasurement(await this.#countPromptTokens(messages, signal), this.#source);
248
+ }
249
+ signal?.throwIfAborted();
250
+ try {
251
+ const timeout = AbortSignal.timeout(this.#fetchTimeoutMs);
252
+ const response = await this.#fetch(this.#promptTokensUrl, {
253
+ method: "POST",
254
+ headers: { "Content-Type": "application/json", ...this.#headers },
255
+ body: JSON.stringify({
256
+ model: this.#model,
257
+ messages,
258
+ ...this.#reasoningBody(),
259
+ }),
260
+ signal: signal === undefined ? timeout : AbortSignal.any([signal, timeout]),
261
+ });
262
+ if (!response.ok) {
263
+ return estimatePromptTokens(messages, `llama-server input-token endpoint returned HTTP ${response.status}`);
264
+ }
265
+ const body = await response.json();
266
+ if (!Number.isInteger(body.input_tokens) || body.input_tokens < 0) {
267
+ return estimatePromptTokens(messages, "llama-server input-token endpoint returned no non-negative integer input_tokens");
268
+ }
269
+ return {
270
+ kind: "exact",
271
+ tokens: body.input_tokens,
272
+ source: "llama-server:/v1/chat/completions/input_tokens",
273
+ };
274
+ }
275
+ catch (cause) {
276
+ signal?.throwIfAborted();
277
+ return estimatePromptTokens(messages, `llama-server input-token measurement failed: ${cause instanceof Error ? cause.message : String(cause)}`);
278
+ }
279
+ }
203
280
  calculateCost(usage) { return this.#calculateCost(usage); }
204
- // Maps the reasoning INTENT (off | adaptive | on+budget, #33) to the
205
- // backend's wire mechanism — including under a transported grammar. The #32
206
- // clamp (force reasoning_effort "none" under response_format) is LIFTED:
207
- // canary-verified live that fireworks masks ONLY the content channel — the
208
- // reasoning channel rides beside it unmasked, and the plurnk grammar's
209
- // reasoning?/preplan regions absorb any in-band spillover. The old measured
210
- // failures (low→cycles, high→spirals) were pre-max_tokens-cap; with a bounded
211
- // cap the matrix ACCEPTs across efforts (reasoning-rails matrix
212
- // F9). Clamping was the root of the plan-less regression (service#331).
213
- #reasoningBody() {
281
+ calculateCharge(usage) {
282
+ return this.#calculateCharge?.(usage)
283
+ ?? { kind: "unknown", reason: "no provider rate or settled charge is available" };
284
+ }
285
+ // Reasoning activation and allowance are independent of grammar transport;
286
+ // only the response representation becomes lossless when evidence is needed.
287
+ // The llama-server template mapping is owned by {§llama-reasoning-request}.
288
+ #reasoningBody(preserveGrammarSentence = false) {
214
289
  const { mode, budget } = this.#reasoning;
215
290
  const on = mode !== "off";
216
291
  switch (this.#reasoningStyle) {
217
- // Native-channel styles. "template" ALWAYS emits — the explicit
218
- // enable_thinking:false is the only working off-switch on llama-server
219
- // (§13). Activation only; budget is enforced by the box's
220
- // --reasoning-budget launch flag (per-request numerics ignored, F7).
221
- //
222
- // #488 postmortem: intent maps IDENTICALLY under a transported
223
- // grammar. The brief rails-win-the-channel clamp (enable_thinking
224
- // forced false under a grammar) is REVERTED — specimens proved the
225
- // SANCTIONED think block is the protection, not the hazard: the
226
- // server auto-gates the grammar around it and content decodes
227
- // constrained (26-run baseline green; zero grammar rejects across
228
- // the #488 "railless" specimens). Closing the channel starved a
229
- // reasoning-tuned model into ESCAPING mid-content into the raw
230
- // thought channel — discarded server-side, decode unconstrained,
231
- // 12,288 tokens billed for 1,033 visible chars. The escape is
232
- // surfaced instead (vanished-token telemetry + meta rail state).
233
- case "template": return { chat_template_kwargs: { enable_thinking: on } };
292
+ case "template": {
293
+ const allowance = mode === "off"
294
+ ? 0
295
+ : mode === "on" ? budget : this.reasoningReserve;
296
+ return {
297
+ chat_template_kwargs: { enable_thinking: on },
298
+ reasoning_format: preserveGrammarSentence ? "none" : "auto",
299
+ ...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
300
+ };
301
+ }
234
302
  case "think": return on ? { think: true } : {};
235
303
  case "include_reasoning": return on ? { include_reasoning: true } : {};
236
304
  // effort tiers from the budget; off/adaptive omit the field (the
237
305
  // API's default depth is its adaptive).
238
306
  case "effort": return mode === "on" ? { reasoning_effort: effortFromBudget(budget) } : {};
239
307
  // Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
240
- // reason-by-default model (DeepSeek V4: default 'high') reasoning (#30).
308
+ // reason-by-default model (DeepSeek V4: default 'high') reasoning.
241
309
  // ADAPTIVE omits the field: the backend's own default posture IS the
242
310
  // adaptive semantics, and the literal "adaptive" is MiniMax-M3-only —
243
- // fireworks 400s it for every other model (wire-verified, #403; the
311
+ // Fireworks 400s it for every other model (wire-verified; the
244
312
  // 1.0.2 adaptive default refused to boot on it). V4 gotcha: integer
245
313
  // efforts 400.
246
314
  case "effort_explicit": return mode === "off"
247
315
  ? { reasoning_effort: "none" }
248
316
  : mode === "on" ? { reasoning_effort: effortFromBudget(budget) } : {};
317
+ // {§deepseek-reasoning-request}
318
+ case "thinking_effort": return mode === "off"
319
+ ? { thinking: { type: "disabled" } }
320
+ : mode === "on" ? {
321
+ thinking: { type: "enabled" },
322
+ reasoning_effort: effortFromBudget(budget),
323
+ } : {};
249
324
  // Anthropic compat: explicit thinking object. off → disabled; on →
250
325
  // enabled with budget_tokens; adaptive → omit (the API default).
251
326
  case "anthropic": return mode === "off"
@@ -254,7 +329,7 @@ export default class AiSdkProvider {
254
329
  case "none": return {};
255
330
  }
256
331
  }
257
- // Per-run slot affinity (#11): the consumer passes WHICH run this is; the
332
+ // Per-worker slot affinity: the consumer passes which worker this is; the
258
333
  // provider owns WHICH slot serves it. Sticky per workerId, round-robin across
259
334
  // new runs (distinct runs → distinct slots while slots last), LRU-bounded
260
335
  // bookkeeping so a long-lived daemon never grows the map unboundedly —
@@ -277,19 +352,19 @@ export default class AiSdkProvider {
277
352
  this.#runSlots.set(workerId, slot);
278
353
  return { id_slot: slot };
279
354
  }
280
- // Optional local llama-server GBNF transport (SPEC §13). Unsupported
355
+ // Optional local llama-server GBNF transport ({§gbnf-response-observation}). Unsupported
281
356
  // backends receive no grammar-related field.
282
357
  #grammarBody(grammar) {
283
358
  if (grammar === undefined)
284
359
  return {};
285
360
  switch (this.#grammarStyle) {
286
361
  // Greedy decoding under hard constraint loops without a repeat-penalty
287
- // floor (#9, SPEC §13) — llama.cpp spells it `repeat_penalty`.
362
+ // floor — llama.cpp spells it `repeat_penalty`.
288
363
  case "llamacpp": return { grammar, repeat_penalty: this.#repeatPenalty };
289
364
  case "none": return {};
290
365
  }
291
366
  }
292
- // Anti-degeneration DEFAULT on EVERY request (#426), keyed to the backend's wire
367
+ // Anti-degeneration default on every request, keyed to the backend's wire
293
368
  // convention - NOT grammar-bound. GBNF is a local constraint, so a cloud
294
369
  // alias runs the sampler bare: firefast (deepseek/fireworks) ran 4/86 bench turns
295
370
  // straight to the token cap on pure looped repetition (run52). Ships next to
@@ -300,7 +375,7 @@ export default class AiSdkProvider {
300
375
  // live: together/deepinfra/fireworks; it is OpenAI's own param, so real OpenAI takes it too).
301
376
  #repetitionPenaltyBody() {
302
377
  switch (this.#grammarStyle) {
303
- // #567: repeat_penalty + optional DRY (repeated-SEQUENCE penalty) + a wider
378
+ // repeat_penalty + optional DRY (repeated-sequence penalty) + a wider
304
379
  // repeat_last_n window — the loop-breaking tools a llama.cpp backend serves.
305
380
  // Each rides only when its operator knob is set; absent = the box's default.
306
381
  case "llamacpp": return {
@@ -315,12 +390,12 @@ export default class AiSdkProvider {
315
390
  case "none": return this.#frequencyPenalty > 0 ? { frequency_penalty: this.#frequencyPenalty } : {};
316
391
  }
317
392
  }
318
- // First-party telemetry headers (SPEC §5): forwarded ONLY when the spec
393
+ // First-party telemetry headers ({§provider-request-authority}): forwarded only when the spec
319
394
  // opted in (the plurnk endpoint). The gate is here, not at the call site, so
320
395
  // attributions/client/strikes can never reach a third-party backend even if
321
396
  // the consumer passes them to the wrong provider. Empty values emit no header
322
397
  // — EXCEPT strikes, where 0 is a real value (clean streak) distinct from
323
- // absent (consumer didn't report); contract per plurnk-service#313. Strikes
398
+ // absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
324
399
  // ride HTTP headers only — the packet never carries them (the model must
325
400
  // never see strike state; engine accounting is not a metric to game).
326
401
  #metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn) {
@@ -333,11 +408,11 @@ export default class AiSdkProvider {
333
408
  h["Plurnk-Client"] = client;
334
409
  if (strikes !== undefined && Number.isInteger(strikes) && strikes >= 0)
335
410
  h["Plurnk-Strikes"] = String(strikes);
336
- // Worker identity (#26, wire-name completed #486/#511): the opaque workerId
411
+ // Worker identity: the opaque workerId
337
412
  // the consumer already supplies, forwarded so the endpoint can key
338
413
  // per-worker affinity/telemetry — same gate as every first-party signal.
339
414
  h["Plurnk-Worker-Id"] = workerId;
340
- // Root worker of the lineage (#522): the no-parent ancestor of this turn's
415
+ // Root worker of the lineage ({§worker-primary}): the no-parent ancestor of this turn's
341
416
  // worker tree. The consumer classifies primary-vs-spawned by equality
342
417
  // (primaryWorkerId == workerId ⇒ the primary/root worker). The provider
343
418
  // EMITS what the consumer supplies and never invents a primary; the
@@ -346,7 +421,7 @@ export default class AiSdkProvider {
346
421
  // the endpoint to surface, not a provider default.
347
422
  if (primaryWorkerId !== undefined && primaryWorkerId.length > 0)
348
423
  h["Plurnk-Worker-Primary"] = primaryWorkerId;
349
- // Turn coordinate (#404, extends #26 per #391): workspace/loop/turn, the
424
+ // Turn coordinate ({§lifecycle-terms}): workspace/loop/turn, the
350
425
  // daemon-side sequence the endpoint can never scrape from the wire.
351
426
  // Coordinates are 1-based — 0 is not a real value, so no strikes-style
352
427
  // zero exception; absent/empty/0 emits no header.
@@ -358,30 +433,7 @@ export default class AiSdkProvider {
358
433
  h["Plurnk-Turn"] = String(turn);
359
434
  return h;
360
435
  }
361
- // Enforcement verification (SPEC §13). When a grammar was actually transported
362
- // (grammarStyle !== "none"), the backend MUST have constrained the output;
363
- // some silently drop the grammar field or mislabel the channel, and without
364
- // this check we would return unconstrained output as if enforced. STRICT: any
365
- // non-accept verdict (reject, or an incomplete/never-terminated match) is a
366
- // grammar_unenforced failure. A grammar our own validator can't parse — even
367
- // though the backend accepted it (a port-vs-llama.cpp gap) — is a non-fatal
368
- // verify gap: warn, don't fail a transport that may have worked. This is a
369
- // conformance check against the grammar we already hold, NOT a plurnk-DSL
370
- // parse (§8) — it stays grammar-generic and backend-agnostic.
371
- // Validate output against the grammar. Returns the verdict, or null on the
372
- // verify GAP — a grammar our own validator can't parse (a port-vs-llama.cpp
373
- // gap): warn, don't manufacture a conflict from a check that didn't run.
374
- #grammarVerdict(grammar, content) {
375
- try {
376
- return validateGbnf(grammar, content);
377
- }
378
- catch (cause) {
379
- // Once per (code, message) — #40: this fires PER TURN otherwise.
380
- emitWarningOnce(`${this.#source}: could not verify grammar enforcement — the transported grammar did not parse in @plurnk/gbnf (${cause.message})`, "PLURNK_GRAMMAR_UNVERIFIABLE");
381
- return null;
382
- }
383
- }
384
- // PLURNK_PROVIDERS_GBNF_DEBUG (SPEC §13): validate the supplied GBNF locally and fail
436
+ // PLURNK_PROVIDERS_GBNF_DEBUG ({§gbnf-response-observation}): validate the supplied GBNF locally and fail
385
437
  // hard if it's malformed, BEFORE any wire call — and the grammar is NOT
386
438
  // transported, so the request runs unconstrained. A debug aid to catch invalid
387
439
  // grammars (e.g. while editing the plurnk grammar) without a model round-trip;
@@ -396,7 +448,7 @@ export default class AiSdkProvider {
396
448
  throw new Error(`grammar validation (PLURNK_PROVIDERS_GBNF_DEBUG): invalid GBNF — ${cause.message}`, { cause });
397
449
  }
398
450
  }
399
- // Per-turn metadata bag (#23): pass the backend's non-standard top-level fields
451
+ // Per-turn metadata bag: pass the backend's non-standard top-level fields
400
452
  // through verbatim. Providers do not reinterpret vendor currency or account
401
453
  // metadata; a monetary value carries its own amount and currency.
402
454
  #buildMeta(chunkMetadata) {
@@ -407,7 +459,8 @@ export default class AiSdkProvider {
407
459
  // penalties, stop, seed, …) merged UNDER the managed body: model, messages,
408
460
  // reasoning, grammar (+ its repeat-penalty floor), max_tokens and slot always
409
461
  // win, and reserved transport/protocol keys are stripped so the passthrough
410
- // can't smuggle a grammar, a stream toggle, or a backend slot (SPEC §8).
462
+ // can't smuggle a grammar, a stream toggle, or a backend slot
463
+ // ({§provider-request-authority}).
411
464
  #samplingBody(sampling) {
412
465
  if (sampling === undefined)
413
466
  return {};
@@ -418,48 +471,49 @@ export default class AiSdkProvider {
418
471
  return out;
419
472
  }
420
473
  async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }) {
421
- // Boundary validation (SPEC §2): the worker identity is required.
474
+ // {§provider-interface} The worker identity is required.
422
475
  if (workerId === undefined || workerId.length === 0)
423
476
  throw new Error("generate: workerId is required — the worker's stable, opaque identity");
424
- // Reject before any wire call when already aborted (SPEC §10.8).
477
+ // Reject before any wire call when already aborted
478
+ // ({§provider-failure-normalization}).
425
479
  signal?.throwIfAborted();
426
- // Grammar handling (SPEC §13). PLURNK_PROVIDERS_GBNF_DEBUG validates the supplied
427
- // grammar locally and throws on a malformed one, then WITHHOLDS it so the
428
- // model generates UNCONSTRAINED — and the free output is still verified
429
- // against the grammar (below), surfacing exactly where the model's natural
430
- // output and the grammar conflict. Otherwise the grammar is sent when the
431
- // backend supports it (grammarStyle !== "none").
480
+ // Grammar handling ({§gbnf-response-observation}). Debug validates the
481
+ // supplied grammar before the call but withholds it from the backend.
432
482
  const wantGrammar = grammar !== undefined && this.#grammarStyle !== "none";
433
483
  if (wantGrammar && this.#gbnfDebug)
434
484
  this.#assertGrammarValid(grammar);
435
485
  const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
486
+ const preserveGrammarSentence = wantGrammar
487
+ && this.#reasoningStyle === "template";
436
488
  // Assembly order = precedence: the family's sampling DEFAULTS
437
- // (PLURNK_PROVIDERS_TEMPERATURE — universal, #30 measured it on grammar
489
+ // (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
438
490
  // paths and the name promises every request) < the caller's `sampling`
439
491
  // < the managed fields, which always win.
440
492
  const body = {
441
- // #507: floors suppressed on router-owned-tuning providers (plurnk) —
493
+ // Floors are suppressed on router-owned-tuning providers (plurnk) —
442
494
  // the router's per-model tuning must not be overridden by client floors.
443
495
  ...(this.#tuningFloors ? { temperature: this.#temperature, ...this.#repetitionPenaltyBody() } : {}),
444
496
  ...this.#samplingBody(sampling),
445
497
  ...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
446
498
  model: this.#model,
447
499
  messages,
448
- ...this.#reasoningBody(),
500
+ ...this.#reasoningBody(preserveGrammarSentence),
449
501
  ...this.#grammarBody(sendGrammar),
450
502
  ...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
451
- // #36: request per-token logprobs only when enabled (managed field —
503
+ // Request per-token logprobs only when enabled (managed field —
452
504
  // reserved from caller sampling; the env flag is the single control).
453
505
  ...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
454
506
  ...this.#slotBody(workerId),
455
- // #518: prompt-cache affinity -- workerId as the OpenAI-standard
507
+ // Prompt-cache affinity -- workerId as the OpenAI-standard
456
508
  // prompt_cache_key routes a worker's turns to one serverless replica so
457
509
  // its stable prefix caches (managed; reserved from caller sampling).
458
510
  ...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
459
511
  };
460
512
  // Per-request headers = static auth/routing + any first-party telemetry.
461
513
  const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
462
- const headers = Object.keys(metaHeaders).length > 0 ? { ...this.#headers, ...metaHeaders } : this.#headers;
514
+ const headers = Object.keys(metaHeaders).length === 0
515
+ ? this.#headers
516
+ : { ...this.#headers, ...metaHeaders };
463
517
  let raw;
464
518
  try {
465
519
  raw = this.#languageModel === undefined
@@ -513,13 +567,13 @@ export default class AiSdkProvider {
513
567
  catch (err) {
514
568
  if (signal?.aborted)
515
569
  throw err;
516
- const pe = toProviderError(err, this.#source);
570
+ const pe = toProviderError(err, this.#source, this.#errorDetailLimit);
517
571
  if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
518
572
  throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
519
573
  }
520
574
  throw pe;
521
575
  }
522
- // #539: llama-server --special renders EOG tokens as text, so a turn ending
576
+ // llama-server --special renders EOG tokens as text, so a turn ending
523
577
  // via raw EOS carries a trailing <eos> the grammar never sanctioned - it both
524
578
  // false-rejects the rail verdict and leaks a control token into the packet.
525
579
  // Strip the server-reported eos_token from the tail ONCE, before the verdict
@@ -527,64 +581,118 @@ export default class AiSdkProvider {
527
581
  // wire text for forensics.
528
582
  if (this.#eosText !== undefined)
529
583
  raw.content = stripTrailingSpecial(raw.content, this.#eosText);
530
- // Grammar conformance (§13): bytes always flow; the verdict is an
531
- // observation. Same check whether the grammar was transported
532
- // (sendGrammar) or withheld (PLURNK_PROVIDERS_GBNF_DEBUG filter mode) — a
533
- // non-accept verdict attaches a grammar_unenforced telemetry event
534
- // (message + divergence position) and the response returns normally.
535
- // Discard/retry/escalate/self-correct is the consumer's policy.
536
- let telemetry;
537
- let railsMeta;
538
- const usage = raw.usage;
539
- const observedGrammar = sendGrammar ?? (wantGrammar && this.#gbnfDebug ? grammar : undefined);
540
- if (observedGrammar !== undefined) {
541
- const verdict = this.#grammarVerdict(observedGrammar, raw.content);
542
- if (verdict !== null && verdict.status !== "accept") {
543
- telemetry = [{ source: this.#source, kind: "grammar_unenforced", message: describeUnenforced(verdict), position: verdict.pos }];
584
+ const grammarInput = raw.content;
585
+ const projectedReasoning = preserveGrammarSentence && !raw.reasoningProjected
586
+ ? projectTemplateReasoning(raw.content)
587
+ : projectTaggedReasoning(raw.content, raw.reasoning, this.#reasoningResponseStyle);
588
+ // Preserve the exact sentence seen at the grammar boundary. Constrained
589
+ // template turns request `reasoning_format: "none"`, so even an empty
590
+ // channel remains observable. An unexpectedly projected response cannot
591
+ // supply independent pre-projection evidence.
592
+ let grammarEvidence;
593
+ if (wantGrammar) {
594
+ if (preserveGrammarSentence) {
595
+ if (!raw.reasoningProjected) {
596
+ grammarEvidence = {
597
+ input: grammarInput,
598
+ contentStart: projectedReasoning.projected ? projectedReasoning.contentStart : 0,
599
+ transported: sendGrammar !== undefined,
600
+ };
601
+ }
544
602
  }
545
- // #488 per-request loud state: rail attachment + conformance verdict
546
- // ride `meta` into the consumer's turn row, so a drill reads rail
547
- // presence PER TURN from the run db instead of inferring it from
548
- // output shape (the #488 misdiagnosis, twice).
549
- railsMeta = { railsAttached: sendGrammar !== undefined, railsVerdict: verdict?.status ?? "unverifiable" };
550
- // #488 channel-escape detector (the run105 class): completion tokens
603
+ else if (projectedReasoning.projected) {
604
+ grammarEvidence = {
605
+ input: grammarInput,
606
+ contentStart: projectedReasoning.contentStart,
607
+ transported: sendGrammar !== undefined,
608
+ };
609
+ }
610
+ else {
611
+ grammarEvidence = {
612
+ input: grammarInput,
613
+ contentStart: 0,
614
+ transported: sendGrammar !== undefined,
615
+ };
616
+ }
617
+ }
618
+ if (projectedReasoning.projected) {
619
+ raw.content = projectedReasoning.content;
620
+ raw.reasoning = projectedReasoning.reasoning;
621
+ raw.usage = attributeUnitemizedReasoning(raw.usage, raw.reasoning, raw.content);
622
+ }
623
+ let notices;
624
+ const usage = raw.usage;
625
+ if (sendGrammar !== undefined && this.tokenize !== undefined) {
626
+ // Channel-escape detector: completion tokens
551
627
  // billed far beyond every visible channel mean the decode ESCAPED into
552
- // a server-discarded reasoning block mid-emission — unconstrained,
553
- // invisible, billed (12,288 billed vs 1,033 chars visible, live).
554
- // countTokens OVERCOUNTS text (chars/2 upper bound), so billed
555
- // exceeding visible-plus-slack is real vanishing, not estimator noise.
556
- const visible = this.#countTokens(raw.content) + this.#countTokens(raw.reasoning);
557
- if (sendGrammar !== undefined && usage.completion > visible + 64) {
558
- (telemetry ??= []).push({
559
- source: this.#source,
560
- kind: "grammar_unenforced",
561
- message: `decode escaped the grammar: ${usage.completion} completion tokens billed but only ~${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
562
- position: [...raw.content].length,
563
- });
628
+ // a server-discarded reasoning block mid-emission. This diagnostic
629
+ // requires the serving vocabulary; an estimate cannot prove absence.
630
+ try {
631
+ const [contentTokens, reasoningTokens] = await Promise.all([
632
+ this.tokenize(raw.content),
633
+ this.tokenize(raw.reasoning),
634
+ ]);
635
+ const visible = contentTokens.length + reasoningTokens.length;
636
+ if (usage.completion > visible + 64) {
637
+ (notices ??= []).push({
638
+ source: this.#source,
639
+ kind: "grammar_unenforced",
640
+ level: "warn",
641
+ message: `decode escaped the grammar: ${usage.completion} completion tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
642
+ position: [...raw.content].length,
643
+ });
644
+ }
645
+ }
646
+ catch (cause) {
647
+ emitWarningOnce(`${this.#source}: exact visible-token diagnostic unavailable (${cause instanceof Error ? cause.message : String(cause)})`, "PLURNK_VISIBLE_TOKEN_COUNT_UNAVAILABLE");
564
648
  }
565
649
  }
566
- const builtMeta = this.#buildMeta(raw.metadata);
567
- const meta = railsMeta !== undefined ? { ...builtMeta, ...railsMeta } : builtMeta;
650
+ const meta = this.#buildMeta(raw.metadata);
568
651
  const logprobs = raw.logprobs.length > 0 ? raw.logprobs : undefined;
569
652
  const meanLogprob = logprobs !== undefined
570
653
  ? logprobs.reduce((sum, token) => sum + token.logprob, 0) / logprobs.length
571
654
  : undefined;
572
- return {
573
- assistant: {
574
- content: raw.content,
575
- reasoning: raw.reasoning.length > 0 ? raw.reasoning : null,
576
- ...(raw.reasoningEncrypted.length > 0
577
- ? { reasoningEncrypted: raw.reasoningEncrypted }
578
- : {}),
579
- usage,
580
- finishReason: raw.finishReason,
581
- model: raw.model,
582
- ...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
583
- },
655
+ const assistant = {
656
+ content: raw.content,
657
+ reasoning: raw.reasoning.length > 0 ? raw.reasoning : null,
658
+ ...(raw.reasoningEncrypted.length > 0
659
+ ? { reasoningEncrypted: raw.reasoningEncrypted }
660
+ : {}),
661
+ usage,
662
+ model: raw.model,
663
+ ...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
664
+ };
665
+ const normalizedCharge = this.#normalizeCharge?.(raw.chargeEvidence);
666
+ const charge = normalizedCharge === undefined
667
+ ? undefined
668
+ : validateAuthoritativeCharge(normalizedCharge);
669
+ const evidence = {
584
670
  assistantRaw: raw,
671
+ ...(charge === undefined ? {} : { charge }),
672
+ ...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
585
673
  ...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
586
674
  ...(meta !== undefined ? { meta } : {}),
587
- ...(telemetry !== undefined ? { telemetry } : {}),
675
+ ...(notices !== undefined ? { notices } : {}),
676
+ };
677
+ if (raw.finishReason === "resource_interrupted") {
678
+ const attempt = {
679
+ assistant: { ...assistant, finishReason: raw.finishReason },
680
+ ...evidence,
681
+ };
682
+ throw new ProviderError(this.#source, "resource_interrupted", "The provider interrupted generation because inference resources were unavailable.", {
683
+ attempt,
684
+ extensions: {
685
+ stage: "provider-response",
686
+ finishReason: "resource_interrupted",
687
+ ...(raw.rawFinishReason === undefined
688
+ ? {}
689
+ : { rawFinishReason: raw.rawFinishReason }),
690
+ },
691
+ });
692
+ }
693
+ return {
694
+ assistant: { ...assistant, finishReason: raw.finishReason },
695
+ ...evidence,
588
696
  };
589
697
  }
590
698
  }