@plurnk/plurnk-providers 1.3.11 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/.env.defaults +40 -23
  2. package/README.md +65 -4
  3. package/SPEC.md +222 -56
  4. package/dist/AiSdkProvider.d.ts +18 -5
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +240 -153
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +14 -15
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +26 -10
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +8 -2
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +41 -8
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/ProviderRegistry.d.ts +4 -1
  17. package/dist/ProviderRegistry.d.ts.map +1 -1
  18. package/dist/ProviderRegistry.js +7 -3
  19. package/dist/ProviderRegistry.js.map +1 -1
  20. package/dist/aiSdkTransport.d.ts +4 -2
  21. package/dist/aiSdkTransport.d.ts.map +1 -1
  22. package/dist/aiSdkTransport.js +18 -3
  23. package/dist/aiSdkTransport.js.map +1 -1
  24. package/dist/catalogProvider.d.ts +3 -1
  25. package/dist/catalogProvider.d.ts.map +1 -1
  26. package/dist/catalogProvider.js +20 -7
  27. package/dist/catalogProvider.js.map +1 -1
  28. package/dist/compatibleProvider.d.ts.map +1 -1
  29. package/dist/compatibleProvider.js +11 -4
  30. package/dist/compatibleProvider.js.map +1 -1
  31. package/dist/cost.d.ts +11 -0
  32. package/dist/cost.d.ts.map +1 -0
  33. package/dist/cost.js +64 -0
  34. package/dist/cost.js.map +1 -0
  35. package/dist/discover.d.ts +2 -0
  36. package/dist/discover.d.ts.map +1 -1
  37. package/dist/discover.js +15 -9
  38. package/dist/discover.js.map +1 -1
  39. package/dist/env.d.ts +4 -0
  40. package/dist/env.d.ts.map +1 -1
  41. package/dist/env.js +29 -9
  42. package/dist/env.js.map +1 -1
  43. package/dist/errors.d.ts +27 -0
  44. package/dist/errors.d.ts.map +1 -0
  45. package/dist/errors.js +150 -0
  46. package/dist/errors.js.map +1 -0
  47. package/dist/index.d.ts +11 -5
  48. package/dist/index.d.ts.map +1 -1
  49. package/dist/index.js +7 -4
  50. package/dist/index.js.map +1 -1
  51. package/dist/notices.d.ts +10 -0
  52. package/dist/notices.d.ts.map +1 -0
  53. package/dist/notices.js +11 -0
  54. package/dist/notices.js.map +1 -0
  55. package/dist/ollama.d.ts.map +1 -1
  56. package/dist/ollama.js +3 -3
  57. package/dist/ollama.js.map +1 -1
  58. package/dist/openai.d.ts +1 -1
  59. package/dist/openai.d.ts.map +1 -1
  60. package/dist/promptTokens.d.ts +4 -0
  61. package/dist/promptTokens.d.ts.map +1 -0
  62. package/dist/promptTokens.js +32 -0
  63. package/dist/promptTokens.js.map +1 -0
  64. package/dist/sdkModels.d.ts.map +1 -1
  65. package/dist/sdkModels.js +4 -3
  66. package/dist/sdkModels.js.map +1 -1
  67. package/dist/types.d.ts +43 -16
  68. package/dist/types.d.ts.map +1 -1
  69. package/dist/types.js +1 -1
  70. package/dist/types.js.map +1 -1
  71. package/dist/usage.d.ts +3 -0
  72. package/dist/usage.d.ts.map +1 -1
  73. package/dist/usage.js +26 -14
  74. package/dist/usage.js.map +1 -1
  75. package/dist/warnings.js +0 -0
  76. package/dist/warnings.js.map +1 -1
  77. package/package.json +13 -9
  78. package/src/AiSdkProvider.test.ts +480 -159
  79. package/src/AiSdkProvider.ts +320 -196
  80. package/src/Mock.test.ts +29 -14
  81. package/src/Mock.ts +33 -15
  82. package/src/Pool.test.ts +43 -6
  83. package/src/Pool.ts +56 -10
  84. package/src/ProviderRegistry.test.ts +158 -9
  85. package/src/ProviderRegistry.ts +19 -6
  86. package/src/aiSdkTransport.ts +25 -6
  87. package/src/boundaries.test.ts +8 -3
  88. package/src/catalogProvider.test.ts +17 -0
  89. package/src/catalogProvider.ts +25 -10
  90. package/src/compatibleProvider.test.ts +96 -0
  91. package/src/compatibleProvider.ts +15 -6
  92. package/src/cost.test.ts +63 -0
  93. package/src/cost.ts +83 -0
  94. package/src/defaults.test.ts +1 -0
  95. package/src/discover.test.ts +48 -7
  96. package/src/discover.ts +31 -21
  97. package/src/env.test.ts +38 -23
  98. package/src/env.ts +45 -18
  99. package/src/errors.test.ts +148 -0
  100. package/src/errors.ts +207 -0
  101. package/src/index.ts +29 -7
  102. package/src/lexicon-guard.test.ts +6 -6
  103. package/src/notices.ts +22 -0
  104. package/src/ollama.test.ts +64 -0
  105. package/src/ollama.ts +6 -3
  106. package/src/openai.ts +3 -0
  107. package/src/promptTokens.ts +41 -0
  108. package/src/sdkModels.test.ts +7 -0
  109. package/src/sdkModels.ts +4 -8
  110. package/src/types.ts +106 -64
  111. package/src/usage.test.ts +15 -4
  112. package/src/usage.ts +32 -14
  113. package/src/warnings.test.ts +10 -10
  114. package/src/warnings.ts +0 -0
  115. package/dist/OpenAICompat.d.ts +0 -76
  116. package/dist/OpenAICompat.d.ts.map +0 -1
  117. package/dist/OpenAICompat.js +0 -555
  118. package/dist/OpenAICompat.js.map +0 -1
  119. package/dist/openaiStream.d.ts +0 -47
  120. package/dist/openaiStream.d.ts.map +0 -1
  121. package/dist/openaiStream.js +0 -280
  122. package/dist/openaiStream.js.map +0 -1
  123. package/dist/standardProviders.d.ts +0 -31
  124. package/dist/standardProviders.d.ts.map +0 -1
  125. package/dist/standardProviders.js +0 -518
  126. package/dist/standardProviders.js.map +0 -1
  127. package/dist/telemetry.d.ts +0 -24
  128. package/dist/telemetry.d.ts.map +0 -1
  129. package/dist/telemetry.js +0 -85
  130. package/dist/telemetry.js.map +0 -1
  131. package/src/telemetry.test.ts +0 -69
  132. package/src/telemetry.ts +0 -116
@@ -6,10 +6,12 @@
6
6
  // ordinary vendor protocol. The compatible URL path remains only for PLURNK
7
7
  // extensions and local endpoint probes the SDK cannot represent.
8
8
  import { executeAiSdkModel, executeOpenAICompatible } from "./aiSdkTransport.js";
9
- import { toProviderError, ProviderError } from "./telemetry.js";
9
+ import { toProviderError, ProviderError } from "./errors.js";
10
+ import { attributeUnitemizedReasoning } from "./usage.js";
10
11
  import { validateGbnf } from "@plurnk/gbnf";
12
+ import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.js";
11
13
  import { emitWarningOnce } from "./warnings.js";
12
- // #539: drop trailing occurrences of a server-rendered EOG marker. llama-server
14
+ // Drop trailing occurrences of a server-rendered EOG marker. llama-server
13
15
  // under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
14
16
  // trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
15
17
  // can never eat body content (a body ending in the literal marker isn't producible
@@ -22,6 +24,32 @@ const stripTrailingSpecial = (content, marker) => {
22
24
  out = out.slice(0, -marker.length);
23
25
  return out;
24
26
  };
27
+ // {§provider-tagged-reasoning} Only the model-contract position is structural:
28
+ // one exact leading envelope. Parsing after stream assembly keeps SSE and JSON
29
+ // on one path and leaves later literal tags in the visible suffix untouched.
30
+ const projectTaggedReasoning = (content, structuredReasoning, style) => {
31
+ const opening = "<think>";
32
+ if (style !== "think-tags" || structuredReasoning.length > 0 || !content.startsWith(opening)) {
33
+ return { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
34
+ }
35
+ const closing = "</think>";
36
+ const closingIndex = content.indexOf(closing, opening.length);
37
+ if (closingIndex === -1) {
38
+ return {
39
+ content: "",
40
+ reasoning: content.slice(opening.length),
41
+ projected: true,
42
+ contentStart: [...content].length,
43
+ };
44
+ }
45
+ const suffixStart = closingIndex + closing.length;
46
+ return {
47
+ content: content.slice(suffixStart),
48
+ reasoning: content.slice(opening.length, closingIndex),
49
+ projected: true,
50
+ contentStart: [...content.slice(0, suffixStart)].length,
51
+ };
52
+ };
25
53
  // Shared budget→effort breakpoints (xai and google had identical copies).
26
54
  export const effortFromBudget = (budget) => {
27
55
  if (budget <= 1000)
@@ -30,41 +58,27 @@ export const effortFromBudget = (budget) => {
30
58
  return "medium";
31
59
  return "high";
32
60
  };
33
- // chars/2 upper bound (see ./tokenizers.ts) — overcounts safely, never under.
34
- const heuristicTokens = (text) => (text.length === 0 ? 0 : Math.ceil(text.length / 2));
35
61
  // Body keys the provider owns — a caller's `sampling` passthrough may not set
36
- // these. Two families (#477 audit):
62
+ // these. Two families:
37
63
  // transport/managed — grammar transport, the stream/JSON choice, slot pinning,
38
- // data capture (SPEC §8: backend-specific fields never cross the contract);
64
+ // data capture ({§provider-evidence}: backend-specific fields never cross the contract);
39
65
  // contract invariants — `n` (atomic single completion: choices[0] is the
40
66
  // response; n>1 = paid, dropped output), the tool-calling family (tools-in-
41
67
  // body doctrine, §2: native tool_calls return null content = a broken turn),
42
68
  // modalities/audio (text-only contract), prediction (decode semantics, not
43
69
  // sampling), and the token caps (the envelope is the managed maxTokens —
44
- // sampling must not bypass the consumer's #425 cap).
70
+ // sampling must not bypass the consumer's cap).
45
71
  // Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
46
72
  // platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
47
73
  // metadata, store, verbosity) pass through; the managed floors spread UNDER
48
74
  // sampling stay deliberately caller-overridable.
49
75
  const RESERVED_BODY_KEYS = new Set([
50
76
  "model", "messages", "stream", "stream_options", "grammar", "response_format", "id_slot", "logprobs", "top_logprobs",
77
+ "reasoning_format", "reasoning_effort", "thinking", "think", "include_reasoning", "chat_template_kwargs", "thinking_budget_tokens", // lexicon-allow: backend wire fields
51
78
  "n", "tools", "tool_choice", "functions", "function_call", "parallel_tool_calls",
52
79
  "modalities", "audio", "prediction", "max_tokens", "max_completion_tokens",
53
80
  "prompt_cache_key",
54
81
  ]);
55
- // Render a non-accept verdict into a terse, factual grammar_unenforced message
56
- // (SPEC §12 message policy: no guidance prose). `reject` names the diverging code
57
- // point + what the grammar would have accepted; `incomplete` names the valid-prefix
58
- // length that never reached a terminal state.
59
- const describeUnenforced = (v) => {
60
- if (v.status === "reject") {
61
- const expected = v.expected.length > 0
62
- ? v.expected.map((e) => `${e.rule} accepts ${e.accepts}`).join(", ")
63
- : "end of input";
64
- return `grammar not enforced: output rejected by the transported grammar at code point ${v.pos} (${JSON.stringify(v.char)}); expected ${expected}`;
65
- }
66
- return `grammar not enforced: output is an incomplete match of the transported grammar — a valid prefix of ${v.pos} code points that never terminated`;
67
- };
68
82
  export default class AiSdkProvider {
69
83
  #model;
70
84
  #url;
@@ -86,8 +100,11 @@ export default class AiSdkProvider {
86
100
  #dryAllowedLength;
87
101
  #repeatLastN;
88
102
  #reasoningStyle;
89
- #countTokens;
103
+ #reasoningResponseStyle;
104
+ #countPromptTokens;
105
+ #promptTokensUrl;
90
106
  #calculateCost;
107
+ #calculateCharge;
91
108
  #source;
92
109
  #grammarStyle;
93
110
  #promptCacheKey;
@@ -98,6 +115,7 @@ export default class AiSdkProvider {
98
115
  #supportsSlotPinning;
99
116
  #slotCount;
100
117
  #retryAttempts;
118
+ #errorDetailLimit;
101
119
  #topLogprobs;
102
120
  #reasoningReserve;
103
121
  #completionReserve;
@@ -105,7 +123,8 @@ export default class AiSdkProvider {
105
123
  #rawBody;
106
124
  #servedModel;
107
125
  #requiresMaxTokens;
108
- // Optional capability (SPEC §2): exact tokenization served by the backend's
126
+ attributions;
127
+ // Optional capability ({§provider-local-capabilities}): exact tokenization served by the backend's
109
128
  // own vocab. Assigned in the constructor ONLY when the config carries a
110
129
  // tokenizeUrl (llama-server), so `provider.tokenize === undefined` remains
111
130
  // the honest capability signal for every other backend.
@@ -114,6 +133,7 @@ export default class AiSdkProvider {
114
133
  this.#model = config.model;
115
134
  this.#url = config.url;
116
135
  this.#languageModel = config.languageModel;
136
+ this.attributions = config.attributions;
117
137
  if ((this.#url === undefined) === (this.#languageModel === undefined)) {
118
138
  throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
119
139
  }
@@ -137,9 +157,16 @@ export default class AiSdkProvider {
137
157
  this.#dryAllowedLength = config.dryAllowedLength;
138
158
  this.#repeatLastN = config.repeatLastN;
139
159
  this.#retryAttempts = config.retryAttempts;
160
+ this.#errorDetailLimit = config.errorDetailLimit;
140
161
  this.#reasoningStyle = config.reasoningStyle ?? "none";
141
- this.#countTokens = config.countTokens ?? heuristicTokens;
162
+ this.#reasoningResponseStyle = config.reasoningResponseStyle ?? "verbatim";
163
+ if (config.countPromptTokens !== undefined && config.promptTokensUrl !== undefined) {
164
+ throw new Error(`${config.source ?? "provider"}: configure countPromptTokens or promptTokensUrl, not both`);
165
+ }
166
+ this.#countPromptTokens = config.countPromptTokens ?? ((messages) => estimatePromptTokens(messages));
167
+ this.#promptTokensUrl = config.promptTokensUrl;
142
168
  this.#calculateCost = config.calculateCost ?? (() => 0);
169
+ this.#calculateCharge = config.calculateCharge;
143
170
  this.#source = config.source ?? "provider";
144
171
  this.#grammarStyle = config.grammarStyle ?? "none";
145
172
  this.#promptCacheKey = config.promptCacheKey ?? false;
@@ -159,6 +186,13 @@ export default class AiSdkProvider {
159
186
  this.#rawBody = config.rawBody ?? false;
160
187
  this.#servedModel = config.servedModel;
161
188
  this.#requiresMaxTokens = config.requiresMaxTokens;
189
+ const reasoningReserve = this.reasoningReserve;
190
+ if (this.#reasoningStyle === "template"
191
+ && this.#reasoning.mode === "on"
192
+ && reasoningReserve !== null
193
+ && this.#reasoning.budget > reasoningReserve) {
194
+ throw new Error(`${this.#source}: PLURNK_PROVIDERS_REASONING_BUDGET (${this.#reasoning.budget}) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE (${reasoningReserve})`);
195
+ }
162
196
  const { tokenizeUrl } = config;
163
197
  if (tokenizeUrl !== undefined) {
164
198
  this.tokenize = async (text) => {
@@ -179,7 +213,7 @@ export default class AiSdkProvider {
179
213
  }
180
214
  }
181
215
  get contextWindow() { return this.#contextWindow; }
182
- // #507: envelope reserves — absolute pins stand alone; percentages need the
216
+ // {§provider-generation-envelope} Envelope reserves — absolute pins stand alone; percentages need the
183
217
  // detected window; null = underivable (no claim for core's no-cap path).
184
218
  #resolveReserve(spec) {
185
219
  if (spec === undefined)
@@ -191,61 +225,92 @@ export default class AiSdkProvider {
191
225
  get reasoningReserve() { return this.#resolveReserve(this.#reasoningReserve); }
192
226
  get completionReserve() { return this.#resolveReserve(this.#completionReserve); }
193
227
  get model() { return this.#model; }
194
- // #37: backend's self-reported served id; undefined when unprobed/unknown.
228
+ // Backend's self-reported served id; undefined when unprobed/unknown.
195
229
  get servedModel() { return this.#servedModel; }
196
- // #43: resolved "decodes unbounded without a cap" fact; undefined = no claim.
230
+ // Resolved "decodes unbounded without a cap" fact; undefined = no claim.
197
231
  get requiresMaxTokens() { return this.#requiresMaxTokens; }
198
- // Resolved capability (#34): will a transported grammar actually constrain
232
+ // Resolved capability: will a transported grammar actually constrain
199
233
  // this backend's decode? Introspectable so a consumer can verify the rails
200
234
  // are LIVE without spending a generation on a forcing-grammar probe.
201
235
  get constrainsOutput() { return this.#grammarStyle !== "none"; }
202
- countTokens(text) { return this.#countTokens(text); }
236
+ async countPromptTokens(messages, signal) {
237
+ if (this.#promptTokensUrl === undefined) {
238
+ return assertPromptTokenMeasurement(await this.#countPromptTokens(messages, signal), this.#source);
239
+ }
240
+ signal?.throwIfAborted();
241
+ try {
242
+ const timeout = AbortSignal.timeout(this.#fetchTimeoutMs);
243
+ const response = await this.#fetch(this.#promptTokensUrl, {
244
+ method: "POST",
245
+ headers: { "Content-Type": "application/json", ...this.#headers },
246
+ body: JSON.stringify({
247
+ model: this.#model,
248
+ messages,
249
+ ...this.#reasoningBody(),
250
+ }),
251
+ signal: signal === undefined ? timeout : AbortSignal.any([signal, timeout]),
252
+ });
253
+ if (!response.ok) {
254
+ return estimatePromptTokens(messages, `llama-server input-token endpoint returned HTTP ${response.status}`);
255
+ }
256
+ const body = await response.json();
257
+ if (!Number.isInteger(body.input_tokens) || body.input_tokens < 0) {
258
+ return estimatePromptTokens(messages, "llama-server input-token endpoint returned no non-negative integer input_tokens");
259
+ }
260
+ return {
261
+ kind: "exact",
262
+ tokens: body.input_tokens,
263
+ source: "llama-server:/v1/chat/completions/input_tokens",
264
+ };
265
+ }
266
+ catch (cause) {
267
+ signal?.throwIfAborted();
268
+ return estimatePromptTokens(messages, `llama-server input-token measurement failed: ${cause instanceof Error ? cause.message : String(cause)}`);
269
+ }
270
+ }
203
271
  calculateCost(usage) { return this.#calculateCost(usage); }
204
- // Maps the reasoning INTENT (off | adaptive | on+budget, #33) to the
205
- // backend's wire mechanism — including under a transported grammar. The #32
206
- // clamp (force reasoning_effort "none" under response_format) is LIFTED:
207
- // canary-verified live that fireworks masks ONLY the content channel — the
208
- // reasoning channel rides beside it unmasked, and the plurnk grammar's
209
- // reasoning?/preplan regions absorb any in-band spillover. The old measured
210
- // failures (low→cycles, high→spirals) were pre-max_tokens-cap; with a bounded
211
- // cap the matrix ACCEPTs across efforts (reasoning-rails matrix
212
- // F9). Clamping was the root of the plan-less regression (service#331).
272
+ calculateCharge(usage) {
273
+ return this.#calculateCharge?.(usage)
274
+ ?? { kind: "unknown", reason: "no provider rate or settled charge is available" };
275
+ }
276
+ // Reasoning intent maps independently of grammar transport. The llama-server
277
+ // template mapping is owned by {§llama-reasoning-request}.
213
278
  #reasoningBody() {
214
279
  const { mode, budget } = this.#reasoning;
215
280
  const on = mode !== "off";
216
281
  switch (this.#reasoningStyle) {
217
- // Native-channel styles. "template" ALWAYS emits — the explicit
218
- // enable_thinking:false is the only working off-switch on llama-server
219
- // (§13). Activation only; budget is enforced by the box's
220
- // --reasoning-budget launch flag (per-request numerics ignored, F7).
221
- //
222
- // #488 postmortem: intent maps IDENTICALLY under a transported
223
- // grammar. The brief rails-win-the-channel clamp (enable_thinking
224
- // forced false under a grammar) is REVERTED — specimens proved the
225
- // SANCTIONED think block is the protection, not the hazard: the
226
- // server auto-gates the grammar around it and content decodes
227
- // constrained (26-run baseline green; zero grammar rejects across
228
- // the #488 "railless" specimens). Closing the channel starved a
229
- // reasoning-tuned model into ESCAPING mid-content into the raw
230
- // thought channel — discarded server-side, decode unconstrained,
231
- // 12,288 tokens billed for 1,033 visible chars. The escape is
232
- // surfaced instead (vanished-token telemetry + meta rail state).
233
- case "template": return { chat_template_kwargs: { enable_thinking: on } };
282
+ case "template": {
283
+ const allowance = mode === "off"
284
+ ? 0
285
+ : mode === "on" ? budget : this.reasoningReserve;
286
+ return {
287
+ chat_template_kwargs: { enable_thinking: on },
288
+ reasoning_format: "auto",
289
+ ...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
290
+ };
291
+ }
234
292
  case "think": return on ? { think: true } : {};
235
293
  case "include_reasoning": return on ? { include_reasoning: true } : {};
236
294
  // effort tiers from the budget; off/adaptive omit the field (the
237
295
  // API's default depth is its adaptive).
238
296
  case "effort": return mode === "on" ? { reasoning_effort: effortFromBudget(budget) } : {};
239
297
  // Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
240
- // reason-by-default model (DeepSeek V4: default 'high') reasoning (#30).
298
+ // reason-by-default model (DeepSeek V4: default 'high') reasoning.
241
299
  // ADAPTIVE omits the field: the backend's own default posture IS the
242
300
  // adaptive semantics, and the literal "adaptive" is MiniMax-M3-only —
243
- // fireworks 400s it for every other model (wire-verified, #403; the
301
+ // Fireworks 400s it for every other model (wire-verified; the
244
302
  // 1.0.2 adaptive default refused to boot on it). V4 gotcha: integer
245
303
  // efforts 400.
246
304
  case "effort_explicit": return mode === "off"
247
305
  ? { reasoning_effort: "none" }
248
306
  : mode === "on" ? { reasoning_effort: effortFromBudget(budget) } : {};
307
+ // {§deepseek-reasoning-request}
308
+ case "thinking_effort": return mode === "off"
309
+ ? { thinking: { type: "disabled" } }
310
+ : mode === "on" ? {
311
+ thinking: { type: "enabled" },
312
+ reasoning_effort: effortFromBudget(budget),
313
+ } : {};
249
314
  // Anthropic compat: explicit thinking object. off → disabled; on →
250
315
  // enabled with budget_tokens; adaptive → omit (the API default).
251
316
  case "anthropic": return mode === "off"
@@ -254,7 +319,7 @@ export default class AiSdkProvider {
254
319
  case "none": return {};
255
320
  }
256
321
  }
257
- // Per-run slot affinity (#11): the consumer passes WHICH run this is; the
322
+ // Per-worker slot affinity: the consumer passes which worker this is; the
258
323
  // provider owns WHICH slot serves it. Sticky per workerId, round-robin across
259
324
  // new runs (distinct runs → distinct slots while slots last), LRU-bounded
260
325
  // bookkeeping so a long-lived daemon never grows the map unboundedly —
@@ -277,19 +342,19 @@ export default class AiSdkProvider {
277
342
  this.#runSlots.set(workerId, slot);
278
343
  return { id_slot: slot };
279
344
  }
280
- // Optional local llama-server GBNF transport (SPEC §13). Unsupported
345
+ // Optional local llama-server GBNF transport ({§gbnf-response-observation}). Unsupported
281
346
  // backends receive no grammar-related field.
282
347
  #grammarBody(grammar) {
283
348
  if (grammar === undefined)
284
349
  return {};
285
350
  switch (this.#grammarStyle) {
286
351
  // Greedy decoding under hard constraint loops without a repeat-penalty
287
- // floor (#9, SPEC §13) — llama.cpp spells it `repeat_penalty`.
352
+ // floor — llama.cpp spells it `repeat_penalty`.
288
353
  case "llamacpp": return { grammar, repeat_penalty: this.#repeatPenalty };
289
354
  case "none": return {};
290
355
  }
291
356
  }
292
- // Anti-degeneration DEFAULT on EVERY request (#426), keyed to the backend's wire
357
+ // Anti-degeneration default on every request, keyed to the backend's wire
293
358
  // convention - NOT grammar-bound. GBNF is a local constraint, so a cloud
294
359
  // alias runs the sampler bare: firefast (deepseek/fireworks) ran 4/86 bench turns
295
360
  // straight to the token cap on pure looped repetition (run52). Ships next to
@@ -300,7 +365,7 @@ export default class AiSdkProvider {
300
365
  // live: together/deepinfra/fireworks; it is OpenAI's own param, so real OpenAI takes it too).
301
366
  #repetitionPenaltyBody() {
302
367
  switch (this.#grammarStyle) {
303
- // #567: repeat_penalty + optional DRY (repeated-SEQUENCE penalty) + a wider
368
+ // repeat_penalty + optional DRY (repeated-sequence penalty) + a wider
304
369
  // repeat_last_n window — the loop-breaking tools a llama.cpp backend serves.
305
370
  // Each rides only when its operator knob is set; absent = the box's default.
306
371
  case "llamacpp": return {
@@ -315,12 +380,12 @@ export default class AiSdkProvider {
315
380
  case "none": return this.#frequencyPenalty > 0 ? { frequency_penalty: this.#frequencyPenalty } : {};
316
381
  }
317
382
  }
318
- // First-party telemetry headers (SPEC §5): forwarded ONLY when the spec
383
+ // First-party telemetry headers ({§provider-request-authority}): forwarded only when the spec
319
384
  // opted in (the plurnk endpoint). The gate is here, not at the call site, so
320
385
  // attributions/client/strikes can never reach a third-party backend even if
321
386
  // the consumer passes them to the wrong provider. Empty values emit no header
322
387
  // — EXCEPT strikes, where 0 is a real value (clean streak) distinct from
323
- // absent (consumer didn't report); contract per plurnk-service#313. Strikes
388
+ // absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
324
389
  // ride HTTP headers only — the packet never carries them (the model must
325
390
  // never see strike state; engine accounting is not a metric to game).
326
391
  #metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn) {
@@ -333,11 +398,11 @@ export default class AiSdkProvider {
333
398
  h["Plurnk-Client"] = client;
334
399
  if (strikes !== undefined && Number.isInteger(strikes) && strikes >= 0)
335
400
  h["Plurnk-Strikes"] = String(strikes);
336
- // Worker identity (#26, wire-name completed #486/#511): the opaque workerId
401
+ // Worker identity: the opaque workerId
337
402
  // the consumer already supplies, forwarded so the endpoint can key
338
403
  // per-worker affinity/telemetry — same gate as every first-party signal.
339
404
  h["Plurnk-Worker-Id"] = workerId;
340
- // Root worker of the lineage (#522): the no-parent ancestor of this turn's
405
+ // Root worker of the lineage ({§worker-primary}): the no-parent ancestor of this turn's
341
406
  // worker tree. The consumer classifies primary-vs-spawned by equality
342
407
  // (primaryWorkerId == workerId ⇒ the primary/root worker). The provider
343
408
  // EMITS what the consumer supplies and never invents a primary; the
@@ -346,7 +411,7 @@ export default class AiSdkProvider {
346
411
  // the endpoint to surface, not a provider default.
347
412
  if (primaryWorkerId !== undefined && primaryWorkerId.length > 0)
348
413
  h["Plurnk-Worker-Primary"] = primaryWorkerId;
349
- // Turn coordinate (#404, extends #26 per #391): workspace/loop/turn, the
414
+ // Turn coordinate ({§lifecycle-terms}): workspace/loop/turn, the
350
415
  // daemon-side sequence the endpoint can never scrape from the wire.
351
416
  // Coordinates are 1-based — 0 is not a real value, so no strikes-style
352
417
  // zero exception; absent/empty/0 emits no header.
@@ -358,30 +423,7 @@ export default class AiSdkProvider {
358
423
  h["Plurnk-Turn"] = String(turn);
359
424
  return h;
360
425
  }
361
- // Enforcement verification (SPEC §13). When a grammar was actually transported
362
- // (grammarStyle !== "none"), the backend MUST have constrained the output;
363
- // some silently drop the grammar field or mislabel the channel, and without
364
- // this check we would return unconstrained output as if enforced. STRICT: any
365
- // non-accept verdict (reject, or an incomplete/never-terminated match) is a
366
- // grammar_unenforced failure. A grammar our own validator can't parse — even
367
- // though the backend accepted it (a port-vs-llama.cpp gap) — is a non-fatal
368
- // verify gap: warn, don't fail a transport that may have worked. This is a
369
- // conformance check against the grammar we already hold, NOT a plurnk-DSL
370
- // parse (§8) — it stays grammar-generic and backend-agnostic.
371
- // Validate output against the grammar. Returns the verdict, or null on the
372
- // verify GAP — a grammar our own validator can't parse (a port-vs-llama.cpp
373
- // gap): warn, don't manufacture a conflict from a check that didn't run.
374
- #grammarVerdict(grammar, content) {
375
- try {
376
- return validateGbnf(grammar, content);
377
- }
378
- catch (cause) {
379
- // Once per (code, message) — #40: this fires PER TURN otherwise.
380
- emitWarningOnce(`${this.#source}: could not verify grammar enforcement — the transported grammar did not parse in @plurnk/gbnf (${cause.message})`, "PLURNK_GRAMMAR_UNVERIFIABLE");
381
- return null;
382
- }
383
- }
384
- // PLURNK_PROVIDERS_GBNF_DEBUG (SPEC §13): validate the supplied GBNF locally and fail
426
+ // PLURNK_PROVIDERS_GBNF_DEBUG ({§gbnf-response-observation}): validate the supplied GBNF locally and fail
385
427
  // hard if it's malformed, BEFORE any wire call — and the grammar is NOT
386
428
  // transported, so the request runs unconstrained. A debug aid to catch invalid
387
429
  // grammars (e.g. while editing the plurnk grammar) without a model round-trip;
@@ -396,7 +438,7 @@ export default class AiSdkProvider {
396
438
  throw new Error(`grammar validation (PLURNK_PROVIDERS_GBNF_DEBUG): invalid GBNF — ${cause.message}`, { cause });
397
439
  }
398
440
  }
399
- // Per-turn metadata bag (#23): pass the backend's non-standard top-level fields
441
+ // Per-turn metadata bag: pass the backend's non-standard top-level fields
400
442
  // through verbatim. Providers do not reinterpret vendor currency or account
401
443
  // metadata; a monetary value carries its own amount and currency.
402
444
  #buildMeta(chunkMetadata) {
@@ -407,7 +449,8 @@ export default class AiSdkProvider {
407
449
  // penalties, stop, seed, …) merged UNDER the managed body: model, messages,
408
450
  // reasoning, grammar (+ its repeat-penalty floor), max_tokens and slot always
409
451
  // win, and reserved transport/protocol keys are stripped so the passthrough
410
- // can't smuggle a grammar, a stream toggle, or a backend slot (SPEC §8).
452
+ // can't smuggle a grammar, a stream toggle, or a backend slot
453
+ // ({§provider-request-authority}).
411
454
  #samplingBody(sampling) {
412
455
  if (sampling === undefined)
413
456
  return {};
@@ -418,27 +461,24 @@ export default class AiSdkProvider {
418
461
  return out;
419
462
  }
420
463
  async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }) {
421
- // Boundary validation (SPEC §2): the worker identity is required.
464
+ // {§provider-interface} The worker identity is required.
422
465
  if (workerId === undefined || workerId.length === 0)
423
466
  throw new Error("generate: workerId is required — the worker's stable, opaque identity");
424
- // Reject before any wire call when already aborted (SPEC §10.8).
467
+ // Reject before any wire call when already aborted
468
+ // ({§provider-failure-normalization}).
425
469
  signal?.throwIfAborted();
426
- // Grammar handling (SPEC §13). PLURNK_PROVIDERS_GBNF_DEBUG validates the supplied
427
- // grammar locally and throws on a malformed one, then WITHHOLDS it so the
428
- // model generates UNCONSTRAINED — and the free output is still verified
429
- // against the grammar (below), surfacing exactly where the model's natural
430
- // output and the grammar conflict. Otherwise the grammar is sent when the
431
- // backend supports it (grammarStyle !== "none").
470
+ // Grammar handling ({§gbnf-response-observation}). Debug validates the
471
+ // supplied grammar before the call but withholds it from the backend.
432
472
  const wantGrammar = grammar !== undefined && this.#grammarStyle !== "none";
433
473
  if (wantGrammar && this.#gbnfDebug)
434
474
  this.#assertGrammarValid(grammar);
435
475
  const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
436
476
  // Assembly order = precedence: the family's sampling DEFAULTS
437
- // (PLURNK_PROVIDERS_TEMPERATURE — universal, #30 measured it on grammar
477
+ // (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
438
478
  // paths and the name promises every request) < the caller's `sampling`
439
479
  // < the managed fields, which always win.
440
480
  const body = {
441
- // #507: floors suppressed on router-owned-tuning providers (plurnk) —
481
+ // Floors are suppressed on router-owned-tuning providers (plurnk) —
442
482
  // the router's per-model tuning must not be overridden by client floors.
443
483
  ...(this.#tuningFloors ? { temperature: this.#temperature, ...this.#repetitionPenaltyBody() } : {}),
444
484
  ...this.#samplingBody(sampling),
@@ -448,11 +488,11 @@ export default class AiSdkProvider {
448
488
  ...this.#reasoningBody(),
449
489
  ...this.#grammarBody(sendGrammar),
450
490
  ...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
451
- // #36: request per-token logprobs only when enabled (managed field —
491
+ // Request per-token logprobs only when enabled (managed field —
452
492
  // reserved from caller sampling; the env flag is the single control).
453
493
  ...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
454
494
  ...this.#slotBody(workerId),
455
- // #518: prompt-cache affinity -- workerId as the OpenAI-standard
495
+ // Prompt-cache affinity -- workerId as the OpenAI-standard
456
496
  // prompt_cache_key routes a worker's turns to one serverless replica so
457
497
  // its stable prefix caches (managed; reserved from caller sampling).
458
498
  ...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
@@ -513,13 +553,13 @@ export default class AiSdkProvider {
513
553
  catch (err) {
514
554
  if (signal?.aborted)
515
555
  throw err;
516
- const pe = toProviderError(err, this.#source);
556
+ const pe = toProviderError(err, this.#source, this.#errorDetailLimit);
517
557
  if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
518
558
  throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
519
559
  }
520
560
  throw pe;
521
561
  }
522
- // #539: llama-server --special renders EOG tokens as text, so a turn ending
562
+ // llama-server --special renders EOG tokens as text, so a turn ending
523
563
  // via raw EOS carries a trailing <eos> the grammar never sanctioned - it both
524
564
  // false-rejects the rail verdict and leaks a control token into the packet.
525
565
  // Strip the server-reported eos_token from the tail ONCE, before the verdict
@@ -527,64 +567,111 @@ export default class AiSdkProvider {
527
567
  // wire text for forensics.
528
568
  if (this.#eosText !== undefined)
529
569
  raw.content = stripTrailingSpecial(raw.content, this.#eosText);
530
- // Grammar conformance (§13): bytes always flow; the verdict is an
531
- // observation. Same check whether the grammar was transported
532
- // (sendGrammar) or withheld (PLURNK_PROVIDERS_GBNF_DEBUG filter mode) — a
533
- // non-accept verdict attaches a grammar_unenforced telemetry event
534
- // (message + divergence position) and the response returns normally.
535
- // Discard/retry/escalate/self-correct is the consumer's policy.
536
- let telemetry;
537
- let railsMeta;
538
- const usage = raw.usage;
539
- const observedGrammar = sendGrammar ?? (wantGrammar && this.#gbnfDebug ? grammar : undefined);
540
- if (observedGrammar !== undefined) {
541
- const verdict = this.#grammarVerdict(observedGrammar, raw.content);
542
- if (verdict !== null && verdict.status !== "accept") {
543
- telemetry = [{ source: this.#source, kind: "grammar_unenforced", message: describeUnenforced(verdict), position: verdict.pos }];
570
+ const taggedReasoning = projectTaggedReasoning(raw.content, raw.reasoning, this.#reasoningResponseStyle);
571
+ // Preserve the exact sentence seen at the grammar boundary. llama-server's
572
+ // `reasoning_format: "auto"` projects one raw Harmony enclosure into the
573
+ // reasoning/content fields; the wire field's presence is the proof that the
574
+ // projection occurred. The provider represents this evidence and never grades it.
575
+ let grammarEvidence;
576
+ if (wantGrammar) {
577
+ if (taggedReasoning.projected) {
578
+ grammarEvidence = {
579
+ input: raw.content,
580
+ contentStart: taggedReasoning.contentStart,
581
+ transported: sendGrammar !== undefined,
582
+ };
583
+ }
584
+ else if (this.#reasoningStyle === "template" && this.#reasoning.mode !== "off") {
585
+ if (raw.reasoningProjected) {
586
+ const prefix = `<|channel>thought\n${raw.reasoning}<channel|>`;
587
+ grammarEvidence = {
588
+ input: `${prefix}${raw.content}`,
589
+ contentStart: [...prefix].length,
590
+ transported: sendGrammar !== undefined,
591
+ };
592
+ }
593
+ }
594
+ else {
595
+ grammarEvidence = {
596
+ input: raw.content,
597
+ contentStart: 0,
598
+ transported: sendGrammar !== undefined,
599
+ };
544
600
  }
545
- // #488 per-request loud state: rail attachment + conformance verdict
546
- // ride `meta` into the consumer's turn row, so a drill reads rail
547
- // presence PER TURN from the run db instead of inferring it from
548
- // output shape (the #488 misdiagnosis, twice).
549
- railsMeta = { railsAttached: sendGrammar !== undefined, railsVerdict: verdict?.status ?? "unverifiable" };
550
- // #488 channel-escape detector (the run105 class): completion tokens
601
+ }
602
+ if (taggedReasoning.projected) {
603
+ raw.content = taggedReasoning.content;
604
+ raw.reasoning = taggedReasoning.reasoning;
605
+ raw.usage = attributeUnitemizedReasoning(raw.usage, raw.reasoning, raw.content);
606
+ }
607
+ let notices;
608
+ const usage = raw.usage;
609
+ if (sendGrammar !== undefined && this.tokenize !== undefined) {
610
+ // Channel-escape detector: completion tokens
551
611
  // billed far beyond every visible channel mean the decode ESCAPED into
552
- // a server-discarded reasoning block mid-emission — unconstrained,
553
- // invisible, billed (12,288 billed vs 1,033 chars visible, live).
554
- // countTokens OVERCOUNTS text (chars/2 upper bound), so billed
555
- // exceeding visible-plus-slack is real vanishing, not estimator noise.
556
- const visible = this.#countTokens(raw.content) + this.#countTokens(raw.reasoning);
557
- if (sendGrammar !== undefined && usage.completion > visible + 64) {
558
- (telemetry ??= []).push({
559
- source: this.#source,
560
- kind: "grammar_unenforced",
561
- message: `decode escaped the grammar: ${usage.completion} completion tokens billed but only ~${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
562
- position: [...raw.content].length,
563
- });
612
+ // a server-discarded reasoning block mid-emission. This diagnostic
613
+ // requires the serving vocabulary; an estimate cannot prove absence.
614
+ try {
615
+ const [contentTokens, reasoningTokens] = await Promise.all([
616
+ this.tokenize(raw.content),
617
+ this.tokenize(raw.reasoning),
618
+ ]);
619
+ const visible = contentTokens.length + reasoningTokens.length;
620
+ if (usage.completion > visible + 64) {
621
+ (notices ??= []).push({
622
+ source: this.#source,
623
+ kind: "grammar_unenforced",
624
+ level: "warn",
625
+ message: `decode escaped the grammar: ${usage.completion} completion tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
626
+ position: [...raw.content].length,
627
+ });
628
+ }
629
+ }
630
+ catch (cause) {
631
+ emitWarningOnce(`${this.#source}: exact visible-token diagnostic unavailable (${cause instanceof Error ? cause.message : String(cause)})`, "PLURNK_VISIBLE_TOKEN_COUNT_UNAVAILABLE");
564
632
  }
565
633
  }
566
- const builtMeta = this.#buildMeta(raw.metadata);
567
- const meta = railsMeta !== undefined ? { ...builtMeta, ...railsMeta } : builtMeta;
634
+ const meta = this.#buildMeta(raw.metadata);
568
635
  const logprobs = raw.logprobs.length > 0 ? raw.logprobs : undefined;
569
636
  const meanLogprob = logprobs !== undefined
570
637
  ? logprobs.reduce((sum, token) => sum + token.logprob, 0) / logprobs.length
571
638
  : undefined;
572
- return {
573
- assistant: {
574
- content: raw.content,
575
- reasoning: raw.reasoning.length > 0 ? raw.reasoning : null,
576
- ...(raw.reasoningEncrypted.length > 0
577
- ? { reasoningEncrypted: raw.reasoningEncrypted }
578
- : {}),
579
- usage,
580
- finishReason: raw.finishReason,
581
- model: raw.model,
582
- ...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
583
- },
639
+ const assistant = {
640
+ content: raw.content,
641
+ reasoning: raw.reasoning.length > 0 ? raw.reasoning : null,
642
+ ...(raw.reasoningEncrypted.length > 0
643
+ ? { reasoningEncrypted: raw.reasoningEncrypted }
644
+ : {}),
645
+ usage,
646
+ model: raw.model,
647
+ ...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
648
+ };
649
+ const evidence = {
584
650
  assistantRaw: raw,
651
+ ...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
585
652
  ...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
586
653
  ...(meta !== undefined ? { meta } : {}),
587
- ...(telemetry !== undefined ? { telemetry } : {}),
654
+ ...(notices !== undefined ? { notices } : {}),
655
+ };
656
+ if (raw.finishReason === "resource_interrupted") {
657
+ const attempt = {
658
+ assistant: { ...assistant, finishReason: raw.finishReason },
659
+ ...evidence,
660
+ };
661
+ throw new ProviderError(this.#source, "resource_interrupted", "The provider interrupted generation because inference resources were unavailable.", {
662
+ attempt,
663
+ extensions: {
664
+ stage: "provider-response",
665
+ finishReason: "resource_interrupted",
666
+ ...(raw.rawFinishReason === undefined
667
+ ? {}
668
+ : { rawFinishReason: raw.rawFinishReason }),
669
+ },
670
+ });
671
+ }
672
+ return {
673
+ assistant: { ...assistant, finishReason: raw.finishReason },
674
+ ...evidence,
588
675
  };
589
676
  }
590
677
  }