@plurnk/plurnk-providers 1.3.12 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (139) hide show
  1. package/.env.defaults +47 -38
  2. package/README.md +68 -4
  3. package/SPEC.md +245 -62
  4. package/dist/AiSdkProvider.d.ts +19 -5
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +264 -156
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +15 -17
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +27 -10
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +8 -2
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +41 -8
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/ProviderRegistry.d.ts +4 -1
  17. package/dist/ProviderRegistry.d.ts.map +1 -1
  18. package/dist/ProviderRegistry.js +7 -3
  19. package/dist/ProviderRegistry.js.map +1 -1
  20. package/dist/accounting.d.ts +3 -0
  21. package/dist/accounting.d.ts.map +1 -0
  22. package/dist/accounting.js +84 -0
  23. package/dist/accounting.js.map +1 -0
  24. package/dist/aiSdkTransport.d.ts +5 -2
  25. package/dist/aiSdkTransport.d.ts.map +1 -1
  26. package/dist/aiSdkTransport.js +91 -5
  27. package/dist/aiSdkTransport.js.map +1 -1
  28. package/dist/catalogProvider.d.ts +5 -2
  29. package/dist/catalogProvider.d.ts.map +1 -1
  30. package/dist/catalogProvider.js +24 -11
  31. package/dist/catalogProvider.js.map +1 -1
  32. package/dist/compatibleProvider.d.ts.map +1 -1
  33. package/dist/compatibleProvider.js +11 -4
  34. package/dist/compatibleProvider.js.map +1 -1
  35. package/dist/cost.d.ts +11 -0
  36. package/dist/cost.d.ts.map +1 -0
  37. package/dist/cost.js +61 -0
  38. package/dist/cost.js.map +1 -0
  39. package/dist/discover.d.ts +2 -0
  40. package/dist/discover.d.ts.map +1 -1
  41. package/dist/discover.js +15 -9
  42. package/dist/discover.js.map +1 -1
  43. package/dist/env.d.ts +4 -6
  44. package/dist/env.d.ts.map +1 -1
  45. package/dist/env.js +27 -29
  46. package/dist/env.js.map +1 -1
  47. package/dist/errors.d.ts +27 -0
  48. package/dist/errors.d.ts.map +1 -0
  49. package/dist/errors.js +152 -0
  50. package/dist/errors.js.map +1 -0
  51. package/dist/index.d.ts +12 -6
  52. package/dist/index.d.ts.map +1 -1
  53. package/dist/index.js +8 -5
  54. package/dist/index.js.map +1 -1
  55. package/dist/notices.d.ts +10 -0
  56. package/dist/notices.d.ts.map +1 -0
  57. package/dist/notices.js +11 -0
  58. package/dist/notices.js.map +1 -0
  59. package/dist/ollama.d.ts.map +1 -1
  60. package/dist/ollama.js +3 -3
  61. package/dist/ollama.js.map +1 -1
  62. package/dist/openai.d.ts +1 -1
  63. package/dist/openai.d.ts.map +1 -1
  64. package/dist/promptTokens.d.ts +4 -0
  65. package/dist/promptTokens.d.ts.map +1 -0
  66. package/dist/promptTokens.js +32 -0
  67. package/dist/promptTokens.js.map +1 -0
  68. package/dist/sdkModels.d.ts +2 -0
  69. package/dist/sdkModels.d.ts.map +1 -1
  70. package/dist/sdkModels.js +17 -6
  71. package/dist/sdkModels.js.map +1 -1
  72. package/dist/types.d.ts +52 -16
  73. package/dist/types.d.ts.map +1 -1
  74. package/dist/types.js +1 -1
  75. package/dist/types.js.map +1 -1
  76. package/dist/usage.d.ts +4 -0
  77. package/dist/usage.d.ts.map +1 -1
  78. package/dist/usage.js +64 -19
  79. package/dist/usage.js.map +1 -1
  80. package/dist/warnings.js +0 -0
  81. package/dist/warnings.js.map +1 -1
  82. package/package.json +15 -10
  83. package/src/AiSdkProvider.test.ts +750 -169
  84. package/src/AiSdkProvider.ts +354 -200
  85. package/src/Mock.test.ts +29 -14
  86. package/src/Mock.ts +36 -15
  87. package/src/Pool.test.ts +43 -6
  88. package/src/Pool.ts +56 -10
  89. package/src/ProviderRegistry.test.ts +158 -9
  90. package/src/ProviderRegistry.ts +19 -6
  91. package/src/accounting.test.ts +58 -0
  92. package/src/accounting.ts +88 -0
  93. package/src/aiSdkTransport.ts +101 -8
  94. package/src/boundaries.test.ts +9 -3
  95. package/src/catalogProvider.test.ts +43 -15
  96. package/src/catalogProvider.ts +32 -16
  97. package/src/compatibleProvider.test.ts +96 -0
  98. package/src/compatibleProvider.ts +15 -6
  99. package/src/cost.test.ts +64 -0
  100. package/src/cost.ts +78 -0
  101. package/src/defaults.test.ts +1 -0
  102. package/src/discover.test.ts +48 -7
  103. package/src/discover.ts +31 -21
  104. package/src/env.test.ts +38 -48
  105. package/src/env.ts +43 -40
  106. package/src/errors.test.ts +148 -0
  107. package/src/errors.ts +208 -0
  108. package/src/index.ts +30 -8
  109. package/src/lexicon-guard.test.ts +6 -6
  110. package/src/notices.ts +22 -0
  111. package/src/ollama.test.ts +64 -0
  112. package/src/ollama.ts +6 -3
  113. package/src/openai.ts +3 -0
  114. package/src/promptTokens.ts +41 -0
  115. package/src/sdkModels.test.ts +29 -3
  116. package/src/sdkModels.ts +19 -11
  117. package/src/types.ts +125 -64
  118. package/src/usage.test.ts +24 -5
  119. package/src/usage.ts +72 -21
  120. package/src/warnings.test.ts +10 -10
  121. package/src/warnings.ts +0 -0
  122. package/dist/OpenAICompat.d.ts +0 -76
  123. package/dist/OpenAICompat.d.ts.map +0 -1
  124. package/dist/OpenAICompat.js +0 -555
  125. package/dist/OpenAICompat.js.map +0 -1
  126. package/dist/openaiStream.d.ts +0 -47
  127. package/dist/openaiStream.d.ts.map +0 -1
  128. package/dist/openaiStream.js +0 -280
  129. package/dist/openaiStream.js.map +0 -1
  130. package/dist/standardProviders.d.ts +0 -31
  131. package/dist/standardProviders.d.ts.map +0 -1
  132. package/dist/standardProviders.js +0 -518
  133. package/dist/standardProviders.js.map +0 -1
  134. package/dist/telemetry.d.ts +0 -24
  135. package/dist/telemetry.d.ts.map +0 -1
  136. package/dist/telemetry.js +0 -85
  137. package/dist/telemetry.js.map +0 -1
  138. package/src/telemetry.test.ts +0 -69
  139. package/src/telemetry.ts +0 -116
@@ -6,28 +6,25 @@
6
6
  // ordinary vendor protocol. The compatible URL path remains only for PLURNK
7
7
  // extensions and local endpoint probes the SDK cannot represent.
8
8
 
9
- import type { ChatMessage, FinishReason, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
10
- import type { Reasoning, ReserveSpec } from "./env.ts";
9
+ import type { AuthoritativeChargeNormalizer, ChatMessage, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
10
+ import type { ProviderCost } from "@plurnk/plurnk-contracts";
11
+ import type { Reasoning, ReasoningResponseStyle, ReserveSpec } from "./env.ts";
11
12
  import { executeAiSdkModel, executeOpenAICompatible } from "./aiSdkTransport.ts";
12
13
  import type { LanguageModel } from "ai";
13
- import { toProviderError, ProviderError, type TelemetryEvent } from "./telemetry.ts";
14
- import { validateGbnf, type Verdict } from "@plurnk/gbnf";
14
+ import { toProviderError, ProviderError } from "./errors.ts";
15
+ import { attributeUnitemizedReasoning } from "./usage.ts";
16
+ import type { ProviderNotice } from "./notices.ts";
17
+ import { validateGbnf } from "@plurnk/gbnf";
18
+ import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.ts";
15
19
  import { emitWarningOnce } from "./warnings.ts";
20
+ import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
21
+ import { validateAuthoritativeCharge } from "./cost.ts";
16
22
 
17
23
  export type ProviderFetch = typeof globalThis.fetch;
18
24
 
19
- // How the reasoning intent (PLURNK_PROVIDERS_REASONING: off | adaptive | on, plus
20
- // REASONING_BUDGET iff on — #32/#33) translates to each backend's wire mechanism
21
- // (SPEC §4); the per-style mapping lives in #reasoningBody. Non-obvious ones:
22
- // "template" ALWAYS emits enable_thinking — the explicit false is llama-server's
23
- // only working off-switch (§13); "anthropic" uses the `thinking` object and IGNORES
24
- // reasoning_effort; "effort_explicit" (fireworks) sends the EXPLICIT "none" for OFF
25
- // instead of omitting — reason-by-DEFAULT models (DeepSeek V4 defaults
26
- // 'high') keep reasoning when the field is omitted, fatal under an active grammar
27
- // (#30). Intent maps IDENTICALLY with or without a transported grammar — fireworks
28
- // masks only the content channel, so reasoning and rails coexist in one call
29
- // (canary-verified; the #32 clamp is lifted — it caused the plan-less service#331).
30
- export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "template" | "anthropic";
25
+ // Backend wire spellings for the resolved reasoning intent. The switch beside each
26
+ // mapping retains any backend-specific omission/explicit-disable constraint.
27
+ export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "thinking_effort" | "template" | "anthropic";
31
28
 
32
29
  // GBNF transport is a local llama-server capability. "none" means no
33
30
  // service-managed constrained sampling; endpoint-owned settings are not inferred.
@@ -37,17 +34,21 @@ export type AiSdkProviderConfig = {
37
34
  model: string;
38
35
  url?: string; // OpenAI-compatible chat-completions URL
39
36
  languageModel?: LanguageModel; // native AI SDK provider model
37
+ attributions?: (context: PluginAttributionContext) => PluginAttribution;
40
38
  fetchTimeoutMs: number;
41
39
  streamIdleTimeoutMs?: number; // streamed body inter-chunk deadline; zero/unset disables
42
40
  headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
43
41
  fetch?: ProviderFetch; // per-instance request executor; default globalThis.fetch
44
- contextWindow?: number | null; // default null; caller resolves-or-fails (#419), narrows to required with the interface
42
+ contextWindow?: number | null; // default null; caller resolves-or-fails, narrows to required with the interface
45
43
  reasoningStyle?: ReasoningStyle; // default "none"
46
- countTokens?: (text: string) => number; // default chars/2 upper-bound heuristic
44
+ reasoningResponseStyle?: ReasoningResponseStyle; // {§provider-tagged-reasoning}; default "verbatim"
45
+ countPromptTokens?: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
47
46
  calculateCost?: (usage: ProviderUsage) => number; // default () => 0
48
- source?: string; // telemetry source, e.g. "provider:openai"; default "provider"
47
+ calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }>;
48
+ normalizeCharge?: AuthoritativeChargeNormalizer;
49
+ source?: string; // notice/problem source, e.g. "provider:openai"; default "provider"
49
50
  grammarStyle?: GrammarStyle; // how a GBNF grammar is carried; default "none" (not sent)
50
- // #518: send the OpenAI-standard `prompt_cache_key` set to workerId, so a
51
+ // Send the OpenAI-standard `prompt_cache_key` set to workerId, so a
51
52
  // serverless backend with REPLICA-LOCAL prompt caching (fireworks, verified)
52
53
  // pins a worker's turns to one replica and claims its stable prefix. Default
53
54
  // false -- a backend that strict-validates unknown fields 400s, so enable only
@@ -59,21 +60,24 @@ export type AiSdkProviderConfig = {
59
60
  gbnfDebug?: boolean; // PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar locally + throw on invalid, but DON'T transport it (run unconstrained); default false
60
61
  streaming?: boolean; // SSE transport (default true); false → one non-streamed JSON
61
62
  firstPartyMetadata?: boolean; // forward per-turn attributions + client as Plurnk-* headers (plurnk only); default false
62
- apiKeyRejectedMessage?: string; // #537: friendly hint when a PRESENT key is 401/403-rejected (distinct from unset); default undefined
63
- eosText?: string; // #539: server-reported eos_token, stripped from the content tail (--special renders it as text); default undefined
64
- // Slot affinity wiring (provider-INTERNAL — never consumer-facing, #11).
63
+ apiKeyRejectedMessage?: string; // friendly hint when a present key is 401/403-rejected (distinct from unset); default undefined
64
+ eosText?: string; // server-reported eos_token, stripped from the content tail (--special renders it as text); default undefined
65
+ // Slot affinity wiring is provider-internal, never consumer-facing.
65
66
  supportsSlotPinning?: boolean; // backend accepts an `id_slot` body field (llama-server); default false
66
67
  slotCount?: number | null; // probed slot count for pinning backends; default null
67
68
  // Backend-served exact tokenization (llama-server /tokenize). When set, the
68
69
  // provider exposes the optional `tokenize()` capability — the model's OWN
69
70
  // vocab, no client-side tokenizer data needed; default unset (capability absent).
70
71
  tokenizeUrl?: string;
71
- // #37: the backend's self-reported served model id (from the /v1/models probe),
72
+ // Provider-authoritative count of a complete chat-completions request. This
73
+ // is distinct from /tokenize, which sees content but not the chat template.
74
+ promptTokensUrl?: string;
75
+ // The backend's self-reported served model id (from the /v1/models probe),
72
76
  // surfaced as Provider.servedModel. For a local llama-server the wire `model` is
73
77
  // the alias; this is the real name (the .gguf) the tokenizer seam maps. Absent
74
78
  // when no probe ran or it read no row.
75
79
  servedModel?: string;
76
- // #43: backend decodes unbounded without a caller cap (llama-server n_predict
80
+ // Backend decodes unbounded without a caller cap (llama-server n_predict
77
81
  // to the wall) — surfaced as Provider.requiresMaxTokens so consumers can
78
82
  // boot-refuse an envelope-less local alias. Default unset (no claim).
79
83
  requiresMaxTokens?: boolean;
@@ -81,36 +85,39 @@ export type AiSdkProviderConfig = {
81
85
  // (PLURNK_PROVIDERS_REASONING + _BUDGET, read via reasoningFromEnv):
82
86
  // { mode: off|adaptive|on, budget: iff on }. The provider maps it to the
83
87
  // backend's mechanism via reasoningStyle; budget is only ever a magnitude,
84
- // never a hidden activation flag (#33).
88
+ // never a hidden activation flag.
85
89
  reasoning: Reasoning;
86
90
  // Decode tuning: no in-code defaults; the canonical measured values (0.2 /
87
91
  // 1.15) ship as the floor in .env.defaults (alias-scopable). `temperature` is the
88
- // DEFAULT for EVERY request, spread UNDER caller sampling (#30/endpoint#7).
92
+ // DEFAULT for EVERY request, spread UNDER caller sampling.
89
93
  // `repeatPenalty` is the FLOOR the provider manages wherever a grammar rides
90
- // (greedy-under-mask loops without it, #9) — the VALUE is operator config;
94
+ // (greedy-under-mask loops without it) — the VALUE is operator config;
91
95
  // WHERE it applies stays mechanism.
92
96
  temperature: number;
93
97
  repeatPenalty: number;
94
- // #426: anti-degeneration guard on the CLOUD path (grammarStyle "none"), where the
98
+ // Anti-degeneration guard on the cloud path (grammarStyle "none"), where the
95
99
  // repeat_penalty multiplier isn't available - the OpenAI-standard frequency_penalty.
96
100
  // Optional (default 0 = off) so an out-of-date plugin that omits it just runs unguarded
97
101
  // rather than failing construction; the standard factory always supplies it.
98
102
  frequencyPenalty?: number;
99
- // #567: llama.cpp anti-repetition-LOOP controls, sent on the llamacpp path (DRY is a
100
- // llama.cpp sampler). DRY penalizes repeated SEQUENCES with a penalty escalating in run
101
- // length — the tool for a plan-restart loop a single-token repeat_penalty over a short
102
- // window can't see. The GENERIC DRY defaults (0.8/1.75/2) ship as a floor in .env.defaults
103
- // like repeatPenalty — applied for a detected llama.cpp backend, customer-overridable;
104
- // absent (a plugin omitting them) = the box's own default. repeatLastN widens the window.
103
+ // Optional llama.cpp anti-repetition controls. DRY can suppress long
104
+ // repeated sequences, but it can also corrupt exact repetition required by
105
+ // PLURNK operations. The portable default is off; these fields ride only
106
+ // after an explicit operator opt-in. repeatLastN widens the repeat window.
105
107
  dryMultiplier?: number;
106
108
  dryBase?: number;
107
109
  dryAllowedLength?: number;
108
110
  repeatLastN?: number;
109
111
  // Transient-failure retry budget — REQUIRED, no in-code default
110
112
  // (PLURNK_PROVIDERS_RETRY_ATTEMPTS, a non-negative int): 0 = surface the
111
- // first failure; N = up to N retries on a transient error (§4, #18).
113
+ // first failure; N = up to N retries on a transient error
114
+ // ({§provider-failure-normalization}).
112
115
  retryAttempts: number;
113
- // Data-capture knobs (#36), OFF by default — the flag IS the isolation, so a
116
+ // Maximum characters retained from an upstream diagnostic in the public
117
+ // ProviderError Problem. Standard factories supply the env-owned value;
118
+ // direct construction may omit it to preserve the complete diagnostic.
119
+ errorDetailLimit?: number;
120
+ // Data-capture knobs ({§provider-evidence}), off by default — the flag is the isolation, so a
114
121
  // serving turn requests nothing and carries nothing. `topLogprobs`: when a
115
122
  // non-negative int, request `logprobs:true, top_logprobs:<n>` and surface the
116
123
  // per-token confidence on assistant.logprobs (PLURNK_PROVIDERS_TOP_LOGPROBS;
@@ -119,7 +126,7 @@ export type AiSdkProviderConfig = {
119
126
  // gated per-alias.
120
127
  topLogprobs?: number | null;
121
128
  rawBody?: boolean;
122
- // #507 (owner-ruled): the generation-envelope reserves, env-read via
129
+ // {§provider-generation-envelope} The generation-envelope reserves, env-read via
123
130
  // envelopeFromEnv — a percentage of the DETECTED window or an absolute token
124
131
  // count. Optional so an out-of-date sibling keeps constructing (no claim);
125
132
  // the standard factory always supplies them. Resolved against contextWindow
@@ -127,13 +134,13 @@ export type AiSdkProviderConfig = {
127
134
  // derives correctly.
128
135
  reasoningReserve?: ReserveSpec;
129
136
  completionReserve?: ReserveSpec;
130
- // #507: the plurnk.ai router owns tuning (SPEC §5) — false suppresses the
137
+ // The plurnk.ai router owns tuning — false suppresses the
131
138
  // client-side temperature/penalty FLOORS on this provider (caller `sampling`
132
139
  // still passes through verbatim). Default true (floors ride).
133
140
  tuningFloors?: boolean;
134
141
  };
135
142
 
136
- // #539: drop trailing occurrences of a server-rendered EOG marker. llama-server
143
+ // Drop trailing occurrences of a server-rendered EOG marker. llama-server
137
144
  // under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
138
145
  // trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
139
146
  // can never eat body content (a body ending in the literal marker isn't producible
@@ -145,6 +152,58 @@ const stripTrailingSpecial = (content: string, marker: string): string => {
145
152
  return out;
146
153
  };
147
154
 
155
+ type TaggedReasoningProjection = {
156
+ readonly content: string;
157
+ readonly reasoning: string;
158
+ readonly projected: boolean;
159
+ readonly contentStart: number;
160
+ };
161
+
162
+ const projectLeadingReasoning = (
163
+ content: string,
164
+ structuredReasoning: string,
165
+ opening: string,
166
+ closing: string,
167
+ ): TaggedReasoningProjection => {
168
+ if (structuredReasoning.length > 0 || !content.startsWith(opening)) {
169
+ return { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
170
+ }
171
+ const closingIndex = content.indexOf(closing, opening.length);
172
+ if (closingIndex === -1) {
173
+ return {
174
+ content: "",
175
+ reasoning: content.slice(opening.length),
176
+ projected: true,
177
+ contentStart: [...content].length,
178
+ };
179
+ }
180
+ const suffixStart = closingIndex + closing.length;
181
+ return {
182
+ content: content.slice(suffixStart),
183
+ reasoning: content.slice(opening.length, closingIndex),
184
+ projected: true,
185
+ contentStart: [...content.slice(0, suffixStart)].length,
186
+ };
187
+ };
188
+
189
+ // {§provider-tagged-reasoning} Only the model-contract position is structural:
190
+ // one exact leading envelope. Parsing after stream assembly keeps SSE and JSON
191
+ // on one path and leaves later literal tags in the visible suffix untouched.
192
+ const projectTaggedReasoning = (
193
+ content: string,
194
+ structuredReasoning: string,
195
+ style: ReasoningResponseStyle,
196
+ ): TaggedReasoningProjection => style === "think-tags"
197
+ ? projectLeadingReasoning(content, structuredReasoning, "<think>", "</think>")
198
+ : { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
199
+
200
+ // llama-server's template reasoning parser can project this leading channel out
201
+ // of the OpenAI-compatible response. Grammar evidence needs the sentence before
202
+ // that lossy projection, so constrained template turns request it verbatim and
203
+ // split the observed enclosure here.
204
+ const projectTemplateReasoning = (content: string): TaggedReasoningProjection =>
205
+ projectLeadingReasoning(content, "", "<|channel>thought\n", "<channel|>");
206
+
148
207
  // Shared budget→effort breakpoints (xai and google had identical copies).
149
208
  export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
150
209
  if (budget <= 1000) return "low";
@@ -152,44 +211,28 @@ export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
152
211
  return "high";
153
212
  };
154
213
 
155
- // chars/2 upper bound (see ./tokenizers.ts) — overcounts safely, never under.
156
- const heuristicTokens = (text: string): number => (text.length === 0 ? 0 : Math.ceil(text.length / 2));
157
-
158
214
  // Body keys the provider owns — a caller's `sampling` passthrough may not set
159
- // these. Two families (#477 audit):
215
+ // these. Two families:
160
216
  // transport/managed — grammar transport, the stream/JSON choice, slot pinning,
161
- // data capture (SPEC §8: backend-specific fields never cross the contract);
217
+ // data capture ({§provider-evidence}: backend-specific fields never cross the contract);
162
218
  // contract invariants — `n` (atomic single completion: choices[0] is the
163
219
  // response; n>1 = paid, dropped output), the tool-calling family (tools-in-
164
220
  // body doctrine, §2: native tool_calls return null content = a broken turn),
165
221
  // modalities/audio (text-only contract), prediction (decode semantics, not
166
222
  // sampling), and the token caps (the envelope is the managed maxTokens —
167
- // sampling must not bypass the consumer's #425 cap).
223
+ // sampling must not bypass the consumer's cap).
168
224
  // Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
169
225
  // platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
170
226
  // metadata, store, verbosity) pass through; the managed floors spread UNDER
171
227
  // sampling stay deliberately caller-overridable.
172
228
  const RESERVED_BODY_KEYS: ReadonlySet<string> = new Set([
173
229
  "model", "messages", "stream", "stream_options", "grammar", "response_format", "id_slot", "logprobs", "top_logprobs",
230
+ "reasoning_format", "reasoning_effort", "thinking", "think", "include_reasoning", "chat_template_kwargs", "thinking_budget_tokens", // lexicon-allow: backend wire fields
174
231
  "n", "tools", "tool_choice", "functions", "function_call", "parallel_tool_calls",
175
232
  "modalities", "audio", "prediction", "max_tokens", "max_completion_tokens",
176
233
  "prompt_cache_key",
177
234
  ]);
178
235
 
179
- // Render a non-accept verdict into a terse, factual grammar_unenforced message
180
- // (SPEC §12 message policy: no guidance prose). `reject` names the diverging code
181
- // point + what the grammar would have accepted; `incomplete` names the valid-prefix
182
- // length that never reached a terminal state.
183
- const describeUnenforced = (v: Exclude<Verdict, { status: "accept" }>): string => {
184
- if (v.status === "reject") {
185
- const expected = v.expected.length > 0
186
- ? v.expected.map((e) => `${e.rule} accepts ${e.accepts}`).join(", ")
187
- : "end of input";
188
- return `grammar not enforced: output rejected by the transported grammar at code point ${v.pos} (${JSON.stringify(v.char)}); expected ${expected}`;
189
- }
190
- return `grammar not enforced: output is an incomplete match of the transported grammar — a valid prefix of ${v.pos} code points that never terminated`;
191
- };
192
-
193
236
  export default class AiSdkProvider implements Provider {
194
237
  #model: string;
195
238
  #url: string | undefined;
@@ -211,8 +254,12 @@ export default class AiSdkProvider implements Provider {
211
254
  #dryAllowedLength: number | undefined;
212
255
  #repeatLastN: number | undefined;
213
256
  #reasoningStyle: ReasoningStyle;
214
- #countTokens: (text: string) => number;
257
+ #reasoningResponseStyle: ReasoningResponseStyle;
258
+ #countPromptTokens: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
259
+ #promptTokensUrl: string | undefined;
215
260
  #calculateCost: (usage: ProviderUsage) => number;
261
+ #calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }>;
262
+ #normalizeCharge?: AuthoritativeChargeNormalizer;
216
263
  #source: string;
217
264
  #grammarStyle: GrammarStyle;
218
265
  #promptCacheKey: boolean;
@@ -223,6 +270,7 @@ export default class AiSdkProvider implements Provider {
223
270
  #supportsSlotPinning: boolean;
224
271
  #slotCount: number | null;
225
272
  #retryAttempts: number;
273
+ #errorDetailLimit: number | undefined;
226
274
  #topLogprobs: number | null;
227
275
  #reasoningReserve: ReserveSpec | undefined;
228
276
  #completionReserve: ReserveSpec | undefined;
@@ -230,17 +278,18 @@ export default class AiSdkProvider implements Provider {
230
278
  #rawBody: boolean;
231
279
  #servedModel: string | undefined;
232
280
  #requiresMaxTokens: boolean | undefined;
281
+ readonly attributions?: (context: PluginAttributionContext) => PluginAttribution;
233
282
 
234
- // Optional capability (SPEC §2): exact tokenization served by the backend's
283
+ // Optional capability ({§provider-local-capabilities}): exact tokenization served by the backend's
235
284
  // own vocab. Assigned in the constructor ONLY when the config carries a
236
285
  // tokenizeUrl (llama-server), so `provider.tokenize === undefined` remains
237
286
  // the honest capability signal for every other backend.
238
287
  tokenize?: (text: string) => Promise<number[]>;
239
-
240
288
  constructor(config: AiSdkProviderConfig) {
241
289
  this.#model = config.model;
242
290
  this.#url = config.url;
243
291
  this.#languageModel = config.languageModel;
292
+ this.attributions = config.attributions;
244
293
  if ((this.#url === undefined) === (this.#languageModel === undefined)) {
245
294
  throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
246
295
  }
@@ -264,9 +313,17 @@ export default class AiSdkProvider implements Provider {
264
313
  this.#dryAllowedLength = config.dryAllowedLength;
265
314
  this.#repeatLastN = config.repeatLastN;
266
315
  this.#retryAttempts = config.retryAttempts;
316
+ this.#errorDetailLimit = config.errorDetailLimit;
267
317
  this.#reasoningStyle = config.reasoningStyle ?? "none";
268
- this.#countTokens = config.countTokens ?? heuristicTokens;
318
+ this.#reasoningResponseStyle = config.reasoningResponseStyle ?? "verbatim";
319
+ if (config.countPromptTokens !== undefined && config.promptTokensUrl !== undefined) {
320
+ throw new Error(`${config.source ?? "provider"}: configure countPromptTokens or promptTokensUrl, not both`);
321
+ }
322
+ this.#countPromptTokens = config.countPromptTokens ?? ((messages) => estimatePromptTokens(messages));
323
+ this.#promptTokensUrl = config.promptTokensUrl;
269
324
  this.#calculateCost = config.calculateCost ?? (() => 0);
325
+ this.#calculateCharge = config.calculateCharge;
326
+ this.#normalizeCharge = config.normalizeCharge;
270
327
  this.#source = config.source ?? "provider";
271
328
  this.#grammarStyle = config.grammarStyle ?? "none";
272
329
  this.#promptCacheKey = config.promptCacheKey ?? false;
@@ -286,6 +343,13 @@ export default class AiSdkProvider implements Provider {
286
343
  this.#rawBody = config.rawBody ?? false;
287
344
  this.#servedModel = config.servedModel;
288
345
  this.#requiresMaxTokens = config.requiresMaxTokens;
346
+ const reasoningReserve = this.reasoningReserve;
347
+ if (this.#reasoningStyle === "template"
348
+ && this.#reasoning.mode === "on"
349
+ && reasoningReserve !== null
350
+ && this.#reasoning.budget! > reasoningReserve) {
351
+ throw new Error(`${this.#source}: PLURNK_PROVIDERS_REASONING_BUDGET (${this.#reasoning.budget}) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE (${reasoningReserve})`);
352
+ }
289
353
  const { tokenizeUrl } = config;
290
354
  if (tokenizeUrl !== undefined) {
291
355
  this.tokenize = async (text: string): Promise<number[]> => {
@@ -306,7 +370,7 @@ export default class AiSdkProvider implements Provider {
306
370
  }
307
371
 
308
372
  get contextWindow(): number | null { return this.#contextWindow; }
309
- // #507: envelope reserves — absolute pins stand alone; percentages need the
373
+ // {§provider-generation-envelope} Envelope reserves — absolute pins stand alone; percentages need the
310
374
  // detected window; null = underivable (no claim for core's no-cap path).
311
375
  #resolveReserve(spec: ReserveSpec | undefined): number | null {
312
376
  if (spec === undefined) return null;
@@ -316,63 +380,110 @@ export default class AiSdkProvider implements Provider {
316
380
  get reasoningReserve(): number | null { return this.#resolveReserve(this.#reasoningReserve); }
317
381
  get completionReserve(): number | null { return this.#resolveReserve(this.#completionReserve); }
318
382
  get model(): string { return this.#model; }
319
- // #37: backend's self-reported served id; undefined when unprobed/unknown.
383
+ // Backend's self-reported served id; undefined when unprobed/unknown.
320
384
  get servedModel(): string | undefined { return this.#servedModel; }
321
- // #43: resolved "decodes unbounded without a cap" fact; undefined = no claim.
385
+ // Resolved "decodes unbounded without a cap" fact; undefined = no claim.
322
386
  get requiresMaxTokens(): boolean | undefined { return this.#requiresMaxTokens; }
323
- // Resolved capability (#34): will a transported grammar actually constrain
387
+ // Resolved capability: will a transported grammar actually constrain
324
388
  // this backend's decode? Introspectable so a consumer can verify the rails
325
389
  // are LIVE without spending a generation on a forcing-grammar probe.
326
390
  get constrainsOutput(): boolean { return this.#grammarStyle !== "none"; }
327
391
 
328
- countTokens(text: string): number { return this.#countTokens(text); }
392
+ async countPromptTokens(
393
+ messages: readonly ChatMessage[],
394
+ signal?: AbortSignal,
395
+ ): Promise<PromptTokenMeasurement> {
396
+ if (this.#promptTokensUrl === undefined) {
397
+ return assertPromptTokenMeasurement(
398
+ await this.#countPromptTokens(messages, signal),
399
+ this.#source,
400
+ );
401
+ }
402
+
403
+ signal?.throwIfAborted();
404
+ try {
405
+ const timeout = AbortSignal.timeout(this.#fetchTimeoutMs);
406
+ const response = await this.#fetch(this.#promptTokensUrl, {
407
+ method: "POST",
408
+ headers: { "Content-Type": "application/json", ...this.#headers },
409
+ body: JSON.stringify({
410
+ model: this.#model,
411
+ messages,
412
+ ...this.#reasoningBody(),
413
+ }),
414
+ signal: signal === undefined ? timeout : AbortSignal.any([signal, timeout]),
415
+ });
416
+ if (!response.ok) {
417
+ return estimatePromptTokens(
418
+ messages,
419
+ `llama-server input-token endpoint returned HTTP ${response.status}`,
420
+ );
421
+ }
422
+ const body = await response.json() as { input_tokens?: unknown };
423
+ if (!Number.isInteger(body.input_tokens) || (body.input_tokens as number) < 0) {
424
+ return estimatePromptTokens(
425
+ messages,
426
+ "llama-server input-token endpoint returned no non-negative integer input_tokens",
427
+ );
428
+ }
429
+ return {
430
+ kind: "exact",
431
+ tokens: body.input_tokens as number,
432
+ source: "llama-server:/v1/chat/completions/input_tokens",
433
+ };
434
+ } catch (cause) {
435
+ signal?.throwIfAborted();
436
+ return estimatePromptTokens(
437
+ messages,
438
+ `llama-server input-token measurement failed: ${cause instanceof Error ? cause.message : String(cause)}`,
439
+ );
440
+ }
441
+ }
329
442
  calculateCost(usage: ProviderUsage): number { return this.#calculateCost(usage); }
443
+ calculateCharge(usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }> {
444
+ return this.#calculateCharge?.(usage)
445
+ ?? { kind: "unknown", reason: "no provider rate or settled charge is available" };
446
+ }
330
447
 
331
- // Maps the reasoning INTENT (off | adaptive | on+budget, #33) to the
332
- // backend's wire mechanism — including under a transported grammar. The #32
333
- // clamp (force reasoning_effort "none" under response_format) is LIFTED:
334
- // canary-verified live that fireworks masks ONLY the content channel — the
335
- // reasoning channel rides beside it unmasked, and the plurnk grammar's
336
- // reasoning?/preplan regions absorb any in-band spillover. The old measured
337
- // failures (low→cycles, high→spirals) were pre-max_tokens-cap; with a bounded
338
- // cap the matrix ACCEPTs across efforts (reasoning-rails matrix
339
- // F9). Clamping was the root of the plan-less regression (service#331).
340
- #reasoningBody(): Record<string, unknown> {
448
+ // Reasoning activation and allowance are independent of grammar transport;
449
+ // only the response representation becomes lossless when evidence is needed.
450
+ // The llama-server template mapping is owned by {§llama-reasoning-request}.
451
+ #reasoningBody(preserveGrammarSentence = false): Record<string, unknown> {
341
452
  const { mode, budget } = this.#reasoning;
342
453
  const on = mode !== "off";
343
454
  switch (this.#reasoningStyle) {
344
- // Native-channel styles. "template" ALWAYS emits — the explicit
345
- // enable_thinking:false is the only working off-switch on llama-server
346
- // (§13). Activation only; budget is enforced by the box's
347
- // --reasoning-budget launch flag (per-request numerics ignored, F7).
348
- //
349
- // #488 postmortem: intent maps IDENTICALLY under a transported
350
- // grammar. The brief rails-win-the-channel clamp (enable_thinking
351
- // forced false under a grammar) is REVERTED — specimens proved the
352
- // SANCTIONED think block is the protection, not the hazard: the
353
- // server auto-gates the grammar around it and content decodes
354
- // constrained (26-run baseline green; zero grammar rejects across
355
- // the #488 "railless" specimens). Closing the channel starved a
356
- // reasoning-tuned model into ESCAPING mid-content into the raw
357
- // thought channel — discarded server-side, decode unconstrained,
358
- // 12,288 tokens billed for 1,033 visible chars. The escape is
359
- // surfaced instead (vanished-token telemetry + meta rail state).
360
- case "template": return { chat_template_kwargs: { enable_thinking: on } };
455
+ case "template": {
456
+ const allowance = mode === "off"
457
+ ? 0
458
+ : mode === "on" ? budget : this.reasoningReserve;
459
+ return {
460
+ chat_template_kwargs: { enable_thinking: on },
461
+ reasoning_format: preserveGrammarSentence ? "none" : "auto",
462
+ ...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
463
+ };
464
+ }
361
465
  case "think": return on ? { think: true } : {};
362
466
  case "include_reasoning": return on ? { include_reasoning: true } : {};
363
467
  // effort tiers from the budget; off/adaptive omit the field (the
364
468
  // API's default depth is its adaptive).
365
469
  case "effort": return mode === "on" ? { reasoning_effort: effortFromBudget(budget!) } : {};
366
470
  // Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
367
- // reason-by-default model (DeepSeek V4: default 'high') reasoning (#30).
471
+ // reason-by-default model (DeepSeek V4: default 'high') reasoning.
368
472
  // ADAPTIVE omits the field: the backend's own default posture IS the
369
473
  // adaptive semantics, and the literal "adaptive" is MiniMax-M3-only —
370
- // fireworks 400s it for every other model (wire-verified, #403; the
474
+ // Fireworks 400s it for every other model (wire-verified; the
371
475
  // 1.0.2 adaptive default refused to boot on it). V4 gotcha: integer
372
476
  // efforts 400.
373
477
  case "effort_explicit": return mode === "off"
374
478
  ? { reasoning_effort: "none" }
375
479
  : mode === "on" ? { reasoning_effort: effortFromBudget(budget!) } : {};
480
+ // {§deepseek-reasoning-request}
481
+ case "thinking_effort": return mode === "off"
482
+ ? { thinking: { type: "disabled" } }
483
+ : mode === "on" ? {
484
+ thinking: { type: "enabled" },
485
+ reasoning_effort: effortFromBudget(budget!),
486
+ } : {};
376
487
  // Anthropic compat: explicit thinking object. off → disabled; on →
377
488
  // enabled with budget_tokens; adaptive → omit (the API default).
378
489
  case "anthropic": return mode === "off"
@@ -382,7 +493,7 @@ export default class AiSdkProvider implements Provider {
382
493
  }
383
494
  }
384
495
 
385
- // Per-run slot affinity (#11): the consumer passes WHICH run this is; the
496
+ // Per-worker slot affinity: the consumer passes which worker this is; the
386
497
  // provider owns WHICH slot serves it. Sticky per workerId, round-robin across
387
498
  // new runs (distinct runs → distinct slots while slots last), LRU-bounded
388
499
  // bookkeeping so a long-lived daemon never grows the map unboundedly —
@@ -405,19 +516,19 @@ export default class AiSdkProvider implements Provider {
405
516
  return { id_slot: slot };
406
517
  }
407
518
 
408
- // Optional local llama-server GBNF transport (SPEC §13). Unsupported
519
+ // Optional local llama-server GBNF transport ({§gbnf-response-observation}). Unsupported
409
520
  // backends receive no grammar-related field.
410
521
  #grammarBody(grammar: string | undefined): Record<string, unknown> {
411
522
  if (grammar === undefined) return {};
412
523
  switch (this.#grammarStyle) {
413
524
  // Greedy decoding under hard constraint loops without a repeat-penalty
414
- // floor (#9, SPEC §13) — llama.cpp spells it `repeat_penalty`.
525
+ // floor — llama.cpp spells it `repeat_penalty`.
415
526
  case "llamacpp": return { grammar, repeat_penalty: this.#repeatPenalty };
416
527
  case "none": return {};
417
528
  }
418
529
  }
419
530
 
420
- // Anti-degeneration DEFAULT on EVERY request (#426), keyed to the backend's wire
531
+ // Anti-degeneration default on every request, keyed to the backend's wire
421
532
  // convention - NOT grammar-bound. GBNF is a local constraint, so a cloud
422
533
  // alias runs the sampler bare: firefast (deepseek/fireworks) ran 4/86 bench turns
423
534
  // straight to the token cap on pure looped repetition (run52). Ships next to
@@ -428,7 +539,7 @@ export default class AiSdkProvider implements Provider {
428
539
  // live: together/deepinfra/fireworks; it is OpenAI's own param, so real OpenAI takes it too).
429
540
  #repetitionPenaltyBody(): Record<string, unknown> {
430
541
  switch (this.#grammarStyle) {
431
- // #567: repeat_penalty + optional DRY (repeated-SEQUENCE penalty) + a wider
542
+ // repeat_penalty + optional DRY (repeated-sequence penalty) + a wider
432
543
  // repeat_last_n window — the loop-breaking tools a llama.cpp backend serves.
433
544
  // Each rides only when its operator knob is set; absent = the box's default.
434
545
  case "llamacpp": return {
@@ -444,12 +555,12 @@ export default class AiSdkProvider implements Provider {
444
555
  }
445
556
  }
446
557
 
447
- // First-party telemetry headers (SPEC §5): forwarded ONLY when the spec
558
+ // First-party telemetry headers ({§provider-request-authority}): forwarded only when the spec
448
559
  // opted in (the plurnk endpoint). The gate is here, not at the call site, so
449
560
  // attributions/client/strikes can never reach a third-party backend even if
450
561
  // the consumer passes them to the wrong provider. Empty values emit no header
451
562
  // — EXCEPT strikes, where 0 is a real value (clean streak) distinct from
452
- // absent (consumer didn't report); contract per plurnk-service#313. Strikes
563
+ // absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
453
564
  // ride HTTP headers only — the packet never carries them (the model must
454
565
  // never see strike state; engine accounting is not a metric to game).
455
566
  #metadataHeaders(attributions: string[] | undefined, client: string | undefined, strikes: number | undefined, workerId: string, primaryWorkerId: string | undefined, workspaceId: string | undefined, loop: number | undefined, turn: number | undefined): Record<string, string> {
@@ -458,11 +569,11 @@ export default class AiSdkProvider implements Provider {
458
569
  if (attributions !== undefined && attributions.length > 0) h["Plurnk-Attribution"] = JSON.stringify(attributions);
459
570
  if (client !== undefined && client.length > 0) h["Plurnk-Client"] = client;
460
571
  if (strikes !== undefined && Number.isInteger(strikes) && strikes >= 0) h["Plurnk-Strikes"] = String(strikes);
461
- // Worker identity (#26, wire-name completed #486/#511): the opaque workerId
572
+ // Worker identity: the opaque workerId
462
573
  // the consumer already supplies, forwarded so the endpoint can key
463
574
  // per-worker affinity/telemetry — same gate as every first-party signal.
464
575
  h["Plurnk-Worker-Id"] = workerId;
465
- // Root worker of the lineage (#522): the no-parent ancestor of this turn's
576
+ // Root worker of the lineage ({§worker-primary}): the no-parent ancestor of this turn's
466
577
  // worker tree. The consumer classifies primary-vs-spawned by equality
467
578
  // (primaryWorkerId == workerId ⇒ the primary/root worker). The provider
468
579
  // EMITS what the consumer supplies and never invents a primary; the
@@ -470,7 +581,7 @@ export default class AiSdkProvider implements Provider {
470
581
  // own, where it equals workerId). Absence is the consumer's violation for
471
582
  // the endpoint to surface, not a provider default.
472
583
  if (primaryWorkerId !== undefined && primaryWorkerId.length > 0) h["Plurnk-Worker-Primary"] = primaryWorkerId;
473
- // Turn coordinate (#404, extends #26 per #391): workspace/loop/turn, the
584
+ // Turn coordinate ({§lifecycle-terms}): workspace/loop/turn, the
474
585
  // daemon-side sequence the endpoint can never scrape from the wire.
475
586
  // Coordinates are 1-based — 0 is not a real value, so no strikes-style
476
587
  // zero exception; absent/empty/0 emits no header.
@@ -480,33 +591,7 @@ export default class AiSdkProvider implements Provider {
480
591
  return h;
481
592
  }
482
593
 
483
- // Enforcement verification (SPEC §13). When a grammar was actually transported
484
- // (grammarStyle !== "none"), the backend MUST have constrained the output;
485
- // some silently drop the grammar field or mislabel the channel, and without
486
- // this check we would return unconstrained output as if enforced. STRICT: any
487
- // non-accept verdict (reject, or an incomplete/never-terminated match) is a
488
- // grammar_unenforced failure. A grammar our own validator can't parse — even
489
- // though the backend accepted it (a port-vs-llama.cpp gap) — is a non-fatal
490
- // verify gap: warn, don't fail a transport that may have worked. This is a
491
- // conformance check against the grammar we already hold, NOT a plurnk-DSL
492
- // parse (§8) — it stays grammar-generic and backend-agnostic.
493
- // Validate output against the grammar. Returns the verdict, or null on the
494
- // verify GAP — a grammar our own validator can't parse (a port-vs-llama.cpp
495
- // gap): warn, don't manufacture a conflict from a check that didn't run.
496
- #grammarVerdict(grammar: string, content: string): Verdict | null {
497
- try {
498
- return validateGbnf(grammar, content);
499
- } catch (cause) {
500
- // Once per (code, message) — #40: this fires PER TURN otherwise.
501
- emitWarningOnce(
502
- `${this.#source}: could not verify grammar enforcement — the transported grammar did not parse in @plurnk/gbnf (${(cause as Error).message})`,
503
- "PLURNK_GRAMMAR_UNVERIFIABLE",
504
- );
505
- return null;
506
- }
507
- }
508
-
509
- // PLURNK_PROVIDERS_GBNF_DEBUG (SPEC §13): validate the supplied GBNF locally and fail
594
+ // PLURNK_PROVIDERS_GBNF_DEBUG ({§gbnf-response-observation}): validate the supplied GBNF locally and fail
510
595
  // hard if it's malformed, BEFORE any wire call — and the grammar is NOT
511
596
  // transported, so the request runs unconstrained. A debug aid to catch invalid
512
597
  // grammars (e.g. while editing the plurnk grammar) without a model round-trip;
@@ -521,7 +606,7 @@ export default class AiSdkProvider implements Provider {
521
606
  }
522
607
  }
523
608
 
524
- // Per-turn metadata bag (#23): pass the backend's non-standard top-level fields
609
+ // Per-turn metadata bag: pass the backend's non-standard top-level fields
525
610
  // through verbatim. Providers do not reinterpret vendor currency or account
526
611
  // metadata; a monetary value carries its own amount and currency.
527
612
  #buildMeta(chunkMetadata: Record<string, unknown>): Record<string, unknown> | undefined {
@@ -533,7 +618,8 @@ export default class AiSdkProvider implements Provider {
533
618
  // penalties, stop, seed, …) merged UNDER the managed body: model, messages,
534
619
  // reasoning, grammar (+ its repeat-penalty floor), max_tokens and slot always
535
620
  // win, and reserved transport/protocol keys are stripped so the passthrough
536
- // can't smuggle a grammar, a stream toggle, or a backend slot (SPEC §8).
621
+ // can't smuggle a grammar, a stream toggle, or a backend slot
622
+ // ({§provider-request-authority}).
537
623
  #samplingBody(sampling: Record<string, unknown> | undefined): Record<string, unknown> {
538
624
  if (sampling === undefined) return {};
539
625
  const out: Record<string, unknown> = {};
@@ -542,41 +628,40 @@ export default class AiSdkProvider implements Provider {
542
628
  }
543
629
 
544
630
  async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }: { messages: ChatMessage[]; workerId: string; primaryWorkerId?: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse> {
545
- // Boundary validation (SPEC §2): the worker identity is required.
631
+ // {§provider-interface} The worker identity is required.
546
632
  if (workerId === undefined || workerId.length === 0) throw new Error("generate: workerId is required — the worker's stable, opaque identity");
547
- // Reject before any wire call when already aborted (SPEC §10.8).
633
+ // Reject before any wire call when already aborted
634
+ // ({§provider-failure-normalization}).
548
635
  signal?.throwIfAborted();
549
636
 
550
- // Grammar handling (SPEC §13). PLURNK_PROVIDERS_GBNF_DEBUG validates the supplied
551
- // grammar locally and throws on a malformed one, then WITHHOLDS it so the
552
- // model generates UNCONSTRAINED — and the free output is still verified
553
- // against the grammar (below), surfacing exactly where the model's natural
554
- // output and the grammar conflict. Otherwise the grammar is sent when the
555
- // backend supports it (grammarStyle !== "none").
637
+ // Grammar handling ({§gbnf-response-observation}). Debug validates the
638
+ // supplied grammar before the call but withholds it from the backend.
556
639
  const wantGrammar = grammar !== undefined && this.#grammarStyle !== "none";
557
640
  if (wantGrammar && this.#gbnfDebug) this.#assertGrammarValid(grammar!);
558
641
  const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
642
+ const preserveGrammarSentence = wantGrammar
643
+ && this.#reasoningStyle === "template";
559
644
 
560
645
  // Assembly order = precedence: the family's sampling DEFAULTS
561
- // (PLURNK_PROVIDERS_TEMPERATURE — universal, #30 measured it on grammar
646
+ // (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
562
647
  // paths and the name promises every request) < the caller's `sampling`
563
648
  // < the managed fields, which always win.
564
649
  const body: Record<string, unknown> = {
565
- // #507: floors suppressed on router-owned-tuning providers (plurnk) —
650
+ // Floors are suppressed on router-owned-tuning providers (plurnk) —
566
651
  // the router's per-model tuning must not be overridden by client floors.
567
652
  ...(this.#tuningFloors ? { temperature: this.#temperature, ...this.#repetitionPenaltyBody() } : {}),
568
653
  ...this.#samplingBody(sampling),
569
654
  ...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
570
655
  model: this.#model,
571
656
  messages,
572
- ...this.#reasoningBody(),
657
+ ...this.#reasoningBody(preserveGrammarSentence),
573
658
  ...this.#grammarBody(sendGrammar),
574
659
  ...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
575
- // #36: request per-token logprobs only when enabled (managed field —
660
+ // Request per-token logprobs only when enabled (managed field —
576
661
  // reserved from caller sampling; the env flag is the single control).
577
662
  ...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
578
663
  ...this.#slotBody(workerId),
579
- // #518: prompt-cache affinity -- workerId as the OpenAI-standard
664
+ // Prompt-cache affinity -- workerId as the OpenAI-standard
580
665
  // prompt_cache_key routes a worker's turns to one serverless replica so
581
666
  // its stable prefix caches (managed; reserved from caller sampling).
582
667
  ...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
@@ -584,7 +669,9 @@ export default class AiSdkProvider implements Provider {
584
669
 
585
670
  // Per-request headers = static auth/routing + any first-party telemetry.
586
671
  const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
587
- const headers = Object.keys(metaHeaders).length > 0 ? { ...this.#headers, ...metaHeaders } : this.#headers;
672
+ const headers = Object.keys(metaHeaders).length === 0
673
+ ? this.#headers
674
+ : { ...this.#headers, ...metaHeaders };
588
675
  let raw;
589
676
  try {
590
677
  raw = this.#languageModel === undefined
@@ -636,14 +723,14 @@ export default class AiSdkProvider implements Provider {
636
723
  });
637
724
  } catch (err) {
638
725
  if (signal?.aborted) throw err;
639
- const pe = toProviderError(err, this.#source);
726
+ const pe = toProviderError(err, this.#source, this.#errorDetailLimit);
640
727
  if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
641
728
  throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
642
729
  }
643
730
  throw pe;
644
731
  }
645
732
 
646
- // #539: llama-server --special renders EOG tokens as text, so a turn ending
733
+ // llama-server --special renders EOG tokens as text, so a turn ending
647
734
  // via raw EOS carries a trailing <eos> the grammar never sanctioned - it both
648
735
  // false-rejects the rail verdict and leaks a control token into the packet.
649
736
  // Strip the server-reported eos_token from the tail ONCE, before the verdict
@@ -651,66 +738,133 @@ export default class AiSdkProvider implements Provider {
651
738
  // wire text for forensics.
652
739
  if (this.#eosText !== undefined) raw.content = stripTrailingSpecial(raw.content, this.#eosText);
653
740
 
654
- // Grammar conformance (§13): bytes always flow; the verdict is an
655
- // observation. Same check whether the grammar was transported
656
- // (sendGrammar) or withheld (PLURNK_PROVIDERS_GBNF_DEBUG filter mode) — a
657
- // non-accept verdict attaches a grammar_unenforced telemetry event
658
- // (message + divergence position) and the response returns normally.
659
- // Discard/retry/escalate/self-correct is the consumer's policy.
660
- let telemetry: TelemetryEvent[] | undefined;
661
- let railsMeta: Record<string, unknown> | undefined;
662
- const usage = raw.usage;
663
- const observedGrammar = sendGrammar ?? (wantGrammar && this.#gbnfDebug ? grammar : undefined);
664
- if (observedGrammar !== undefined) {
665
- const verdict = this.#grammarVerdict(observedGrammar, raw.content);
666
- if (verdict !== null && verdict.status !== "accept") {
667
- telemetry = [{ source: this.#source, kind: "grammar_unenforced", message: describeUnenforced(verdict), position: verdict.pos }];
741
+ const grammarInput = raw.content;
742
+ const projectedReasoning = preserveGrammarSentence && !raw.reasoningProjected
743
+ ? projectTemplateReasoning(raw.content)
744
+ : projectTaggedReasoning(
745
+ raw.content,
746
+ raw.reasoning,
747
+ this.#reasoningResponseStyle,
748
+ );
749
+
750
+ // Preserve the exact sentence seen at the grammar boundary. Constrained
751
+ // template turns request `reasoning_format: "none"`, so even an empty
752
+ // channel remains observable. An unexpectedly projected response cannot
753
+ // supply independent pre-projection evidence.
754
+ let grammarEvidence: GrammarEvidence | undefined;
755
+ if (wantGrammar) {
756
+ if (preserveGrammarSentence) {
757
+ if (!raw.reasoningProjected) {
758
+ grammarEvidence = {
759
+ input: grammarInput,
760
+ contentStart: projectedReasoning.projected ? projectedReasoning.contentStart : 0,
761
+ transported: sendGrammar !== undefined,
762
+ };
763
+ }
764
+ } else if (projectedReasoning.projected) {
765
+ grammarEvidence = {
766
+ input: grammarInput,
767
+ contentStart: projectedReasoning.contentStart,
768
+ transported: sendGrammar !== undefined,
769
+ };
770
+ } else {
771
+ grammarEvidence = {
772
+ input: grammarInput,
773
+ contentStart: 0,
774
+ transported: sendGrammar !== undefined,
775
+ };
668
776
  }
669
- // #488 per-request loud state: rail attachment + conformance verdict
670
- // ride `meta` into the consumer's turn row, so a drill reads rail
671
- // presence PER TURN from the run db instead of inferring it from
672
- // output shape (the #488 misdiagnosis, twice).
673
- railsMeta = { railsAttached: sendGrammar !== undefined, railsVerdict: verdict?.status ?? "unverifiable" };
674
- // #488 channel-escape detector (the run105 class): completion tokens
777
+ }
778
+
779
+ if (projectedReasoning.projected) {
780
+ raw.content = projectedReasoning.content;
781
+ raw.reasoning = projectedReasoning.reasoning;
782
+ raw.usage = attributeUnitemizedReasoning(raw.usage, raw.reasoning, raw.content);
783
+ }
784
+
785
+ let notices: ProviderNotice[] | undefined;
786
+ const usage = raw.usage;
787
+ if (sendGrammar !== undefined && this.tokenize !== undefined) {
788
+ // Channel-escape detector: completion tokens
675
789
  // billed far beyond every visible channel mean the decode ESCAPED into
676
- // a server-discarded reasoning block mid-emission — unconstrained,
677
- // invisible, billed (12,288 billed vs 1,033 chars visible, live).
678
- // countTokens OVERCOUNTS text (chars/2 upper bound), so billed
679
- // exceeding visible-plus-slack is real vanishing, not estimator noise.
680
- const visible = this.#countTokens(raw.content) + this.#countTokens(raw.reasoning);
681
- if (sendGrammar !== undefined && usage.completion > visible + 64) {
682
- (telemetry ??= []).push({
683
- source: this.#source,
684
- kind: "grammar_unenforced",
685
- message: `decode escaped the grammar: ${usage.completion} completion tokens billed but only ~${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
686
- position: [...raw.content].length,
687
- });
790
+ // a server-discarded reasoning block mid-emission. This diagnostic
791
+ // requires the serving vocabulary; an estimate cannot prove absence.
792
+ try {
793
+ const [contentTokens, reasoningTokens] = await Promise.all([
794
+ this.tokenize(raw.content),
795
+ this.tokenize(raw.reasoning),
796
+ ]);
797
+ const visible = contentTokens.length + reasoningTokens.length;
798
+ if (usage.completion > visible + 64) {
799
+ (notices ??= []).push({
800
+ source: this.#source,
801
+ kind: "grammar_unenforced",
802
+ level: "warn",
803
+ message: `decode escaped the grammar: ${usage.completion} completion tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
804
+ position: [...raw.content].length,
805
+ });
806
+ }
807
+ } catch (cause) {
808
+ emitWarningOnce(
809
+ `${this.#source}: exact visible-token diagnostic unavailable (${cause instanceof Error ? cause.message : String(cause)})`,
810
+ "PLURNK_VISIBLE_TOKEN_COUNT_UNAVAILABLE",
811
+ );
688
812
  }
689
813
  }
690
814
 
691
- const builtMeta = this.#buildMeta(raw.metadata);
692
- const meta = railsMeta !== undefined ? { ...builtMeta, ...railsMeta } : builtMeta;
815
+ const meta = this.#buildMeta(raw.metadata);
693
816
  const logprobs = raw.logprobs.length > 0 ? raw.logprobs : undefined;
694
817
  const meanLogprob = logprobs !== undefined
695
818
  ? logprobs.reduce((sum, token) => sum + token.logprob, 0) / logprobs.length
696
819
  : undefined;
697
820
 
698
- return {
699
- assistant: {
700
- content: raw.content,
701
- reasoning: raw.reasoning.length > 0 ? raw.reasoning : null,
702
- ...(raw.reasoningEncrypted.length > 0
703
- ? { reasoningEncrypted: raw.reasoningEncrypted }
704
- : {}),
705
- usage,
706
- finishReason: raw.finishReason,
707
- model: raw.model,
708
- ...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
709
- },
821
+ const assistant = {
822
+ content: raw.content,
823
+ reasoning: raw.reasoning.length > 0 ? raw.reasoning : null,
824
+ ...(raw.reasoningEncrypted.length > 0
825
+ ? { reasoningEncrypted: raw.reasoningEncrypted }
826
+ : {}),
827
+ usage,
828
+ model: raw.model,
829
+ ...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
830
+ };
831
+ const normalizedCharge = this.#normalizeCharge?.(raw.chargeEvidence);
832
+ const charge = normalizedCharge === undefined
833
+ ? undefined
834
+ : validateAuthoritativeCharge(normalizedCharge);
835
+ const evidence = {
710
836
  assistantRaw: raw,
837
+ ...(charge === undefined ? {} : { charge }),
838
+ ...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
711
839
  ...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
712
840
  ...(meta !== undefined ? { meta } : {}),
713
- ...(telemetry !== undefined ? { telemetry } : {}),
841
+ ...(notices !== undefined ? { notices } : {}),
842
+ };
843
+ if (raw.finishReason === "resource_interrupted") {
844
+ const attempt: ProviderResponse<"resource_interrupted"> = {
845
+ assistant: { ...assistant, finishReason: raw.finishReason },
846
+ ...evidence,
847
+ };
848
+ throw new ProviderError(
849
+ this.#source,
850
+ "resource_interrupted",
851
+ "The provider interrupted generation because inference resources were unavailable.",
852
+ {
853
+ attempt,
854
+ extensions: {
855
+ stage: "provider-response",
856
+ finishReason: "resource_interrupted",
857
+ ...(raw.rawFinishReason === undefined
858
+ ? {}
859
+ : { rawFinishReason: raw.rawFinishReason }),
860
+ },
861
+ },
862
+ );
863
+ }
864
+ return {
865
+ assistant: { ...assistant, finishReason: raw.finishReason },
866
+ ...evidence,
714
867
  };
715
868
  }
869
+
716
870
  }