@plurnk/plurnk-providers 1.3.11 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/.env.defaults +40 -23
  2. package/README.md +65 -4
  3. package/SPEC.md +222 -56
  4. package/dist/AiSdkProvider.d.ts +18 -5
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +240 -153
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +14 -15
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +26 -10
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +8 -2
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +41 -8
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/ProviderRegistry.d.ts +4 -1
  17. package/dist/ProviderRegistry.d.ts.map +1 -1
  18. package/dist/ProviderRegistry.js +7 -3
  19. package/dist/ProviderRegistry.js.map +1 -1
  20. package/dist/aiSdkTransport.d.ts +4 -2
  21. package/dist/aiSdkTransport.d.ts.map +1 -1
  22. package/dist/aiSdkTransport.js +18 -3
  23. package/dist/aiSdkTransport.js.map +1 -1
  24. package/dist/catalogProvider.d.ts +3 -1
  25. package/dist/catalogProvider.d.ts.map +1 -1
  26. package/dist/catalogProvider.js +20 -7
  27. package/dist/catalogProvider.js.map +1 -1
  28. package/dist/compatibleProvider.d.ts.map +1 -1
  29. package/dist/compatibleProvider.js +11 -4
  30. package/dist/compatibleProvider.js.map +1 -1
  31. package/dist/cost.d.ts +11 -0
  32. package/dist/cost.d.ts.map +1 -0
  33. package/dist/cost.js +64 -0
  34. package/dist/cost.js.map +1 -0
  35. package/dist/discover.d.ts +2 -0
  36. package/dist/discover.d.ts.map +1 -1
  37. package/dist/discover.js +15 -9
  38. package/dist/discover.js.map +1 -1
  39. package/dist/env.d.ts +4 -0
  40. package/dist/env.d.ts.map +1 -1
  41. package/dist/env.js +29 -9
  42. package/dist/env.js.map +1 -1
  43. package/dist/errors.d.ts +27 -0
  44. package/dist/errors.d.ts.map +1 -0
  45. package/dist/errors.js +150 -0
  46. package/dist/errors.js.map +1 -0
  47. package/dist/index.d.ts +11 -5
  48. package/dist/index.d.ts.map +1 -1
  49. package/dist/index.js +7 -4
  50. package/dist/index.js.map +1 -1
  51. package/dist/notices.d.ts +10 -0
  52. package/dist/notices.d.ts.map +1 -0
  53. package/dist/notices.js +11 -0
  54. package/dist/notices.js.map +1 -0
  55. package/dist/ollama.d.ts.map +1 -1
  56. package/dist/ollama.js +3 -3
  57. package/dist/ollama.js.map +1 -1
  58. package/dist/openai.d.ts +1 -1
  59. package/dist/openai.d.ts.map +1 -1
  60. package/dist/promptTokens.d.ts +4 -0
  61. package/dist/promptTokens.d.ts.map +1 -0
  62. package/dist/promptTokens.js +32 -0
  63. package/dist/promptTokens.js.map +1 -0
  64. package/dist/sdkModels.d.ts.map +1 -1
  65. package/dist/sdkModels.js +4 -3
  66. package/dist/sdkModels.js.map +1 -1
  67. package/dist/types.d.ts +43 -16
  68. package/dist/types.d.ts.map +1 -1
  69. package/dist/types.js +1 -1
  70. package/dist/types.js.map +1 -1
  71. package/dist/usage.d.ts +3 -0
  72. package/dist/usage.d.ts.map +1 -1
  73. package/dist/usage.js +26 -14
  74. package/dist/usage.js.map +1 -1
  75. package/dist/warnings.js +0 -0
  76. package/dist/warnings.js.map +1 -1
  77. package/package.json +13 -9
  78. package/src/AiSdkProvider.test.ts +480 -159
  79. package/src/AiSdkProvider.ts +320 -196
  80. package/src/Mock.test.ts +29 -14
  81. package/src/Mock.ts +33 -15
  82. package/src/Pool.test.ts +43 -6
  83. package/src/Pool.ts +56 -10
  84. package/src/ProviderRegistry.test.ts +158 -9
  85. package/src/ProviderRegistry.ts +19 -6
  86. package/src/aiSdkTransport.ts +25 -6
  87. package/src/boundaries.test.ts +8 -3
  88. package/src/catalogProvider.test.ts +17 -0
  89. package/src/catalogProvider.ts +25 -10
  90. package/src/compatibleProvider.test.ts +96 -0
  91. package/src/compatibleProvider.ts +15 -6
  92. package/src/cost.test.ts +63 -0
  93. package/src/cost.ts +83 -0
  94. package/src/defaults.test.ts +1 -0
  95. package/src/discover.test.ts +48 -7
  96. package/src/discover.ts +31 -21
  97. package/src/env.test.ts +38 -23
  98. package/src/env.ts +45 -18
  99. package/src/errors.test.ts +148 -0
  100. package/src/errors.ts +207 -0
  101. package/src/index.ts +29 -7
  102. package/src/lexicon-guard.test.ts +6 -6
  103. package/src/notices.ts +22 -0
  104. package/src/ollama.test.ts +64 -0
  105. package/src/ollama.ts +6 -3
  106. package/src/openai.ts +3 -0
  107. package/src/promptTokens.ts +41 -0
  108. package/src/sdkModels.test.ts +7 -0
  109. package/src/sdkModels.ts +4 -8
  110. package/src/types.ts +106 -64
  111. package/src/usage.test.ts +15 -4
  112. package/src/usage.ts +32 -14
  113. package/src/warnings.test.ts +10 -10
  114. package/src/warnings.ts +0 -0
  115. package/dist/OpenAICompat.d.ts +0 -76
  116. package/dist/OpenAICompat.d.ts.map +0 -1
  117. package/dist/OpenAICompat.js +0 -555
  118. package/dist/OpenAICompat.js.map +0 -1
  119. package/dist/openaiStream.d.ts +0 -47
  120. package/dist/openaiStream.d.ts.map +0 -1
  121. package/dist/openaiStream.js +0 -280
  122. package/dist/openaiStream.js.map +0 -1
  123. package/dist/standardProviders.d.ts +0 -31
  124. package/dist/standardProviders.d.ts.map +0 -1
  125. package/dist/standardProviders.js +0 -518
  126. package/dist/standardProviders.js.map +0 -1
  127. package/dist/telemetry.d.ts +0 -24
  128. package/dist/telemetry.d.ts.map +0 -1
  129. package/dist/telemetry.js +0 -85
  130. package/dist/telemetry.js.map +0 -1
  131. package/src/telemetry.test.ts +0 -69
  132. package/src/telemetry.ts +0 -116
@@ -6,28 +6,24 @@
6
6
  // ordinary vendor protocol. The compatible URL path remains only for PLURNK
7
7
  // extensions and local endpoint probes the SDK cannot represent.
8
8
 
9
- import type { ChatMessage, FinishReason, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
10
- import type { Reasoning, ReserveSpec } from "./env.ts";
9
+ import type { ChatMessage, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
10
+ import type { ProviderCost } from "@plurnk/plurnk-contracts";
11
+ import type { Reasoning, ReasoningResponseStyle, ReserveSpec } from "./env.ts";
11
12
  import { executeAiSdkModel, executeOpenAICompatible } from "./aiSdkTransport.ts";
12
13
  import type { LanguageModel } from "ai";
13
- import { toProviderError, ProviderError, type TelemetryEvent } from "./telemetry.ts";
14
- import { validateGbnf, type Verdict } from "@plurnk/gbnf";
14
+ import { toProviderError, ProviderError } from "./errors.ts";
15
+ import { attributeUnitemizedReasoning } from "./usage.ts";
16
+ import type { ProviderNotice } from "./notices.ts";
17
+ import { validateGbnf } from "@plurnk/gbnf";
18
+ import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.ts";
15
19
  import { emitWarningOnce } from "./warnings.ts";
20
+ import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
16
21
 
17
22
  export type ProviderFetch = typeof globalThis.fetch;
18
23
 
19
- // How the reasoning intent (PLURNK_PROVIDERS_REASONING: off | adaptive | on, plus
20
- // REASONING_BUDGET iff on — #32/#33) translates to each backend's wire mechanism
21
- // (SPEC §4); the per-style mapping lives in #reasoningBody. Non-obvious ones:
22
- // "template" ALWAYS emits enable_thinking — the explicit false is llama-server's
23
- // only working off-switch (§13); "anthropic" uses the `thinking` object and IGNORES
24
- // reasoning_effort; "effort_explicit" (fireworks) sends the EXPLICIT "none" for OFF
25
- // instead of omitting — reason-by-DEFAULT models (DeepSeek V4 defaults
26
- // 'high') keep reasoning when the field is omitted, fatal under an active grammar
27
- // (#30). Intent maps IDENTICALLY with or without a transported grammar — fireworks
28
- // masks only the content channel, so reasoning and rails coexist in one call
29
- // (canary-verified; the #32 clamp is lifted — it caused the plan-less service#331).
30
- export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "template" | "anthropic";
24
+ // Backend wire spellings for the resolved reasoning intent. The switch beside each
25
+ // mapping retains any backend-specific omission/explicit-disable constraint.
26
+ export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "thinking_effort" | "template" | "anthropic";
31
27
 
32
28
  // GBNF transport is a local llama-server capability. "none" means no
33
29
  // service-managed constrained sampling; endpoint-owned settings are not inferred.
@@ -37,17 +33,20 @@ export type AiSdkProviderConfig = {
37
33
  model: string;
38
34
  url?: string; // OpenAI-compatible chat-completions URL
39
35
  languageModel?: LanguageModel; // native AI SDK provider model
36
+ attributions?: (context: PluginAttributionContext) => PluginAttribution;
40
37
  fetchTimeoutMs: number;
41
38
  streamIdleTimeoutMs?: number; // streamed body inter-chunk deadline; zero/unset disables
42
39
  headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
43
40
  fetch?: ProviderFetch; // per-instance request executor; default globalThis.fetch
44
- contextWindow?: number | null; // default null; caller resolves-or-fails (#419), narrows to required with the interface
41
+ contextWindow?: number | null; // default null; caller resolves-or-fails, narrows to required with the interface
45
42
  reasoningStyle?: ReasoningStyle; // default "none"
46
- countTokens?: (text: string) => number; // default chars/2 upper-bound heuristic
43
+ reasoningResponseStyle?: ReasoningResponseStyle; // {§provider-tagged-reasoning}; default "verbatim"
44
+ countPromptTokens?: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
47
45
  calculateCost?: (usage: ProviderUsage) => number; // default () => 0
48
- source?: string; // telemetry source, e.g. "provider:openai"; default "provider"
46
+ calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }>;
47
+ source?: string; // notice/problem source, e.g. "provider:openai"; default "provider"
49
48
  grammarStyle?: GrammarStyle; // how a GBNF grammar is carried; default "none" (not sent)
50
- // #518: send the OpenAI-standard `prompt_cache_key` set to workerId, so a
49
+ // Send the OpenAI-standard `prompt_cache_key` set to workerId, so a
51
50
  // serverless backend with REPLICA-LOCAL prompt caching (fireworks, verified)
52
51
  // pins a worker's turns to one replica and claims its stable prefix. Default
53
52
  // false -- a backend that strict-validates unknown fields 400s, so enable only
@@ -59,21 +58,24 @@ export type AiSdkProviderConfig = {
59
58
  gbnfDebug?: boolean; // PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar locally + throw on invalid, but DON'T transport it (run unconstrained); default false
60
59
  streaming?: boolean; // SSE transport (default true); false → one non-streamed JSON
61
60
  firstPartyMetadata?: boolean; // forward per-turn attributions + client as Plurnk-* headers (plurnk only); default false
62
- apiKeyRejectedMessage?: string; // #537: friendly hint when a PRESENT key is 401/403-rejected (distinct from unset); default undefined
63
- eosText?: string; // #539: server-reported eos_token, stripped from the content tail (--special renders it as text); default undefined
64
- // Slot affinity wiring (provider-INTERNAL — never consumer-facing, #11).
61
+ apiKeyRejectedMessage?: string; // friendly hint when a present key is 401/403-rejected (distinct from unset); default undefined
62
+ eosText?: string; // server-reported eos_token, stripped from the content tail (--special renders it as text); default undefined
63
+ // Slot affinity wiring is provider-internal, never consumer-facing.
65
64
  supportsSlotPinning?: boolean; // backend accepts an `id_slot` body field (llama-server); default false
66
65
  slotCount?: number | null; // probed slot count for pinning backends; default null
67
66
  // Backend-served exact tokenization (llama-server /tokenize). When set, the
68
67
  // provider exposes the optional `tokenize()` capability — the model's OWN
69
68
  // vocab, no client-side tokenizer data needed; default unset (capability absent).
70
69
  tokenizeUrl?: string;
71
- // #37: the backend's self-reported served model id (from the /v1/models probe),
70
+ // Provider-authoritative count of a complete chat-completions request. This
71
+ // is distinct from /tokenize, which sees content but not the chat template.
72
+ promptTokensUrl?: string;
73
+ // The backend's self-reported served model id (from the /v1/models probe),
72
74
  // surfaced as Provider.servedModel. For a local llama-server the wire `model` is
73
75
  // the alias; this is the real name (the .gguf) the tokenizer seam maps. Absent
74
76
  // when no probe ran or it read no row.
75
77
  servedModel?: string;
76
- // #43: backend decodes unbounded without a caller cap (llama-server n_predict
78
+ // Backend decodes unbounded without a caller cap (llama-server n_predict
77
79
  // to the wall) — surfaced as Provider.requiresMaxTokens so consumers can
78
80
  // boot-refuse an envelope-less local alias. Default unset (no claim).
79
81
  requiresMaxTokens?: boolean;
@@ -81,36 +83,39 @@ export type AiSdkProviderConfig = {
81
83
  // (PLURNK_PROVIDERS_REASONING + _BUDGET, read via reasoningFromEnv):
82
84
  // { mode: off|adaptive|on, budget: iff on }. The provider maps it to the
83
85
  // backend's mechanism via reasoningStyle; budget is only ever a magnitude,
84
- // never a hidden activation flag (#33).
86
+ // never a hidden activation flag.
85
87
  reasoning: Reasoning;
86
88
  // Decode tuning: no in-code defaults; the canonical measured values (0.2 /
87
89
  // 1.15) ship as the floor in .env.defaults (alias-scopable). `temperature` is the
88
- // DEFAULT for EVERY request, spread UNDER caller sampling (#30/endpoint#7).
90
+ // DEFAULT for EVERY request, spread UNDER caller sampling.
89
91
  // `repeatPenalty` is the FLOOR the provider manages wherever a grammar rides
90
- // (greedy-under-mask loops without it, #9) — the VALUE is operator config;
92
+ // (greedy-under-mask loops without it) — the VALUE is operator config;
91
93
  // WHERE it applies stays mechanism.
92
94
  temperature: number;
93
95
  repeatPenalty: number;
94
- // #426: anti-degeneration guard on the CLOUD path (grammarStyle "none"), where the
96
+ // Anti-degeneration guard on the cloud path (grammarStyle "none"), where the
95
97
  // repeat_penalty multiplier isn't available - the OpenAI-standard frequency_penalty.
96
98
  // Optional (default 0 = off) so an out-of-date plugin that omits it just runs unguarded
97
99
  // rather than failing construction; the standard factory always supplies it.
98
100
  frequencyPenalty?: number;
99
- // #567: llama.cpp anti-repetition-LOOP controls, sent on the llamacpp path (DRY is a
100
- // llama.cpp sampler). DRY penalizes repeated SEQUENCES with a penalty escalating in run
101
- // length — the tool for a plan-restart loop a single-token repeat_penalty over a short
102
- // window can't see. The GENERIC DRY defaults (0.8/1.75/2) ship as a floor in .env.defaults
103
- // like repeatPenalty — applied for a detected llama.cpp backend, customer-overridable;
104
- // absent (a plugin omitting them) = the box's own default. repeatLastN widens the window.
101
+ // Optional llama.cpp anti-repetition controls. DRY can suppress long
102
+ // repeated sequences, but it can also corrupt exact repetition required by
103
+ // PLURNK operations. The portable default is off; these fields ride only
104
+ // after an explicit operator opt-in. repeatLastN widens the repeat window.
105
105
  dryMultiplier?: number;
106
106
  dryBase?: number;
107
107
  dryAllowedLength?: number;
108
108
  repeatLastN?: number;
109
109
  // Transient-failure retry budget — REQUIRED, no in-code default
110
110
  // (PLURNK_PROVIDERS_RETRY_ATTEMPTS, a non-negative int): 0 = surface the
111
- // first failure; N = up to N retries on a transient error (§4, #18).
111
+ // first failure; N = up to N retries on a transient error
112
+ // ({§provider-failure-normalization}).
112
113
  retryAttempts: number;
113
- // Data-capture knobs (#36), OFF by default — the flag IS the isolation, so a
114
+ // Maximum characters retained from an upstream diagnostic in the public
115
+ // ProviderError Problem. Standard factories supply the env-owned value;
116
+ // direct construction may omit it to preserve the complete diagnostic.
117
+ errorDetailLimit?: number;
118
+ // Data-capture knobs ({§provider-evidence}), off by default — the flag is the isolation, so a
114
119
  // serving turn requests nothing and carries nothing. `topLogprobs`: when a
115
120
  // non-negative int, request `logprobs:true, top_logprobs:<n>` and surface the
116
121
  // per-token confidence on assistant.logprobs (PLURNK_PROVIDERS_TOP_LOGPROBS;
@@ -119,7 +124,7 @@ export type AiSdkProviderConfig = {
119
124
  // gated per-alias.
120
125
  topLogprobs?: number | null;
121
126
  rawBody?: boolean;
122
- // #507 (owner-ruled): the generation-envelope reserves, env-read via
127
+ // {§provider-generation-envelope} The generation-envelope reserves, env-read via
123
128
  // envelopeFromEnv — a percentage of the DETECTED window or an absolute token
124
129
  // count. Optional so an out-of-date sibling keeps constructing (no claim);
125
130
  // the standard factory always supplies them. Resolved against contextWindow
@@ -127,13 +132,13 @@ export type AiSdkProviderConfig = {
127
132
  // derives correctly.
128
133
  reasoningReserve?: ReserveSpec;
129
134
  completionReserve?: ReserveSpec;
130
- // #507: the plurnk.ai router owns tuning (SPEC §5) — false suppresses the
135
+ // The plurnk.ai router owns tuning — false suppresses the
131
136
  // client-side temperature/penalty FLOORS on this provider (caller `sampling`
132
137
  // still passes through verbatim). Default true (floors ride).
133
138
  tuningFloors?: boolean;
134
139
  };
135
140
 
136
- // #539: drop trailing occurrences of a server-rendered EOG marker. llama-server
141
+ // Drop trailing occurrences of a server-rendered EOG marker. llama-server
137
142
  // under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
138
143
  // trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
139
144
  // can never eat body content (a body ending in the literal marker isn't producible
@@ -145,6 +150,44 @@ const stripTrailingSpecial = (content: string, marker: string): string => {
145
150
  return out;
146
151
  };
147
152
 
153
+ type TaggedReasoningProjection = {
154
+ readonly content: string;
155
+ readonly reasoning: string;
156
+ readonly projected: boolean;
157
+ readonly contentStart: number;
158
+ };
159
+
160
+ // {§provider-tagged-reasoning} Only the model-contract position is structural:
161
+ // one exact leading envelope. Parsing after stream assembly keeps SSE and JSON
162
+ // on one path and leaves later literal tags in the visible suffix untouched.
163
+ const projectTaggedReasoning = (
164
+ content: string,
165
+ structuredReasoning: string,
166
+ style: ReasoningResponseStyle,
167
+ ): TaggedReasoningProjection => {
168
+ const opening = "<think>";
169
+ if (style !== "think-tags" || structuredReasoning.length > 0 || !content.startsWith(opening)) {
170
+ return { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
171
+ }
172
+ const closing = "</think>";
173
+ const closingIndex = content.indexOf(closing, opening.length);
174
+ if (closingIndex === -1) {
175
+ return {
176
+ content: "",
177
+ reasoning: content.slice(opening.length),
178
+ projected: true,
179
+ contentStart: [...content].length,
180
+ };
181
+ }
182
+ const suffixStart = closingIndex + closing.length;
183
+ return {
184
+ content: content.slice(suffixStart),
185
+ reasoning: content.slice(opening.length, closingIndex),
186
+ projected: true,
187
+ contentStart: [...content.slice(0, suffixStart)].length,
188
+ };
189
+ };
190
+
148
191
  // Shared budget→effort breakpoints (xai and google had identical copies).
149
192
  export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
150
193
  if (budget <= 1000) return "low";
@@ -152,44 +195,28 @@ export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
152
195
  return "high";
153
196
  };
154
197
 
155
- // chars/2 upper bound (see ./tokenizers.ts) — overcounts safely, never under.
156
- const heuristicTokens = (text: string): number => (text.length === 0 ? 0 : Math.ceil(text.length / 2));
157
-
158
198
  // Body keys the provider owns — a caller's `sampling` passthrough may not set
159
- // these. Two families (#477 audit):
199
+ // these. Two families:
160
200
  // transport/managed — grammar transport, the stream/JSON choice, slot pinning,
161
- // data capture (SPEC §8: backend-specific fields never cross the contract);
201
+ // data capture ({§provider-evidence}: backend-specific fields never cross the contract);
162
202
  // contract invariants — `n` (atomic single completion: choices[0] is the
163
203
  // response; n>1 = paid, dropped output), the tool-calling family (tools-in-
164
204
  // body doctrine, §2: native tool_calls return null content = a broken turn),
165
205
  // modalities/audio (text-only contract), prediction (decode semantics, not
166
206
  // sampling), and the token caps (the envelope is the managed maxTokens —
167
- // sampling must not bypass the consumer's #425 cap).
207
+ // sampling must not bypass the consumer's cap).
168
208
  // Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
169
209
  // platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
170
210
  // metadata, store, verbosity) pass through; the managed floors spread UNDER
171
211
  // sampling stay deliberately caller-overridable.
172
212
  const RESERVED_BODY_KEYS: ReadonlySet<string> = new Set([
173
213
  "model", "messages", "stream", "stream_options", "grammar", "response_format", "id_slot", "logprobs", "top_logprobs",
214
+ "reasoning_format", "reasoning_effort", "thinking", "think", "include_reasoning", "chat_template_kwargs", "thinking_budget_tokens", // lexicon-allow: backend wire fields
174
215
  "n", "tools", "tool_choice", "functions", "function_call", "parallel_tool_calls",
175
216
  "modalities", "audio", "prediction", "max_tokens", "max_completion_tokens",
176
217
  "prompt_cache_key",
177
218
  ]);
178
219
 
179
- // Render a non-accept verdict into a terse, factual grammar_unenforced message
180
- // (SPEC §12 message policy: no guidance prose). `reject` names the diverging code
181
- // point + what the grammar would have accepted; `incomplete` names the valid-prefix
182
- // length that never reached a terminal state.
183
- const describeUnenforced = (v: Exclude<Verdict, { status: "accept" }>): string => {
184
- if (v.status === "reject") {
185
- const expected = v.expected.length > 0
186
- ? v.expected.map((e) => `${e.rule} accepts ${e.accepts}`).join(", ")
187
- : "end of input";
188
- return `grammar not enforced: output rejected by the transported grammar at code point ${v.pos} (${JSON.stringify(v.char)}); expected ${expected}`;
189
- }
190
- return `grammar not enforced: output is an incomplete match of the transported grammar — a valid prefix of ${v.pos} code points that never terminated`;
191
- };
192
-
193
220
  export default class AiSdkProvider implements Provider {
194
221
  #model: string;
195
222
  #url: string | undefined;
@@ -211,8 +238,11 @@ export default class AiSdkProvider implements Provider {
211
238
  #dryAllowedLength: number | undefined;
212
239
  #repeatLastN: number | undefined;
213
240
  #reasoningStyle: ReasoningStyle;
214
- #countTokens: (text: string) => number;
241
+ #reasoningResponseStyle: ReasoningResponseStyle;
242
+ #countPromptTokens: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
243
+ #promptTokensUrl: string | undefined;
215
244
  #calculateCost: (usage: ProviderUsage) => number;
245
+ #calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }>;
216
246
  #source: string;
217
247
  #grammarStyle: GrammarStyle;
218
248
  #promptCacheKey: boolean;
@@ -223,6 +253,7 @@ export default class AiSdkProvider implements Provider {
223
253
  #supportsSlotPinning: boolean;
224
254
  #slotCount: number | null;
225
255
  #retryAttempts: number;
256
+ #errorDetailLimit: number | undefined;
226
257
  #topLogprobs: number | null;
227
258
  #reasoningReserve: ReserveSpec | undefined;
228
259
  #completionReserve: ReserveSpec | undefined;
@@ -230,8 +261,9 @@ export default class AiSdkProvider implements Provider {
230
261
  #rawBody: boolean;
231
262
  #servedModel: string | undefined;
232
263
  #requiresMaxTokens: boolean | undefined;
264
+ readonly attributions?: (context: PluginAttributionContext) => PluginAttribution;
233
265
 
234
- // Optional capability (SPEC §2): exact tokenization served by the backend's
266
+ // Optional capability ({§provider-local-capabilities}): exact tokenization served by the backend's
235
267
  // own vocab. Assigned in the constructor ONLY when the config carries a
236
268
  // tokenizeUrl (llama-server), so `provider.tokenize === undefined` remains
237
269
  // the honest capability signal for every other backend.
@@ -241,6 +273,7 @@ export default class AiSdkProvider implements Provider {
241
273
  this.#model = config.model;
242
274
  this.#url = config.url;
243
275
  this.#languageModel = config.languageModel;
276
+ this.attributions = config.attributions;
244
277
  if ((this.#url === undefined) === (this.#languageModel === undefined)) {
245
278
  throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
246
279
  }
@@ -264,9 +297,16 @@ export default class AiSdkProvider implements Provider {
264
297
  this.#dryAllowedLength = config.dryAllowedLength;
265
298
  this.#repeatLastN = config.repeatLastN;
266
299
  this.#retryAttempts = config.retryAttempts;
300
+ this.#errorDetailLimit = config.errorDetailLimit;
267
301
  this.#reasoningStyle = config.reasoningStyle ?? "none";
268
- this.#countTokens = config.countTokens ?? heuristicTokens;
302
+ this.#reasoningResponseStyle = config.reasoningResponseStyle ?? "verbatim";
303
+ if (config.countPromptTokens !== undefined && config.promptTokensUrl !== undefined) {
304
+ throw new Error(`${config.source ?? "provider"}: configure countPromptTokens or promptTokensUrl, not both`);
305
+ }
306
+ this.#countPromptTokens = config.countPromptTokens ?? ((messages) => estimatePromptTokens(messages));
307
+ this.#promptTokensUrl = config.promptTokensUrl;
269
308
  this.#calculateCost = config.calculateCost ?? (() => 0);
309
+ this.#calculateCharge = config.calculateCharge;
270
310
  this.#source = config.source ?? "provider";
271
311
  this.#grammarStyle = config.grammarStyle ?? "none";
272
312
  this.#promptCacheKey = config.promptCacheKey ?? false;
@@ -286,6 +326,13 @@ export default class AiSdkProvider implements Provider {
286
326
  this.#rawBody = config.rawBody ?? false;
287
327
  this.#servedModel = config.servedModel;
288
328
  this.#requiresMaxTokens = config.requiresMaxTokens;
329
+ const reasoningReserve = this.reasoningReserve;
330
+ if (this.#reasoningStyle === "template"
331
+ && this.#reasoning.mode === "on"
332
+ && reasoningReserve !== null
333
+ && this.#reasoning.budget! > reasoningReserve) {
334
+ throw new Error(`${this.#source}: PLURNK_PROVIDERS_REASONING_BUDGET (${this.#reasoning.budget}) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE (${reasoningReserve})`);
335
+ }
289
336
  const { tokenizeUrl } = config;
290
337
  if (tokenizeUrl !== undefined) {
291
338
  this.tokenize = async (text: string): Promise<number[]> => {
@@ -306,7 +353,7 @@ export default class AiSdkProvider implements Provider {
306
353
  }
307
354
 
308
355
  get contextWindow(): number | null { return this.#contextWindow; }
309
- // #507: envelope reserves — absolute pins stand alone; percentages need the
356
+ // {§provider-generation-envelope} Envelope reserves — absolute pins stand alone; percentages need the
310
357
  // detected window; null = underivable (no claim for core's no-cap path).
311
358
  #resolveReserve(spec: ReserveSpec | undefined): number | null {
312
359
  if (spec === undefined) return null;
@@ -316,63 +363,109 @@ export default class AiSdkProvider implements Provider {
316
363
  get reasoningReserve(): number | null { return this.#resolveReserve(this.#reasoningReserve); }
317
364
  get completionReserve(): number | null { return this.#resolveReserve(this.#completionReserve); }
318
365
  get model(): string { return this.#model; }
319
- // #37: backend's self-reported served id; undefined when unprobed/unknown.
366
+ // Backend's self-reported served id; undefined when unprobed/unknown.
320
367
  get servedModel(): string | undefined { return this.#servedModel; }
321
- // #43: resolved "decodes unbounded without a cap" fact; undefined = no claim.
368
+ // Resolved "decodes unbounded without a cap" fact; undefined = no claim.
322
369
  get requiresMaxTokens(): boolean | undefined { return this.#requiresMaxTokens; }
323
- // Resolved capability (#34): will a transported grammar actually constrain
370
+ // Resolved capability: will a transported grammar actually constrain
324
371
  // this backend's decode? Introspectable so a consumer can verify the rails
325
372
  // are LIVE without spending a generation on a forcing-grammar probe.
326
373
  get constrainsOutput(): boolean { return this.#grammarStyle !== "none"; }
327
374
 
328
- countTokens(text: string): number { return this.#countTokens(text); }
375
+ async countPromptTokens(
376
+ messages: readonly ChatMessage[],
377
+ signal?: AbortSignal,
378
+ ): Promise<PromptTokenMeasurement> {
379
+ if (this.#promptTokensUrl === undefined) {
380
+ return assertPromptTokenMeasurement(
381
+ await this.#countPromptTokens(messages, signal),
382
+ this.#source,
383
+ );
384
+ }
385
+
386
+ signal?.throwIfAborted();
387
+ try {
388
+ const timeout = AbortSignal.timeout(this.#fetchTimeoutMs);
389
+ const response = await this.#fetch(this.#promptTokensUrl, {
390
+ method: "POST",
391
+ headers: { "Content-Type": "application/json", ...this.#headers },
392
+ body: JSON.stringify({
393
+ model: this.#model,
394
+ messages,
395
+ ...this.#reasoningBody(),
396
+ }),
397
+ signal: signal === undefined ? timeout : AbortSignal.any([signal, timeout]),
398
+ });
399
+ if (!response.ok) {
400
+ return estimatePromptTokens(
401
+ messages,
402
+ `llama-server input-token endpoint returned HTTP ${response.status}`,
403
+ );
404
+ }
405
+ const body = await response.json() as { input_tokens?: unknown };
406
+ if (!Number.isInteger(body.input_tokens) || (body.input_tokens as number) < 0) {
407
+ return estimatePromptTokens(
408
+ messages,
409
+ "llama-server input-token endpoint returned no non-negative integer input_tokens",
410
+ );
411
+ }
412
+ return {
413
+ kind: "exact",
414
+ tokens: body.input_tokens as number,
415
+ source: "llama-server:/v1/chat/completions/input_tokens",
416
+ };
417
+ } catch (cause) {
418
+ signal?.throwIfAborted();
419
+ return estimatePromptTokens(
420
+ messages,
421
+ `llama-server input-token measurement failed: ${cause instanceof Error ? cause.message : String(cause)}`,
422
+ );
423
+ }
424
+ }
329
425
  calculateCost(usage: ProviderUsage): number { return this.#calculateCost(usage); }
426
+ calculateCharge(usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }> {
427
+ return this.#calculateCharge?.(usage)
428
+ ?? { kind: "unknown", reason: "no provider rate or settled charge is available" };
429
+ }
330
430
 
331
- // Maps the reasoning INTENT (off | adaptive | on+budget, #33) to the
332
- // backend's wire mechanism — including under a transported grammar. The #32
333
- // clamp (force reasoning_effort "none" under response_format) is LIFTED:
334
- // canary-verified live that fireworks masks ONLY the content channel — the
335
- // reasoning channel rides beside it unmasked, and the plurnk grammar's
336
- // reasoning?/preplan regions absorb any in-band spillover. The old measured
337
- // failures (low→cycles, high→spirals) were pre-max_tokens-cap; with a bounded
338
- // cap the matrix ACCEPTs across efforts (reasoning-rails matrix
339
- // F9). Clamping was the root of the plan-less regression (service#331).
431
+ // Reasoning intent maps independently of grammar transport. The llama-server
432
+ // template mapping is owned by {§llama-reasoning-request}.
340
433
  #reasoningBody(): Record<string, unknown> {
341
434
  const { mode, budget } = this.#reasoning;
342
435
  const on = mode !== "off";
343
436
  switch (this.#reasoningStyle) {
344
- // Native-channel styles. "template" ALWAYS emits — the explicit
345
- // enable_thinking:false is the only working off-switch on llama-server
346
- // (§13). Activation only; budget is enforced by the box's
347
- // --reasoning-budget launch flag (per-request numerics ignored, F7).
348
- //
349
- // #488 postmortem: intent maps IDENTICALLY under a transported
350
- // grammar. The brief rails-win-the-channel clamp (enable_thinking
351
- // forced false under a grammar) is REVERTED — specimens proved the
352
- // SANCTIONED think block is the protection, not the hazard: the
353
- // server auto-gates the grammar around it and content decodes
354
- // constrained (26-run baseline green; zero grammar rejects across
355
- // the #488 "railless" specimens). Closing the channel starved a
356
- // reasoning-tuned model into ESCAPING mid-content into the raw
357
- // thought channel — discarded server-side, decode unconstrained,
358
- // 12,288 tokens billed for 1,033 visible chars. The escape is
359
- // surfaced instead (vanished-token telemetry + meta rail state).
360
- case "template": return { chat_template_kwargs: { enable_thinking: on } };
437
+ case "template": {
438
+ const allowance = mode === "off"
439
+ ? 0
440
+ : mode === "on" ? budget : this.reasoningReserve;
441
+ return {
442
+ chat_template_kwargs: { enable_thinking: on },
443
+ reasoning_format: "auto",
444
+ ...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
445
+ };
446
+ }
361
447
  case "think": return on ? { think: true } : {};
362
448
  case "include_reasoning": return on ? { include_reasoning: true } : {};
363
449
  // effort tiers from the budget; off/adaptive omit the field (the
364
450
  // API's default depth is its adaptive).
365
451
  case "effort": return mode === "on" ? { reasoning_effort: effortFromBudget(budget!) } : {};
366
452
  // Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
367
- // reason-by-default model (DeepSeek V4: default 'high') reasoning (#30).
453
+ // reason-by-default model (DeepSeek V4: default 'high') reasoning.
368
454
  // ADAPTIVE omits the field: the backend's own default posture IS the
369
455
  // adaptive semantics, and the literal "adaptive" is MiniMax-M3-only —
370
- // fireworks 400s it for every other model (wire-verified, #403; the
456
+ // Fireworks 400s it for every other model (wire-verified; the
371
457
  // 1.0.2 adaptive default refused to boot on it). V4 gotcha: integer
372
458
  // efforts 400.
373
459
  case "effort_explicit": return mode === "off"
374
460
  ? { reasoning_effort: "none" }
375
461
  : mode === "on" ? { reasoning_effort: effortFromBudget(budget!) } : {};
462
+ // {§deepseek-reasoning-request}
463
+ case "thinking_effort": return mode === "off"
464
+ ? { thinking: { type: "disabled" } }
465
+ : mode === "on" ? {
466
+ thinking: { type: "enabled" },
467
+ reasoning_effort: effortFromBudget(budget!),
468
+ } : {};
376
469
  // Anthropic compat: explicit thinking object. off → disabled; on →
377
470
  // enabled with budget_tokens; adaptive → omit (the API default).
378
471
  case "anthropic": return mode === "off"
@@ -382,7 +475,7 @@ export default class AiSdkProvider implements Provider {
382
475
  }
383
476
  }
384
477
 
385
- // Per-run slot affinity (#11): the consumer passes WHICH run this is; the
478
+ // Per-worker slot affinity: the consumer passes which worker this is; the
386
479
  // provider owns WHICH slot serves it. Sticky per workerId, round-robin across
387
480
  // new runs (distinct runs → distinct slots while slots last), LRU-bounded
388
481
  // bookkeeping so a long-lived daemon never grows the map unboundedly —
@@ -405,19 +498,19 @@ export default class AiSdkProvider implements Provider {
405
498
  return { id_slot: slot };
406
499
  }
407
500
 
408
- // Optional local llama-server GBNF transport (SPEC §13). Unsupported
501
+ // Optional local llama-server GBNF transport ({§gbnf-response-observation}). Unsupported
409
502
  // backends receive no grammar-related field.
410
503
  #grammarBody(grammar: string | undefined): Record<string, unknown> {
411
504
  if (grammar === undefined) return {};
412
505
  switch (this.#grammarStyle) {
413
506
  // Greedy decoding under hard constraint loops without a repeat-penalty
414
- // floor (#9, SPEC §13) — llama.cpp spells it `repeat_penalty`.
507
+ // floor — llama.cpp spells it `repeat_penalty`.
415
508
  case "llamacpp": return { grammar, repeat_penalty: this.#repeatPenalty };
416
509
  case "none": return {};
417
510
  }
418
511
  }
419
512
 
420
- // Anti-degeneration DEFAULT on EVERY request (#426), keyed to the backend's wire
513
+ // Anti-degeneration default on every request, keyed to the backend's wire
421
514
  // convention - NOT grammar-bound. GBNF is a local constraint, so a cloud
422
515
  // alias runs the sampler bare: firefast (deepseek/fireworks) ran 4/86 bench turns
423
516
  // straight to the token cap on pure looped repetition (run52). Ships next to
@@ -428,7 +521,7 @@ export default class AiSdkProvider implements Provider {
428
521
  // live: together/deepinfra/fireworks; it is OpenAI's own param, so real OpenAI takes it too).
429
522
  #repetitionPenaltyBody(): Record<string, unknown> {
430
523
  switch (this.#grammarStyle) {
431
- // #567: repeat_penalty + optional DRY (repeated-SEQUENCE penalty) + a wider
524
+ // repeat_penalty + optional DRY (repeated-sequence penalty) + a wider
432
525
  // repeat_last_n window — the loop-breaking tools a llama.cpp backend serves.
433
526
  // Each rides only when its operator knob is set; absent = the box's default.
434
527
  case "llamacpp": return {
@@ -444,12 +537,12 @@ export default class AiSdkProvider implements Provider {
444
537
  }
445
538
  }
446
539
 
447
- // First-party telemetry headers (SPEC §5): forwarded ONLY when the spec
540
+ // First-party telemetry headers ({§provider-request-authority}): forwarded only when the spec
448
541
  // opted in (the plurnk endpoint). The gate is here, not at the call site, so
449
542
  // attributions/client/strikes can never reach a third-party backend even if
450
543
  // the consumer passes them to the wrong provider. Empty values emit no header
451
544
  // — EXCEPT strikes, where 0 is a real value (clean streak) distinct from
452
- // absent (consumer didn't report); contract per plurnk-service#313. Strikes
545
+ // absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
453
546
  // ride HTTP headers only — the packet never carries them (the model must
454
547
  // never see strike state; engine accounting is not a metric to game).
455
548
  #metadataHeaders(attributions: string[] | undefined, client: string | undefined, strikes: number | undefined, workerId: string, primaryWorkerId: string | undefined, workspaceId: string | undefined, loop: number | undefined, turn: number | undefined): Record<string, string> {
@@ -458,11 +551,11 @@ export default class AiSdkProvider implements Provider {
458
551
  if (attributions !== undefined && attributions.length > 0) h["Plurnk-Attribution"] = JSON.stringify(attributions);
459
552
  if (client !== undefined && client.length > 0) h["Plurnk-Client"] = client;
460
553
  if (strikes !== undefined && Number.isInteger(strikes) && strikes >= 0) h["Plurnk-Strikes"] = String(strikes);
461
- // Worker identity (#26, wire-name completed #486/#511): the opaque workerId
554
+ // Worker identity: the opaque workerId
462
555
  // the consumer already supplies, forwarded so the endpoint can key
463
556
  // per-worker affinity/telemetry — same gate as every first-party signal.
464
557
  h["Plurnk-Worker-Id"] = workerId;
465
- // Root worker of the lineage (#522): the no-parent ancestor of this turn's
558
+ // Root worker of the lineage ({§worker-primary}): the no-parent ancestor of this turn's
466
559
  // worker tree. The consumer classifies primary-vs-spawned by equality
467
560
  // (primaryWorkerId == workerId ⇒ the primary/root worker). The provider
468
561
  // EMITS what the consumer supplies and never invents a primary; the
@@ -470,7 +563,7 @@ export default class AiSdkProvider implements Provider {
470
563
  // own, where it equals workerId). Absence is the consumer's violation for
471
564
  // the endpoint to surface, not a provider default.
472
565
  if (primaryWorkerId !== undefined && primaryWorkerId.length > 0) h["Plurnk-Worker-Primary"] = primaryWorkerId;
473
- // Turn coordinate (#404, extends #26 per #391): workspace/loop/turn, the
566
+ // Turn coordinate ({§lifecycle-terms}): workspace/loop/turn, the
474
567
  // daemon-side sequence the endpoint can never scrape from the wire.
475
568
  // Coordinates are 1-based — 0 is not a real value, so no strikes-style
476
569
  // zero exception; absent/empty/0 emits no header.
@@ -480,33 +573,7 @@ export default class AiSdkProvider implements Provider {
480
573
  return h;
481
574
  }
482
575
 
483
- // Enforcement verification (SPEC §13). When a grammar was actually transported
484
- // (grammarStyle !== "none"), the backend MUST have constrained the output;
485
- // some silently drop the grammar field or mislabel the channel, and without
486
- // this check we would return unconstrained output as if enforced. STRICT: any
487
- // non-accept verdict (reject, or an incomplete/never-terminated match) is a
488
- // grammar_unenforced failure. A grammar our own validator can't parse — even
489
- // though the backend accepted it (a port-vs-llama.cpp gap) — is a non-fatal
490
- // verify gap: warn, don't fail a transport that may have worked. This is a
491
- // conformance check against the grammar we already hold, NOT a plurnk-DSL
492
- // parse (§8) — it stays grammar-generic and backend-agnostic.
493
- // Validate output against the grammar. Returns the verdict, or null on the
494
- // verify GAP — a grammar our own validator can't parse (a port-vs-llama.cpp
495
- // gap): warn, don't manufacture a conflict from a check that didn't run.
496
- #grammarVerdict(grammar: string, content: string): Verdict | null {
497
- try {
498
- return validateGbnf(grammar, content);
499
- } catch (cause) {
500
- // Once per (code, message) — #40: this fires PER TURN otherwise.
501
- emitWarningOnce(
502
- `${this.#source}: could not verify grammar enforcement — the transported grammar did not parse in @plurnk/gbnf (${(cause as Error).message})`,
503
- "PLURNK_GRAMMAR_UNVERIFIABLE",
504
- );
505
- return null;
506
- }
507
- }
508
-
509
- // PLURNK_PROVIDERS_GBNF_DEBUG (SPEC §13): validate the supplied GBNF locally and fail
576
+ // PLURNK_PROVIDERS_GBNF_DEBUG ({§gbnf-response-observation}): validate the supplied GBNF locally and fail
510
577
  // hard if it's malformed, BEFORE any wire call — and the grammar is NOT
511
578
  // transported, so the request runs unconstrained. A debug aid to catch invalid
512
579
  // grammars (e.g. while editing the plurnk grammar) without a model round-trip;
@@ -521,7 +588,7 @@ export default class AiSdkProvider implements Provider {
521
588
  }
522
589
  }
523
590
 
524
- // Per-turn metadata bag (#23): pass the backend's non-standard top-level fields
591
+ // Per-turn metadata bag: pass the backend's non-standard top-level fields
525
592
  // through verbatim. Providers do not reinterpret vendor currency or account
526
593
  // metadata; a monetary value carries its own amount and currency.
527
594
  #buildMeta(chunkMetadata: Record<string, unknown>): Record<string, unknown> | undefined {
@@ -533,7 +600,8 @@ export default class AiSdkProvider implements Provider {
533
600
  // penalties, stop, seed, …) merged UNDER the managed body: model, messages,
534
601
  // reasoning, grammar (+ its repeat-penalty floor), max_tokens and slot always
535
602
  // win, and reserved transport/protocol keys are stripped so the passthrough
536
- // can't smuggle a grammar, a stream toggle, or a backend slot (SPEC §8).
603
+ // can't smuggle a grammar, a stream toggle, or a backend slot
604
+ // ({§provider-request-authority}).
537
605
  #samplingBody(sampling: Record<string, unknown> | undefined): Record<string, unknown> {
538
606
  if (sampling === undefined) return {};
539
607
  const out: Record<string, unknown> = {};
@@ -542,27 +610,24 @@ export default class AiSdkProvider implements Provider {
542
610
  }
543
611
 
544
612
  async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }: { messages: ChatMessage[]; workerId: string; primaryWorkerId?: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse> {
545
- // Boundary validation (SPEC §2): the worker identity is required.
613
+ // {§provider-interface} The worker identity is required.
546
614
  if (workerId === undefined || workerId.length === 0) throw new Error("generate: workerId is required — the worker's stable, opaque identity");
547
- // Reject before any wire call when already aborted (SPEC §10.8).
615
+ // Reject before any wire call when already aborted
616
+ // ({§provider-failure-normalization}).
548
617
  signal?.throwIfAborted();
549
618
 
550
- // Grammar handling (SPEC §13). PLURNK_PROVIDERS_GBNF_DEBUG validates the supplied
551
- // grammar locally and throws on a malformed one, then WITHHOLDS it so the
552
- // model generates UNCONSTRAINED — and the free output is still verified
553
- // against the grammar (below), surfacing exactly where the model's natural
554
- // output and the grammar conflict. Otherwise the grammar is sent when the
555
- // backend supports it (grammarStyle !== "none").
619
+ // Grammar handling ({§gbnf-response-observation}). Debug validates the
620
+ // supplied grammar before the call but withholds it from the backend.
556
621
  const wantGrammar = grammar !== undefined && this.#grammarStyle !== "none";
557
622
  if (wantGrammar && this.#gbnfDebug) this.#assertGrammarValid(grammar!);
558
623
  const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
559
624
 
560
625
  // Assembly order = precedence: the family's sampling DEFAULTS
561
- // (PLURNK_PROVIDERS_TEMPERATURE — universal, #30 measured it on grammar
626
+ // (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
562
627
  // paths and the name promises every request) < the caller's `sampling`
563
628
  // < the managed fields, which always win.
564
629
  const body: Record<string, unknown> = {
565
- // #507: floors suppressed on router-owned-tuning providers (plurnk) —
630
+ // Floors are suppressed on router-owned-tuning providers (plurnk) —
566
631
  // the router's per-model tuning must not be overridden by client floors.
567
632
  ...(this.#tuningFloors ? { temperature: this.#temperature, ...this.#repetitionPenaltyBody() } : {}),
568
633
  ...this.#samplingBody(sampling),
@@ -572,11 +637,11 @@ export default class AiSdkProvider implements Provider {
572
637
  ...this.#reasoningBody(),
573
638
  ...this.#grammarBody(sendGrammar),
574
639
  ...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
575
- // #36: request per-token logprobs only when enabled (managed field —
640
+ // Request per-token logprobs only when enabled (managed field —
576
641
  // reserved from caller sampling; the env flag is the single control).
577
642
  ...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
578
643
  ...this.#slotBody(workerId),
579
- // #518: prompt-cache affinity -- workerId as the OpenAI-standard
644
+ // Prompt-cache affinity -- workerId as the OpenAI-standard
580
645
  // prompt_cache_key routes a worker's turns to one serverless replica so
581
646
  // its stable prefix caches (managed; reserved from caller sampling).
582
647
  ...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
@@ -636,14 +701,14 @@ export default class AiSdkProvider implements Provider {
636
701
  });
637
702
  } catch (err) {
638
703
  if (signal?.aborted) throw err;
639
- const pe = toProviderError(err, this.#source);
704
+ const pe = toProviderError(err, this.#source, this.#errorDetailLimit);
640
705
  if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
641
706
  throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
642
707
  }
643
708
  throw pe;
644
709
  }
645
710
 
646
- // #539: llama-server --special renders EOG tokens as text, so a turn ending
711
+ // llama-server --special renders EOG tokens as text, so a turn ending
647
712
  // via raw EOS carries a trailing <eos> the grammar never sanctioned - it both
648
713
  // false-rejects the rail verdict and leaks a control token into the packet.
649
714
  // Strip the server-reported eos_token from the tail ONCE, before the verdict
@@ -651,66 +716,125 @@ export default class AiSdkProvider implements Provider {
651
716
  // wire text for forensics.
652
717
  if (this.#eosText !== undefined) raw.content = stripTrailingSpecial(raw.content, this.#eosText);
653
718
 
654
- // Grammar conformance (§13): bytes always flow; the verdict is an
655
- // observation. Same check whether the grammar was transported
656
- // (sendGrammar) or withheld (PLURNK_PROVIDERS_GBNF_DEBUG filter mode) — a
657
- // non-accept verdict attaches a grammar_unenforced telemetry event
658
- // (message + divergence position) and the response returns normally.
659
- // Discard/retry/escalate/self-correct is the consumer's policy.
660
- let telemetry: TelemetryEvent[] | undefined;
661
- let railsMeta: Record<string, unknown> | undefined;
662
- const usage = raw.usage;
663
- const observedGrammar = sendGrammar ?? (wantGrammar && this.#gbnfDebug ? grammar : undefined);
664
- if (observedGrammar !== undefined) {
665
- const verdict = this.#grammarVerdict(observedGrammar, raw.content);
666
- if (verdict !== null && verdict.status !== "accept") {
667
- telemetry = [{ source: this.#source, kind: "grammar_unenforced", message: describeUnenforced(verdict), position: verdict.pos }];
719
+ const taggedReasoning = projectTaggedReasoning(
720
+ raw.content,
721
+ raw.reasoning,
722
+ this.#reasoningResponseStyle,
723
+ );
724
+
725
+ // Preserve the exact sentence seen at the grammar boundary. llama-server's
726
+ // `reasoning_format: "auto"` projects one raw Harmony enclosure into the
727
+ // reasoning/content fields; the wire field's presence is the proof that the
728
+ // projection occurred. The provider represents this evidence and never grades it.
729
+ let grammarEvidence: GrammarEvidence | undefined;
730
+ if (wantGrammar) {
731
+ if (taggedReasoning.projected) {
732
+ grammarEvidence = {
733
+ input: raw.content,
734
+ contentStart: taggedReasoning.contentStart,
735
+ transported: sendGrammar !== undefined,
736
+ };
737
+ } else if (this.#reasoningStyle === "template" && this.#reasoning.mode !== "off") {
738
+ if (raw.reasoningProjected) {
739
+ const prefix = `<|channel>thought\n${raw.reasoning}<channel|>`;
740
+ grammarEvidence = {
741
+ input: `${prefix}${raw.content}`,
742
+ contentStart: [...prefix].length,
743
+ transported: sendGrammar !== undefined,
744
+ };
745
+ }
746
+ } else {
747
+ grammarEvidence = {
748
+ input: raw.content,
749
+ contentStart: 0,
750
+ transported: sendGrammar !== undefined,
751
+ };
668
752
  }
669
- // #488 per-request loud state: rail attachment + conformance verdict
670
- // ride `meta` into the consumer's turn row, so a drill reads rail
671
- // presence PER TURN from the run db instead of inferring it from
672
- // output shape (the #488 misdiagnosis, twice).
673
- railsMeta = { railsAttached: sendGrammar !== undefined, railsVerdict: verdict?.status ?? "unverifiable" };
674
- // #488 channel-escape detector (the run105 class): completion tokens
753
+ }
754
+
755
+ if (taggedReasoning.projected) {
756
+ raw.content = taggedReasoning.content;
757
+ raw.reasoning = taggedReasoning.reasoning;
758
+ raw.usage = attributeUnitemizedReasoning(raw.usage, raw.reasoning, raw.content);
759
+ }
760
+
761
+ let notices: ProviderNotice[] | undefined;
762
+ const usage = raw.usage;
763
+ if (sendGrammar !== undefined && this.tokenize !== undefined) {
764
+ // Channel-escape detector: completion tokens
675
765
  // billed far beyond every visible channel mean the decode ESCAPED into
676
- // a server-discarded reasoning block mid-emission — unconstrained,
677
- // invisible, billed (12,288 billed vs 1,033 chars visible, live).
678
- // countTokens OVERCOUNTS text (chars/2 upper bound), so billed
679
- // exceeding visible-plus-slack is real vanishing, not estimator noise.
680
- const visible = this.#countTokens(raw.content) + this.#countTokens(raw.reasoning);
681
- if (sendGrammar !== undefined && usage.completion > visible + 64) {
682
- (telemetry ??= []).push({
683
- source: this.#source,
684
- kind: "grammar_unenforced",
685
- message: `decode escaped the grammar: ${usage.completion} completion tokens billed but only ~${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
686
- position: [...raw.content].length,
687
- });
766
+ // a server-discarded reasoning block mid-emission. This diagnostic
767
+ // requires the serving vocabulary; an estimate cannot prove absence.
768
+ try {
769
+ const [contentTokens, reasoningTokens] = await Promise.all([
770
+ this.tokenize(raw.content),
771
+ this.tokenize(raw.reasoning),
772
+ ]);
773
+ const visible = contentTokens.length + reasoningTokens.length;
774
+ if (usage.completion > visible + 64) {
775
+ (notices ??= []).push({
776
+ source: this.#source,
777
+ kind: "grammar_unenforced",
778
+ level: "warn",
779
+ message: `decode escaped the grammar: ${usage.completion} completion tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
780
+ position: [...raw.content].length,
781
+ });
782
+ }
783
+ } catch (cause) {
784
+ emitWarningOnce(
785
+ `${this.#source}: exact visible-token diagnostic unavailable (${cause instanceof Error ? cause.message : String(cause)})`,
786
+ "PLURNK_VISIBLE_TOKEN_COUNT_UNAVAILABLE",
787
+ );
688
788
  }
689
789
  }
690
790
 
691
- const builtMeta = this.#buildMeta(raw.metadata);
692
- const meta = railsMeta !== undefined ? { ...builtMeta, ...railsMeta } : builtMeta;
791
+ const meta = this.#buildMeta(raw.metadata);
693
792
  const logprobs = raw.logprobs.length > 0 ? raw.logprobs : undefined;
694
793
  const meanLogprob = logprobs !== undefined
695
794
  ? logprobs.reduce((sum, token) => sum + token.logprob, 0) / logprobs.length
696
795
  : undefined;
697
796
 
698
- return {
699
- assistant: {
700
- content: raw.content,
701
- reasoning: raw.reasoning.length > 0 ? raw.reasoning : null,
702
- ...(raw.reasoningEncrypted.length > 0
703
- ? { reasoningEncrypted: raw.reasoningEncrypted }
704
- : {}),
705
- usage,
706
- finishReason: raw.finishReason,
707
- model: raw.model,
708
- ...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
709
- },
797
+ const assistant = {
798
+ content: raw.content,
799
+ reasoning: raw.reasoning.length > 0 ? raw.reasoning : null,
800
+ ...(raw.reasoningEncrypted.length > 0
801
+ ? { reasoningEncrypted: raw.reasoningEncrypted }
802
+ : {}),
803
+ usage,
804
+ model: raw.model,
805
+ ...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
806
+ };
807
+ const evidence = {
710
808
  assistantRaw: raw,
809
+ ...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
711
810
  ...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
712
811
  ...(meta !== undefined ? { meta } : {}),
713
- ...(telemetry !== undefined ? { telemetry } : {}),
812
+ ...(notices !== undefined ? { notices } : {}),
813
+ };
814
+ if (raw.finishReason === "resource_interrupted") {
815
+ const attempt: ProviderResponse<"resource_interrupted"> = {
816
+ assistant: { ...assistant, finishReason: raw.finishReason },
817
+ ...evidence,
818
+ };
819
+ throw new ProviderError(
820
+ this.#source,
821
+ "resource_interrupted",
822
+ "The provider interrupted generation because inference resources were unavailable.",
823
+ {
824
+ attempt,
825
+ extensions: {
826
+ stage: "provider-response",
827
+ finishReason: "resource_interrupted",
828
+ ...(raw.rawFinishReason === undefined
829
+ ? {}
830
+ : { rawFinishReason: raw.rawFinishReason }),
831
+ },
832
+ },
833
+ );
834
+ }
835
+ return {
836
+ assistant: { ...assistant, finishReason: raw.finishReason },
837
+ ...evidence,
714
838
  };
715
839
  }
716
840
  }