@plurnk/plurnk-providers 1.5.0 → 1.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/.env.defaults +41 -34
  2. package/README.md +15 -0
  3. package/SPEC.md +242 -89
  4. package/dist/AiSdkProvider.d.ts +33 -33
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +442 -133
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +10 -11
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +87 -25
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +9 -24
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +86 -25
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/accounting.d.ts +5 -2
  17. package/dist/accounting.d.ts.map +1 -1
  18. package/dist/accounting.js +100 -16
  19. package/dist/accounting.js.map +1 -1
  20. package/dist/accountingPublic.d.ts +5 -0
  21. package/dist/accountingPublic.d.ts.map +1 -0
  22. package/dist/accountingPublic.js +3 -0
  23. package/dist/accountingPublic.js.map +1 -0
  24. package/dist/aiSdkTransport.d.ts +9 -2
  25. package/dist/aiSdkTransport.d.ts.map +1 -1
  26. package/dist/aiSdkTransport.js +160 -62
  27. package/dist/aiSdkTransport.js.map +1 -1
  28. package/dist/capacity.d.ts +26 -0
  29. package/dist/capacity.d.ts.map +1 -0
  30. package/dist/capacity.js +90 -0
  31. package/dist/capacity.js.map +1 -0
  32. package/dist/catalogProvider.d.ts +8 -3
  33. package/dist/catalogProvider.d.ts.map +1 -1
  34. package/dist/catalogProvider.js +45 -41
  35. package/dist/catalogProvider.js.map +1 -1
  36. package/dist/compatibleProvider.d.ts.map +1 -1
  37. package/dist/compatibleProvider.js +26 -12
  38. package/dist/compatibleProvider.js.map +1 -1
  39. package/dist/cost.d.ts +10 -10
  40. package/dist/cost.d.ts.map +1 -1
  41. package/dist/cost.js +90 -42
  42. package/dist/cost.js.map +1 -1
  43. package/dist/env.d.ts +13 -11
  44. package/dist/env.d.ts.map +1 -1
  45. package/dist/env.js +83 -46
  46. package/dist/env.js.map +1 -1
  47. package/dist/errors.d.ts +17 -3
  48. package/dist/errors.d.ts.map +1 -1
  49. package/dist/errors.js +91 -8
  50. package/dist/errors.js.map +1 -1
  51. package/dist/index.d.ts +7 -6
  52. package/dist/index.d.ts.map +1 -1
  53. package/dist/index.js +5 -3
  54. package/dist/index.js.map +1 -1
  55. package/dist/ollama.js +3 -3
  56. package/dist/ollama.js.map +1 -1
  57. package/dist/promptTokens.d.ts.map +1 -1
  58. package/dist/promptTokens.js +7 -4
  59. package/dist/promptTokens.js.map +1 -1
  60. package/dist/sdkModels.d.ts +7 -2
  61. package/dist/sdkModels.d.ts.map +1 -1
  62. package/dist/sdkModels.js +43 -13
  63. package/dist/sdkModels.js.map +1 -1
  64. package/dist/types.d.ts +55 -33
  65. package/dist/types.d.ts.map +1 -1
  66. package/dist/usage.d.ts +22 -5
  67. package/dist/usage.d.ts.map +1 -1
  68. package/dist/usage.js +169 -83
  69. package/dist/usage.js.map +1 -1
  70. package/package.json +18 -7
  71. package/src/AiSdkProvider.test.ts +964 -206
  72. package/src/AiSdkProvider.ts +545 -155
  73. package/src/Mock.test.ts +69 -30
  74. package/src/Mock.ts +99 -29
  75. package/src/Pool.test.ts +90 -19
  76. package/src/Pool.ts +96 -27
  77. package/src/ProviderRegistry.test.ts +16 -11
  78. package/src/accounting.test.ts +58 -22
  79. package/src/accounting.ts +119 -18
  80. package/src/accountingPublic.ts +9 -0
  81. package/src/aiSdkTransport.test.ts +42 -49
  82. package/src/aiSdkTransport.ts +174 -62
  83. package/src/boundaries.test.ts +2 -0
  84. package/src/capacity.test.ts +92 -0
  85. package/src/capacity.ts +140 -0
  86. package/src/catalogProvider.test.ts +339 -30
  87. package/src/catalogProvider.ts +65 -47
  88. package/src/compatibleProvider.test.ts +7 -5
  89. package/src/compatibleProvider.ts +29 -13
  90. package/src/cost.test.ts +86 -36
  91. package/src/cost.ts +111 -50
  92. package/src/defaults.test.ts +13 -3
  93. package/src/env.test.ts +103 -25
  94. package/src/env.ts +153 -65
  95. package/src/errors.test.ts +80 -2
  96. package/src/errors.ts +107 -8
  97. package/src/index.ts +26 -7
  98. package/src/ollama.test.ts +5 -3
  99. package/src/ollama.ts +3 -3
  100. package/src/promptTokens.ts +8 -5
  101. package/src/sdkModels.test.ts +77 -8
  102. package/src/sdkModels.ts +51 -15
  103. package/src/types.ts +112 -51
  104. package/src/usage.test.ts +112 -116
  105. package/src/usage.ts +214 -93
@@ -5,13 +5,29 @@
5
5
  // Composition, not inheritance: an official AI SDK language model supplies the
6
6
  // ordinary vendor protocol. The compatible URL path remains only for PLURNK
7
7
  // extensions and local endpoint probes the SDK cannot represent.
8
- import { executeAiSdkModel, executeOpenAICompatible } from "./aiSdkTransport.js";
9
- import { toProviderError, ProviderError } from "./errors.js";
10
- import { attributeUnitemizedReasoning } from "./usage.js";
8
+ import { MAX_PROVIDER_TIMEOUT_MS } from "./env.js";
9
+ import { executeAiSdkModel, executeOpenAICompatible, transportFailureEvidence, } from "./aiSdkTransport.js";
10
+ import { prepareRetries } from "ai/internal";
11
+ import { toProviderError, ProviderError, ProviderTimeoutError } from "./errors.js";
11
12
  import { validateGbnf } from "@plurnk/gbnf";
12
13
  import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.js";
13
14
  import { emitWarningOnce } from "./warnings.js";
14
- import { validateAuthoritativeCharge } from "./cost.js";
15
+ import { resolveProviderCost } from "./cost.js";
16
+ import { validateProviderRequestAccounting } from "./accounting.js";
17
+ import { validateProviderUsage } from "./usage.js";
18
+ import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.js";
19
+ class ProviderRequestObserverError extends Error {
20
+ constructor(cause) {
21
+ super("provider request accounting could not be durably settled", { cause });
22
+ this.name = "ProviderRequestObserverError";
23
+ }
24
+ }
25
+ class ProviderRequestAccountingError extends Error {
26
+ constructor(cause) {
27
+ super("provider request accounting could not be normalized", { cause });
28
+ this.name = "ProviderRequestAccountingError";
29
+ }
30
+ }
15
31
  // Drop trailing occurrences of a server-rendered EOG marker. llama-server
16
32
  // under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
17
33
  // trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
@@ -52,11 +68,20 @@ const projectLeadingReasoning = (content, structuredReasoning, opening, closing)
52
68
  const projectTaggedReasoning = (content, structuredReasoning, style) => style === "think-tags"
53
69
  ? projectLeadingReasoning(content, structuredReasoning, "<think>", "</think>")
54
70
  : { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
55
- // llama-server's template reasoning parser can project this leading channel out
56
- // of the OpenAI-compatible response. Grammar evidence needs the sentence before
57
- // that lossy projection, so constrained template turns request it verbatim and
58
- // split the observed enclosure here.
59
- const projectTemplateReasoning = (content) => projectLeadingReasoning(content, "", "<|channel>thought\n", "<channel|>");
71
+ // llama-server's template reasoning parser can project either supported leading
72
+ // reasoning envelope out of the OpenAI-compatible response. Grammar evidence
73
+ // needs the sentence before that lossy projection, so constrained template turns
74
+ // request it verbatim and split the observed enclosure here.
75
+ const projectTemplateReasoning = (content) => {
76
+ for (const [opening, closing] of [
77
+ ["<|channel>thought\n", "<channel|>"],
78
+ ["<think>\n", "</think>"],
79
+ ]) {
80
+ if (content.startsWith(opening))
81
+ return projectLeadingReasoning(content, "", opening, closing);
82
+ }
83
+ return { content, reasoning: "", projected: false, contentStart: 0 };
84
+ };
60
85
  // Shared budget→effort breakpoints (xai and google had identical copies).
61
86
  export const effortFromBudget = (budget) => {
62
87
  if (budget <= 1000)
@@ -65,6 +90,11 @@ export const effortFromBudget = (budget) => {
65
90
  return "medium";
66
91
  return "high";
67
92
  };
93
+ // AI SDK's portable reasoning control has no boolean-enabled value. `medium`
94
+ // is the neutral activation projection for an explicit, unqualified `on`; it
95
+ // changes no PLURNK output budget. An operator reasoning subset, when present, remains
96
+ // the only input to the existing magnitude-to-tier projection.
97
+ const effortFromReasoning = (reasoning) => reasoning.budget === null ? "medium" : effortFromBudget(reasoning.budget);
68
98
  // Body keys the provider owns — a caller's `sampling` passthrough may not set
69
99
  // these. Two families:
70
100
  // transport/managed — grammar transport, the stream/JSON choice, slot pinning,
@@ -73,7 +103,7 @@ export const effortFromBudget = (budget) => {
73
103
  // response; n>1 = paid, dropped output), the tool-calling family (tools-in-
74
104
  // body doctrine, §2: native tool_calls return null content = a broken turn),
75
105
  // modalities/audio (text-only contract), prediction (decode semantics, not
76
- // sampling), and the token caps (the envelope is the managed maxTokens —
106
+ // sampling), and the token caps (the envelope is the managed maxOutputTokens —
77
107
  // sampling must not bypass the consumer's cap).
78
108
  // Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
79
109
  // platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
@@ -91,6 +121,8 @@ export default class AiSdkProvider {
91
121
  #url;
92
122
  #languageModel;
93
123
  #fetchTimeoutMs;
124
+ #operationTimeoutMs;
125
+ #firstContentTimeoutMs;
94
126
  #streamIdleTimeoutMs;
95
127
  #headers;
96
128
  #fetch;
@@ -98,6 +130,11 @@ export default class AiSdkProvider {
98
130
  #apiKeyRejectedMessage;
99
131
  #eosText;
100
132
  #contextWindow;
133
+ #maxInputTokens;
134
+ #maxOutputTokens;
135
+ #outputBudget;
136
+ #reasoningBudget;
137
+ #additiveReasoningProvider;
101
138
  #reasoning;
102
139
  #temperature;
103
140
  #repeatPenalty;
@@ -110,12 +147,13 @@ export default class AiSdkProvider {
110
147
  #reasoningResponseStyle;
111
148
  #countPromptTokens;
112
149
  #promptTokensUrl;
113
- #calculateCost;
114
- #calculateCharge;
115
- #normalizeCharge;
150
+ #estimateCost;
151
+ #normalizeCost;
116
152
  #source;
117
153
  #grammarStyle;
118
- #promptCacheKey;
154
+ #cacheAffinity;
155
+ #systemCacheProviderOptions;
156
+ #reasoningResponseProviderOptions;
119
157
  #serviceTier;
120
158
  #gbnfDebug;
121
159
  #streaming;
@@ -125,12 +163,10 @@ export default class AiSdkProvider {
125
163
  #retryAttempts;
126
164
  #errorDetailLimit;
127
165
  #topLogprobs;
128
- #reasoningReserve;
129
- #completionReserve;
130
166
  #tuningFloors;
131
167
  #rawBody;
132
168
  #servedModel;
133
- #requiresMaxTokens;
169
+ #requiresOutputBudget;
134
170
  attributions;
135
171
  // Optional capability ({§provider-local-capabilities}): exact tokenization served by the backend's
136
172
  // own vocab. Assigned in the constructor ONLY when the config carries a
@@ -145,11 +181,28 @@ export default class AiSdkProvider {
145
181
  if ((this.#url === undefined) === (this.#languageModel === undefined)) {
146
182
  throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
147
183
  }
184
+ for (const [name, value] of [
185
+ ["fetchTimeoutMs", config.fetchTimeoutMs],
186
+ ["operationTimeoutMs", config.operationTimeoutMs],
187
+ ["firstContentTimeoutMs", config.firstContentTimeoutMs],
188
+ ["streamIdleTimeoutMs", config.streamIdleTimeoutMs],
189
+ ]) {
190
+ if (value !== undefined && (!Number.isInteger(value) || value < 0 || value > MAX_PROVIDER_TIMEOUT_MS)) {
191
+ throw new Error(`${config.source ?? "provider"}: ${name} must be an integer from 0 through ${MAX_PROVIDER_TIMEOUT_MS} milliseconds`);
192
+ }
193
+ }
148
194
  this.#fetchTimeoutMs = config.fetchTimeoutMs;
195
+ this.#operationTimeoutMs = config.operationTimeoutMs;
196
+ this.#firstContentTimeoutMs = config.firstContentTimeoutMs;
149
197
  this.#streamIdleTimeoutMs = config.streamIdleTimeoutMs;
150
198
  this.#headers = config.headers ?? {};
151
199
  this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
152
200
  this.#contextWindow = config.contextWindow ?? null;
201
+ this.#maxInputTokens = config.maxInputTokens ?? null;
202
+ this.#maxOutputTokens = config.maxOutputTokens ?? null;
203
+ this.#outputBudget = config.outputBudget ?? null;
204
+ this.#reasoningBudget = config.reasoningBudget ?? null;
205
+ this.#additiveReasoningProvider = config.additiveReasoningProvider;
153
206
  this.#reasoning = config.reasoning;
154
207
  // Loud guard: an out-of-date consumer (stale plugin dist) omitting the
155
208
  // required tuning fields must fail at construction, not silently send
@@ -173,12 +226,33 @@ export default class AiSdkProvider {
173
226
  }
174
227
  this.#countPromptTokens = config.countPromptTokens ?? ((messages) => estimatePromptTokens(messages));
175
228
  this.#promptTokensUrl = config.promptTokensUrl;
176
- this.#calculateCost = config.calculateCost ?? (() => 0);
177
- this.#calculateCharge = config.calculateCharge;
178
- this.#normalizeCharge = config.normalizeCharge;
229
+ this.#estimateCost = config.estimateCost
230
+ ?? (() => ({
231
+ kind: "unknown",
232
+ reason: "the request reported no direct cost and no model rate is configured",
233
+ }));
234
+ this.#normalizeCost = config.normalizeCost;
179
235
  this.#source = config.source ?? "provider";
180
236
  this.#grammarStyle = config.grammarStyle ?? "none";
181
- this.#promptCacheKey = config.promptCacheKey ?? false;
237
+ this.#cacheAffinity = config.cacheAffinity;
238
+ this.#systemCacheProviderOptions = config.systemCacheProviderOptions;
239
+ this.#reasoningResponseProviderOptions = config.reasoningResponseProviderOptions;
240
+ if (this.#languageModel === undefined && this.#cacheAffinity?.target === "provider-option") {
241
+ throw new Error(`${this.#source}: native provider-option cache affinity requires an AI SDK model`);
242
+ }
243
+ if (this.#languageModel !== undefined && this.#cacheAffinity?.target === "body") {
244
+ throw new Error(`${this.#source}: body cache affinity requires an OpenAI-compatible URL`);
245
+ }
246
+ if (this.#languageModel === undefined && this.#systemCacheProviderOptions !== undefined) {
247
+ throw new Error(`${this.#source}: system cache provider options require an AI SDK model`);
248
+ }
249
+ if (this.#languageModel === undefined && this.#reasoningResponseProviderOptions !== undefined) {
250
+ throw new Error(`${this.#source}: reasoning response provider options require an AI SDK model`);
251
+ }
252
+ if (this.#cacheAffinity?.target === "provider-option"
253
+ && Object.hasOwn(this.#reasoningResponseProviderOptions?.[this.#cacheAffinity.provider] ?? {}, this.#cacheAffinity.name)) {
254
+ throw new Error(`${this.#source}: reasoning response options conflict with cache-affinity option ${this.#cacheAffinity.provider}.${this.#cacheAffinity.name}`);
255
+ }
182
256
  this.#serviceTier = config.serviceTier;
183
257
  this.#gbnfDebug = config.gbnfDebug ?? false;
184
258
  this.#streaming = config.streaming ?? true;
@@ -189,18 +263,42 @@ export default class AiSdkProvider {
189
263
  this.#supportsSlotPinning = config.supportsSlotPinning ?? false;
190
264
  this.#slotCount = config.slotCount ?? null;
191
265
  this.#topLogprobs = config.topLogprobs ?? null;
192
- this.#reasoningReserve = config.reasoningReserve;
193
- this.#completionReserve = config.completionReserve;
194
266
  this.#tuningFloors = config.tuningFloors ?? true;
195
267
  this.#rawBody = config.rawBody ?? false;
196
268
  this.#servedModel = config.servedModel;
197
- this.#requiresMaxTokens = config.requiresMaxTokens;
198
- const reasoningReserve = this.reasoningReserve;
199
- if (this.#reasoningStyle === "template"
269
+ this.#requiresOutputBudget = config.requiresOutputBudget;
270
+ for (const [name, value] of [
271
+ ["contextWindow", this.#contextWindow],
272
+ ["maxInputTokens", this.#maxInputTokens],
273
+ ["maxOutputTokens", this.#maxOutputTokens],
274
+ ["outputBudget", this.#outputBudget],
275
+ ["reasoningBudget", this.#reasoningBudget],
276
+ ]) {
277
+ if (value !== null && (!Number.isSafeInteger(value) || value <= 0)) {
278
+ throw new Error(`${this.#source}: ${name} must be a positive safe integer or null`);
279
+ }
280
+ }
281
+ if (this.#reasoningBudget !== null
282
+ && this.#outputBudget !== null
283
+ && this.#reasoningBudget >= this.#outputBudget) {
284
+ throw new Error(`${this.#source}: reasoningBudget must be smaller than the total outputBudget`);
285
+ }
286
+ if (this.#reasoning.budget !== this.#reasoningBudget) {
287
+ throw new Error(`${this.#source}: reasoning intent and generation envelope disagree on reasoningBudget`);
288
+ }
289
+ if (this.#reasoningStyle === "anthropic"
290
+ && this.#reasoning.mode === "on"
291
+ && this.#reasoning.budget === null
292
+ && this.#reasoningBudget === null) {
293
+ throw new Error(`${this.#source}: explicit Anthropic reasoning requires PLURNK_PROVIDERS_REASONING_BUDGET`);
294
+ }
295
+ if (this.#additiveReasoningProvider !== undefined
200
296
  && this.#reasoning.mode === "on"
201
- && reasoningReserve !== null
202
- && this.#reasoning.budget > reasoningReserve) {
203
- throw new Error(`${this.#source}: PLURNK_PROVIDERS_REASONING_BUDGET (${this.#reasoning.budget}) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE (${reasoningReserve})`);
297
+ && this.#reasoningBudget === null) {
298
+ throw new Error(`${this.#source}: explicit ${this.#additiveReasoningProvider} reasoning requires PLURNK_PROVIDERS_REASONING_BUDGET so the total output budget remains bounded`);
299
+ }
300
+ if (this.#requiresOutputBudget === true && this.#outputBudget === null) {
301
+ throw new Error(`${this.#source}: this backend requires a resolved PLURNK_PROVIDERS_OUTPUT_BUDGET`);
204
302
  }
205
303
  const { tokenizeUrl } = config;
206
304
  if (tokenizeUrl !== undefined) {
@@ -209,7 +307,9 @@ export default class AiSdkProvider {
209
307
  method: "POST",
210
308
  headers: { "Content-Type": "application/json", ...this.#headers },
211
309
  body: JSON.stringify({ content: text }),
212
- signal: AbortSignal.timeout(this.#fetchTimeoutMs),
310
+ ...(this.#fetchTimeoutMs > 0
311
+ ? { signal: AbortSignal.timeout(this.#fetchTimeoutMs) }
312
+ : {}),
213
313
  });
214
314
  if (!res.ok)
215
315
  throw new Error(`${this.#source}: tokenize endpoint returned ${res.status}`);
@@ -222,22 +322,22 @@ export default class AiSdkProvider {
222
322
  }
223
323
  }
224
324
  get contextWindow() { return this.#contextWindow; }
225
- // {§provider-generation-envelope} Envelope reserves — absolute pins stand alone; percentages need the
226
- // detected window; null = underivable (no claim for core's no-cap path).
227
- #resolveReserve(spec) {
228
- if (spec === undefined)
229
- return null;
230
- if ("tokens" in spec)
231
- return spec.tokens;
232
- return this.#contextWindow === null ? null : Math.round(spec.percent * this.#contextWindow);
325
+ get maxInputTokens() { return this.#maxInputTokens; }
326
+ get maxOutputTokens() { return this.#maxOutputTokens; }
327
+ get outputBudget() { return this.#outputBudget; }
328
+ get reasoningBudget() { return this.#reasoningBudget; }
329
+ get inputCapacity() {
330
+ return effectiveInputCapacity({
331
+ contextWindow: this.#contextWindow,
332
+ maxInputTokens: this.#maxInputTokens,
333
+ outputBudget: this.#outputBudget,
334
+ });
233
335
  }
234
- get reasoningReserve() { return this.#resolveReserve(this.#reasoningReserve); }
235
- get completionReserve() { return this.#resolveReserve(this.#completionReserve); }
236
336
  get model() { return this.#model; }
237
337
  // Backend's self-reported served id; undefined when unprobed/unknown.
238
338
  get servedModel() { return this.#servedModel; }
239
339
  // Resolved "decodes unbounded without a cap" fact; undefined = no claim.
240
- get requiresMaxTokens() { return this.#requiresMaxTokens; }
340
+ get requiresOutputBudget() { return this.#requiresOutputBudget; }
241
341
  // Resolved capability: will a transported grammar actually constrain
242
342
  // this backend's decode? Introspectable so a consumer can verify the rails
243
343
  // are LIVE without spending a generation on a forcing-grammar probe.
@@ -248,7 +348,14 @@ export default class AiSdkProvider {
248
348
  }
249
349
  signal?.throwIfAborted();
250
350
  try {
251
- const timeout = AbortSignal.timeout(this.#fetchTimeoutMs);
351
+ const timeout = this.#fetchTimeoutMs > 0
352
+ ? AbortSignal.timeout(this.#fetchTimeoutMs)
353
+ : undefined;
354
+ const requestSignal = signal === undefined
355
+ ? timeout
356
+ : timeout === undefined
357
+ ? signal
358
+ : AbortSignal.any([signal, timeout]);
252
359
  const response = await this.#fetch(this.#promptTokensUrl, {
253
360
  method: "POST",
254
361
  headers: { "Content-Type": "application/json", ...this.#headers },
@@ -257,7 +364,7 @@ export default class AiSdkProvider {
257
364
  messages,
258
365
  ...this.#reasoningBody(),
259
366
  }),
260
- signal: signal === undefined ? timeout : AbortSignal.any([signal, timeout]),
367
+ ...(requestSignal === undefined ? {} : { signal: requestSignal }),
261
368
  });
262
369
  if (!response.ok) {
263
370
  return estimatePromptTokens(messages, `llama-server input-token endpoint returned HTTP ${response.status}`);
@@ -277,22 +384,38 @@ export default class AiSdkProvider {
277
384
  return estimatePromptTokens(messages, `llama-server input-token measurement failed: ${cause instanceof Error ? cause.message : String(cause)}`);
278
385
  }
279
386
  }
280
- calculateCost(usage) { return this.#calculateCost(usage); }
281
- calculateCharge(usage) {
282
- return this.#calculateCharge?.(usage)
283
- ?? { kind: "unknown", reason: "no provider rate or settled charge is available" };
387
+ async assessRequestCapacity(messages, maxOutputTokens, signal) {
388
+ const outputBudget = effectiveOutputBudget({
389
+ requested: maxOutputTokens,
390
+ configured: this.#outputBudget,
391
+ maxOutputTokens: this.#maxOutputTokens,
392
+ contextWindow: this.#contextWindow,
393
+ });
394
+ const reasoningBudget = effectiveReasoningBudget({
395
+ configured: this.#reasoningBudget,
396
+ outputBudget,
397
+ });
398
+ return assessRequestCapacity({
399
+ contextWindow: this.#contextWindow,
400
+ maxInputTokens: this.#maxInputTokens,
401
+ maxOutputTokens: this.#maxOutputTokens,
402
+ outputBudget,
403
+ reasoningBudget,
404
+ measurement: await this.countPromptTokens(messages, signal),
405
+ });
284
406
  }
285
407
  // Reasoning activation and allowance are independent of grammar transport;
286
408
  // only the response representation becomes lossless when evidence is needed.
287
409
  // The llama-server template mapping is owned by {§llama-reasoning-request}.
288
- #reasoningBody(preserveGrammarSentence = false) {
289
- const { mode, budget } = this.#reasoning;
410
+ #reasoningBody(preserveGrammarSentence = false, reasoningBudget = this.#reasoningBudget) {
411
+ const { mode } = this.#reasoning;
412
+ const budget = reasoningBudget;
290
413
  const on = mode !== "off";
291
414
  switch (this.#reasoningStyle) {
292
415
  case "template": {
293
416
  const allowance = mode === "off"
294
417
  ? 0
295
- : mode === "on" ? budget : this.reasoningReserve;
418
+ : mode === "on" && budget !== null ? budget : this.#reasoningBudget;
296
419
  return {
297
420
  chat_template_kwargs: { enable_thinking: on },
298
421
  reasoning_format: preserveGrammarSentence ? "none" : "auto",
@@ -301,9 +424,9 @@ export default class AiSdkProvider {
301
424
  }
302
425
  case "think": return on ? { think: true } : {};
303
426
  case "include_reasoning": return on ? { include_reasoning: true } : {};
304
- // effort tiers from the budget; off/adaptive omit the field (the
305
- // API's default depth is its adaptive).
306
- case "effort": return mode === "on" ? { reasoning_effort: effortFromBudget(budget) } : {};
427
+ // Explicit on uses the portable enabled posture or a tier derived
428
+ // from an explicit budget; off/adaptive omit the field.
429
+ case "effort": return mode === "on" ? { reasoning_effort: effortFromReasoning({ mode, budget }) } : {};
307
430
  // Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
308
431
  // reason-by-default model (DeepSeek V4: default 'high') reasoning.
309
432
  // ADAPTIVE omits the field: the backend's own default posture IS the
@@ -313,19 +436,25 @@ export default class AiSdkProvider {
313
436
  // efforts 400.
314
437
  case "effort_explicit": return mode === "off"
315
438
  ? { reasoning_effort: "none" }
316
- : mode === "on" ? { reasoning_effort: effortFromBudget(budget) } : {};
439
+ : mode === "on" ? { reasoning_effort: effortFromReasoning({ mode, budget }) } : {};
317
440
  // {§deepseek-reasoning-request}
318
441
  case "thinking_effort": return mode === "off"
319
442
  ? { thinking: { type: "disabled" } }
320
443
  : mode === "on" ? {
321
444
  thinking: { type: "enabled" },
322
- reasoning_effort: effortFromBudget(budget),
445
+ ...(budget === null ? {} : { reasoning_effort: effortFromBudget(budget) }),
323
446
  } : {};
324
447
  // Anthropic compat: explicit thinking object. off → disabled; on →
325
- // enabled with budget_tokens; adaptive → omit (the API default).
448
+ // enabled with the explicit reasoning subset; adaptive →
449
+ // omit (the API default).
326
450
  case "anthropic": return mode === "off"
327
451
  ? { thinking: { type: "disabled" } }
328
- : mode === "on" ? { thinking: { type: "enabled", budget_tokens: budget } } : {};
452
+ : mode === "on" ? {
453
+ thinking: {
454
+ type: "enabled",
455
+ budget_tokens: budget,
456
+ },
457
+ } : {};
329
458
  case "none": return {};
330
459
  }
331
460
  }
@@ -390,7 +519,7 @@ export default class AiSdkProvider {
390
519
  case "none": return this.#frequencyPenalty > 0 ? { frequency_penalty: this.#frequencyPenalty } : {};
391
520
  }
392
521
  }
393
- // First-party telemetry headers ({§provider-request-authority}): forwarded only when the spec
522
+ // First-party telemetry headers ({§provider-request-authority} {§provider-call-kind}): forwarded only when the spec
394
523
  // opted in (the plurnk endpoint). The gate is here, not at the call site, so
395
524
  // attributions/client/strikes can never reach a third-party backend even if
396
525
  // the consumer passes them to the wrong provider. Empty values emit no header
@@ -398,7 +527,7 @@ export default class AiSdkProvider {
398
527
  // absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
399
528
  // ride HTTP headers only — the packet never carries them (the model must
400
529
  // never see strike state; engine accounting is not a metric to game).
401
- #metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn) {
530
+ #metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind) {
402
531
  if (!this.#firstPartyMetadata)
403
532
  return {};
404
533
  const h = {};
@@ -431,6 +560,8 @@ export default class AiSdkProvider {
431
560
  h["Plurnk-Loop"] = String(loop);
432
561
  if (turn !== undefined && Number.isInteger(turn) && turn >= 1)
433
562
  h["Plurnk-Turn"] = String(turn);
563
+ if (callKind !== undefined)
564
+ h["Plurnk-Call-Kind"] = callKind;
434
565
  return h;
435
566
  }
436
567
  // PLURNK_PROVIDERS_GBNF_DEBUG ({§gbnf-response-observation}): validate the supplied GBNF locally and fail
@@ -470,10 +601,57 @@ export default class AiSdkProvider {
470
601
  out[k] = v;
471
602
  return out;
472
603
  }
473
- async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }) {
604
+ #requestProviderOptions(workerId, reasoningBudget) {
605
+ const responseOptions = this.#reasoning.mode === "off"
606
+ ? undefined
607
+ : this.#reasoningResponseProviderOptions;
608
+ const nativeReasoning = this.#reasoning.mode === "on" && reasoningBudget !== null
609
+ ? this.#additiveReasoningProvider === "anthropic"
610
+ ? { anthropic: { thinking: { type: "enabled", budgetTokens: reasoningBudget } } }
611
+ : this.#additiveReasoningProvider === "bedrock"
612
+ ? { bedrock: { reasoningConfig: { type: "enabled", budgetTokens: reasoningBudget } } }
613
+ : undefined
614
+ : undefined;
615
+ const options = {};
616
+ for (const part of [responseOptions, nativeReasoning]) {
617
+ for (const [provider, values] of Object.entries(part ?? {})) {
618
+ options[provider] = { ...options[provider], ...values };
619
+ }
620
+ }
621
+ if (this.#cacheAffinity?.target === "provider-option") {
622
+ const { provider, name } = this.#cacheAffinity;
623
+ options[provider] = { ...options[provider], [name]: workerId };
624
+ }
625
+ return Object.keys(options).length === 0 ? undefined : options;
626
+ }
627
+ #nativeMaxOutputTokens(outputBudget, reasoningBudget) {
628
+ if (outputBudget === null)
629
+ return undefined;
630
+ return this.#additiveReasoningProvider !== undefined
631
+ && this.#reasoning.mode === "on"
632
+ && reasoningBudget !== null
633
+ ? outputBudget - reasoningBudget
634
+ : outputBudget;
635
+ }
636
+ #accounting(outcome, usage, evidence, status) {
637
+ const knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
638
+ const direct = this.#normalizeCost?.(evidence);
639
+ return validateProviderRequestAccounting({
640
+ provider: this.#source,
641
+ model: this.#model,
642
+ outcome,
643
+ ...(status === undefined ? {} : { status }),
644
+ ...(knownUsage === undefined ? {} : { usage: knownUsage }),
645
+ cost: resolveProviderCost(direct, this.#estimateCost(knownUsage)),
646
+ });
647
+ }
648
+ async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxOutputTokens, attributions, client, strikes, workspaceId, loop, turn, sampling, observeRequest, callKind }) {
474
649
  // {§provider-interface} The worker identity is required.
475
650
  if (workerId === undefined || workerId.length === 0)
476
651
  throw new Error("generate: workerId is required — the worker's stable, opaque identity");
652
+ if (callKind !== undefined && callKind !== "emission" && callKind !== "bare") {
653
+ throw new Error(`generate: unsupported callKind ${JSON.stringify(callKind)}`);
654
+ }
477
655
  // Reject before any wire call when already aborted
478
656
  // ({§provider-failure-normalization}).
479
657
  signal?.throwIfAborted();
@@ -485,6 +663,14 @@ export default class AiSdkProvider {
485
663
  const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
486
664
  const preserveGrammarSentence = wantGrammar
487
665
  && this.#reasoningStyle === "template";
666
+ const capacity = await this.assessRequestCapacity(messages, maxOutputTokens, signal);
667
+ if (capacity.decision === "reject") {
668
+ if (capacity.prompt.kind !== "exact") {
669
+ throw new TypeError(`${this.#source}: only an exact prompt measurement may reject capacity`);
670
+ }
671
+ throw new ProviderError(this.#source, "capacity_exceeded", `The exact provider request uses ${capacity.prompt.tokens} input tokens, exceeding its ${capacity.inputCapacity} token input capacity.`, { capacity, extensions: { capacityStage: "preflight", capacity } });
672
+ }
673
+ const effectiveMaxOutputTokens = capacity.outputBudget ?? undefined;
488
674
  // Assembly order = precedence: the family's sampling DEFAULTS
489
675
  // (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
490
676
  // paths and the name promises every request) < the caller's `sampling`
@@ -497,80 +683,188 @@ export default class AiSdkProvider {
497
683
  ...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
498
684
  model: this.#model,
499
685
  messages,
500
- ...this.#reasoningBody(preserveGrammarSentence),
686
+ ...this.#reasoningBody(preserveGrammarSentence, capacity.reasoningBudget),
501
687
  ...this.#grammarBody(sendGrammar),
502
- ...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
688
+ ...(effectiveMaxOutputTokens !== undefined ? { max_tokens: effectiveMaxOutputTokens } : {}),
503
689
  // Request per-token logprobs only when enabled (managed field —
504
690
  // reserved from caller sampling; the env flag is the single control).
505
691
  ...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
506
692
  ...this.#slotBody(workerId),
507
- // Prompt-cache affinity -- workerId as the OpenAI-standard
508
- // prompt_cache_key routes a worker's turns to one serverless replica so
509
- // its stable prefix caches (managed; reserved from caller sampling).
510
- ...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
693
+ ...(this.#cacheAffinity?.target === "body"
694
+ ? { [this.#cacheAffinity.name]: workerId }
695
+ : {}),
511
696
  };
512
697
  // Per-request headers = static auth/routing + any first-party telemetry.
513
- const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
514
- const headers = Object.keys(metaHeaders).length === 0
515
- ? this.#headers
516
- : { ...this.#headers, ...metaHeaders };
517
- let raw;
518
- try {
519
- raw = this.#languageModel === undefined
520
- ? await executeOpenAICompatible({
521
- url: this.#url,
698
+ const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind);
699
+ const headers = new Headers(this.#headers);
700
+ if (this.#cacheAffinity?.target === "header") {
701
+ headers.set(this.#cacheAffinity.name, workerId);
702
+ }
703
+ for (const [name, value] of Object.entries(metaHeaders))
704
+ headers.set(name, value);
705
+ const requestHeaders = Object.fromEntries(headers.entries());
706
+ const accounting = [];
707
+ const operationTimeout = this.#operationTimeoutMs > 0
708
+ ? AbortSignal.timeout(this.#operationTimeoutMs)
709
+ : undefined;
710
+ const operationSignal = signal === undefined
711
+ ? operationTimeout
712
+ : operationTimeout === undefined
713
+ ? signal
714
+ : AbortSignal.any([signal, operationTimeout]);
715
+ const executeRequest = async () => {
716
+ let settle;
717
+ try {
718
+ settle = await observeRequest?.({
719
+ provider: this.#source,
522
720
  model: this.#model,
523
- headers,
524
- body,
525
- messages,
526
- signal,
527
- fetch: this.#fetch,
528
- fetchTimeoutMs: this.#fetchTimeoutMs,
529
- streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
530
- retryAttempts: this.#retryAttempts,
531
- streaming: this.#streaming,
532
- captureRawBody: this.#rawBody,
533
- })
534
- : await executeAiSdkModel({
535
- languageModel: this.#languageModel,
536
- headers,
537
- messages,
538
- signal,
539
- fetchTimeoutMs: this.#fetchTimeoutMs,
540
- streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
541
- retryAttempts: this.#retryAttempts,
542
- streaming: this.#streaming,
543
- captureRawBody: this.#rawBody,
544
- temperature: this.#tuningFloors
545
- ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
546
- : typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
547
- topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
548
- topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
549
- presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
550
- frequencyPenalty: typeof sampling?.frequency_penalty === "number"
551
- ? sampling.frequency_penalty
552
- : this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
553
- stopSequences: typeof sampling?.stop === "string"
554
- ? [sampling.stop]
555
- : Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
556
- ? sampling.stop
557
- : undefined,
558
- seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
559
- maxOutputTokens: maxTokens,
560
- reasoning: this.#reasoning.mode === "off"
561
- ? "none"
562
- : this.#reasoning.mode === "adaptive"
563
- ? "provider-default"
564
- : effortFromBudget(this.#reasoning.budget),
565
721
  });
722
+ }
723
+ catch (cause) {
724
+ throw new ProviderRequestObserverError(cause);
725
+ }
726
+ const settleAccounting = async (outcome, usage, evidence, status) => {
727
+ let requestAccounting;
728
+ let normalizationFailure;
729
+ try {
730
+ requestAccounting = this.#accounting(outcome, usage, evidence, status);
731
+ }
732
+ catch (cause) {
733
+ normalizationFailure = { cause };
734
+ let knownUsage;
735
+ try {
736
+ knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
737
+ }
738
+ catch {
739
+ knownUsage = undefined;
740
+ }
741
+ requestAccounting = validateProviderRequestAccounting({
742
+ provider: this.#source,
743
+ model: this.#model,
744
+ outcome,
745
+ ...(status === undefined ? {} : { status }),
746
+ ...(knownUsage === undefined ? {} : { usage: knownUsage }),
747
+ cost: {
748
+ kind: "unknown",
749
+ reason: "provider request accounting could not be normalized after physical I/O",
750
+ },
751
+ });
752
+ }
753
+ accounting.push(requestAccounting);
754
+ try {
755
+ await settle?.(requestAccounting);
756
+ }
757
+ catch (cause) {
758
+ throw new ProviderRequestObserverError(cause);
759
+ }
760
+ if (normalizationFailure !== undefined) {
761
+ throw new ProviderRequestAccountingError(normalizationFailure.cause);
762
+ }
763
+ return requestAccounting;
764
+ };
765
+ let response;
766
+ try {
767
+ response = this.#languageModel === undefined
768
+ ? await executeOpenAICompatible({
769
+ url: this.#url,
770
+ model: this.#model,
771
+ headers: requestHeaders,
772
+ body,
773
+ messages,
774
+ signal: operationSignal,
775
+ fetch: this.#fetch,
776
+ fetchTimeoutMs: this.#fetchTimeoutMs,
777
+ firstContentTimeoutMs: this.#firstContentTimeoutMs,
778
+ streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
779
+ streaming: this.#streaming,
780
+ captureRawBody: this.#rawBody,
781
+ })
782
+ : await executeAiSdkModel({
783
+ languageModel: this.#languageModel,
784
+ headers: requestHeaders,
785
+ providerOptions: this.#requestProviderOptions(workerId, capacity.reasoningBudget),
786
+ systemProviderOptions: this.#systemCacheProviderOptions,
787
+ messages,
788
+ signal: operationSignal,
789
+ fetchTimeoutMs: this.#fetchTimeoutMs,
790
+ firstContentTimeoutMs: this.#firstContentTimeoutMs,
791
+ streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
792
+ streaming: this.#streaming,
793
+ captureRawBody: this.#rawBody,
794
+ temperature: this.#tuningFloors
795
+ ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
796
+ : typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
797
+ topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
798
+ topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
799
+ presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
800
+ frequencyPenalty: typeof sampling?.frequency_penalty === "number"
801
+ ? sampling.frequency_penalty
802
+ : this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
803
+ stopSequences: typeof sampling?.stop === "string"
804
+ ? [sampling.stop]
805
+ : Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
806
+ ? sampling.stop
807
+ : undefined,
808
+ seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
809
+ maxOutputTokens: this.#nativeMaxOutputTokens(capacity.outputBudget, capacity.reasoningBudget),
810
+ reasoning: this.#reasoning.mode === "off"
811
+ ? "none"
812
+ : this.#reasoning.mode === "adaptive"
813
+ ? "provider-default"
814
+ : this.#additiveReasoningProvider !== undefined && capacity.reasoningBudget !== null
815
+ ? "provider-default"
816
+ : effortFromReasoning({
817
+ mode: this.#reasoning.mode,
818
+ budget: capacity.reasoningBudget,
819
+ }),
820
+ });
821
+ }
822
+ catch (error) {
823
+ const failure = transportFailureEvidence(error);
824
+ await settleAccounting("error", failure.usage, failure.chargeEvidence, failure.status);
825
+ throw error;
826
+ }
827
+ await settleAccounting("response", response.usage, response.chargeEvidence);
828
+ return response;
829
+ };
830
+ let raw;
831
+ try {
832
+ const { retry } = prepareRetries({
833
+ maxRetries: this.#retryAttempts,
834
+ abortSignal: operationSignal,
835
+ });
836
+ raw = await retry(executeRequest);
566
837
  }
567
838
  catch (err) {
839
+ if (err instanceof ProviderRequestObserverError
840
+ || err instanceof ProviderRequestAccountingError)
841
+ throw err.cause;
568
842
  if (signal?.aborted)
569
843
  throw err;
570
- const pe = toProviderError(err, this.#source, this.#errorDetailLimit);
844
+ if (operationTimeout?.aborted) {
845
+ const timeout = new ProviderTimeoutError("operation", this.#operationTimeoutMs, err);
846
+ throw new ProviderError(this.#source, "deadline_exceeded", timeout.message, {
847
+ status: 504,
848
+ cause: timeout,
849
+ retryable: false,
850
+ extensions: {
851
+ timeoutPhase: timeout.phase,
852
+ timeoutMs: timeout.timeoutMs,
853
+ },
854
+ accounting,
855
+ capacity,
856
+ });
857
+ }
858
+ const pe = toProviderError(err, this.#source, this.#errorDetailLimit, capacity);
571
859
  if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
572
- throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
860
+ throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, {
861
+ status: pe.status,
862
+ cause: err,
863
+ accounting,
864
+ capacity,
865
+ });
573
866
  }
867
+ pe.prependAccounting(accounting);
574
868
  throw pe;
575
869
  }
576
870
  // llama-server --special renders EOG tokens as text, so a turn ending
@@ -585,10 +879,10 @@ export default class AiSdkProvider {
585
879
  const projectedReasoning = preserveGrammarSentence && !raw.reasoningProjected
586
880
  ? projectTemplateReasoning(raw.content)
587
881
  : projectTaggedReasoning(raw.content, raw.reasoning, this.#reasoningResponseStyle);
588
- // Preserve the exact sentence seen at the grammar boundary. Constrained
589
- // template turns request `reasoning_format: "none"`, so even an empty
590
- // channel remains observable. An unexpectedly projected response cannot
591
- // supply independent pre-projection evidence.
882
+ // Preserve the exact pre-projection response. Constrained template turns
883
+ // request `reasoning_format: "none"`, so even an empty channel and any
884
+ // template-provided opener remain observable. An unexpectedly projected
885
+ // response cannot supply independent evidence.
592
886
  let grammarEvidence;
593
887
  if (wantGrammar) {
594
888
  if (preserveGrammarSentence) {
@@ -618,11 +912,12 @@ export default class AiSdkProvider {
618
912
  if (projectedReasoning.projected) {
619
913
  raw.content = projectedReasoning.content;
620
914
  raw.reasoning = projectedReasoning.reasoning;
621
- raw.usage = attributeUnitemizedReasoning(raw.usage, raw.reasoning, raw.content);
622
915
  }
623
916
  let notices;
624
917
  const usage = raw.usage;
625
- if (sendGrammar !== undefined && this.tokenize !== undefined) {
918
+ if (sendGrammar !== undefined
919
+ && this.tokenize !== undefined
920
+ && usage?.outputTokens !== undefined) {
626
921
  // Channel-escape detector: completion tokens
627
922
  // billed far beyond every visible channel mean the decode ESCAPED into
628
923
  // a server-discarded reasoning block mid-emission. This diagnostic
@@ -633,12 +928,12 @@ export default class AiSdkProvider {
633
928
  this.tokenize(raw.reasoning),
634
929
  ]);
635
930
  const visible = contentTokens.length + reasoningTokens.length;
636
- if (usage.completion > visible + 64) {
931
+ if (usage.outputTokens > visible + 64) {
637
932
  (notices ??= []).push({
638
933
  source: this.#source,
639
934
  kind: "grammar_unenforced",
640
935
  level: "warn",
641
- message: `decode escaped the grammar: ${usage.completion} completion tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
936
+ message: `decode escaped the grammar: ${usage.outputTokens} output tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
642
937
  position: [...raw.content].length,
643
938
  });
644
939
  }
@@ -658,22 +953,35 @@ export default class AiSdkProvider {
658
953
  ...(raw.reasoningEncrypted.length > 0
659
954
  ? { reasoningEncrypted: raw.reasoningEncrypted }
660
955
  : {}),
661
- usage,
662
956
  model: raw.model,
663
957
  ...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
664
958
  };
665
- const normalizedCharge = this.#normalizeCharge?.(raw.chargeEvidence);
666
- const charge = normalizedCharge === undefined
667
- ? undefined
668
- : validateAuthoritativeCharge(normalizedCharge);
669
959
  const evidence = {
670
960
  assistantRaw: raw,
671
- ...(charge === undefined ? {} : { charge }),
961
+ accounting,
962
+ capacity,
672
963
  ...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
673
964
  ...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
674
965
  ...(meta !== undefined ? { meta } : {}),
675
966
  ...(notices !== undefined ? { notices } : {}),
676
967
  };
968
+ if (capacity.outputBudget !== null
969
+ && usage?.outputTokens !== undefined
970
+ && usage.outputTokens > capacity.outputBudget) {
971
+ const attempt = {
972
+ assistant: { ...assistant, finishReason: raw.finishReason },
973
+ ...evidence,
974
+ };
975
+ throw new ProviderError(this.#source, "invalid_response", `The provider reported ${usage.outputTokens} output tokens after receiving a total output budget of ${capacity.outputBudget}.`, {
976
+ attempt,
977
+ accounting,
978
+ extensions: {
979
+ stage: "provider-response",
980
+ outputBudget: capacity.outputBudget,
981
+ reportedOutputTokens: usage.outputTokens,
982
+ },
983
+ });
984
+ }
677
985
  if (raw.finishReason === "resource_interrupted") {
678
986
  const attempt = {
679
987
  assistant: { ...assistant, finishReason: raw.finishReason },
@@ -681,6 +989,7 @@ export default class AiSdkProvider {
681
989
  };
682
990
  throw new ProviderError(this.#source, "resource_interrupted", "The provider interrupted generation because inference resources were unavailable.", {
683
991
  attempt,
992
+ accounting,
684
993
  extensions: {
685
994
  stage: "provider-response",
686
995
  finishReason: "resource_interrupted",