@plurnk/plurnk-providers 1.5.0 → 1.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/.env.defaults +41 -34
  2. package/README.md +15 -0
  3. package/SPEC.md +242 -89
  4. package/dist/AiSdkProvider.d.ts +33 -33
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +442 -133
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +10 -11
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +87 -25
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +9 -24
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +86 -25
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/accounting.d.ts +5 -2
  17. package/dist/accounting.d.ts.map +1 -1
  18. package/dist/accounting.js +100 -16
  19. package/dist/accounting.js.map +1 -1
  20. package/dist/accountingPublic.d.ts +5 -0
  21. package/dist/accountingPublic.d.ts.map +1 -0
  22. package/dist/accountingPublic.js +3 -0
  23. package/dist/accountingPublic.js.map +1 -0
  24. package/dist/aiSdkTransport.d.ts +9 -2
  25. package/dist/aiSdkTransport.d.ts.map +1 -1
  26. package/dist/aiSdkTransport.js +160 -62
  27. package/dist/aiSdkTransport.js.map +1 -1
  28. package/dist/capacity.d.ts +26 -0
  29. package/dist/capacity.d.ts.map +1 -0
  30. package/dist/capacity.js +90 -0
  31. package/dist/capacity.js.map +1 -0
  32. package/dist/catalogProvider.d.ts +8 -3
  33. package/dist/catalogProvider.d.ts.map +1 -1
  34. package/dist/catalogProvider.js +45 -41
  35. package/dist/catalogProvider.js.map +1 -1
  36. package/dist/compatibleProvider.d.ts.map +1 -1
  37. package/dist/compatibleProvider.js +26 -12
  38. package/dist/compatibleProvider.js.map +1 -1
  39. package/dist/cost.d.ts +10 -10
  40. package/dist/cost.d.ts.map +1 -1
  41. package/dist/cost.js +90 -42
  42. package/dist/cost.js.map +1 -1
  43. package/dist/env.d.ts +13 -11
  44. package/dist/env.d.ts.map +1 -1
  45. package/dist/env.js +83 -46
  46. package/dist/env.js.map +1 -1
  47. package/dist/errors.d.ts +17 -3
  48. package/dist/errors.d.ts.map +1 -1
  49. package/dist/errors.js +91 -8
  50. package/dist/errors.js.map +1 -1
  51. package/dist/index.d.ts +7 -6
  52. package/dist/index.d.ts.map +1 -1
  53. package/dist/index.js +5 -3
  54. package/dist/index.js.map +1 -1
  55. package/dist/ollama.js +3 -3
  56. package/dist/ollama.js.map +1 -1
  57. package/dist/promptTokens.d.ts.map +1 -1
  58. package/dist/promptTokens.js +7 -4
  59. package/dist/promptTokens.js.map +1 -1
  60. package/dist/sdkModels.d.ts +7 -2
  61. package/dist/sdkModels.d.ts.map +1 -1
  62. package/dist/sdkModels.js +43 -13
  63. package/dist/sdkModels.js.map +1 -1
  64. package/dist/types.d.ts +55 -33
  65. package/dist/types.d.ts.map +1 -1
  66. package/dist/usage.d.ts +22 -5
  67. package/dist/usage.d.ts.map +1 -1
  68. package/dist/usage.js +169 -83
  69. package/dist/usage.js.map +1 -1
  70. package/package.json +18 -7
  71. package/src/AiSdkProvider.test.ts +964 -206
  72. package/src/AiSdkProvider.ts +545 -155
  73. package/src/Mock.test.ts +69 -30
  74. package/src/Mock.ts +99 -29
  75. package/src/Pool.test.ts +90 -19
  76. package/src/Pool.ts +96 -27
  77. package/src/ProviderRegistry.test.ts +16 -11
  78. package/src/accounting.test.ts +58 -22
  79. package/src/accounting.ts +119 -18
  80. package/src/accountingPublic.ts +9 -0
  81. package/src/aiSdkTransport.test.ts +42 -49
  82. package/src/aiSdkTransport.ts +174 -62
  83. package/src/boundaries.test.ts +2 -0
  84. package/src/capacity.test.ts +92 -0
  85. package/src/capacity.ts +140 -0
  86. package/src/catalogProvider.test.ts +339 -30
  87. package/src/catalogProvider.ts +65 -47
  88. package/src/compatibleProvider.test.ts +7 -5
  89. package/src/compatibleProvider.ts +29 -13
  90. package/src/cost.test.ts +86 -36
  91. package/src/cost.ts +111 -50
  92. package/src/defaults.test.ts +13 -3
  93. package/src/env.test.ts +103 -25
  94. package/src/env.ts +153 -65
  95. package/src/errors.test.ts +80 -2
  96. package/src/errors.ts +107 -8
  97. package/src/index.ts +26 -7
  98. package/src/ollama.test.ts +5 -3
  99. package/src/ollama.ts +3 -3
  100. package/src/promptTokens.ts +8 -5
  101. package/src/sdkModels.test.ts +77 -8
  102. package/src/sdkModels.ts +51 -15
  103. package/src/types.ts +112 -51
  104. package/src/usage.test.ts +112 -116
  105. package/src/usage.ts +214 -93
@@ -6,19 +6,42 @@
6
6
  // ordinary vendor protocol. The compatible URL path remains only for PLURNK
7
7
  // extensions and local endpoint probes the SDK cannot represent.
8
8
 
9
- import type { AuthoritativeChargeNormalizer, ChatMessage, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
9
+ import type {
10
+ ChatMessage,
11
+ GrammarEvidence,
12
+ PromptTokenMeasurement,
13
+ Provider,
14
+ ProviderAttempt,
15
+ ProviderCostNormalizer,
16
+ ProviderCallKind,
17
+ ProviderGenerateArgs,
18
+ ProviderRequestAccounting,
19
+ ProviderRequestCapacity,
20
+ ProviderRequestSettlement,
21
+ ProviderResponse,
22
+ ProviderUsage,
23
+ } from "./types.ts";
10
24
  import type { ProviderCost } from "@plurnk/plurnk-contracts";
11
- import type { Reasoning, ReasoningResponseStyle, ReserveSpec } from "./env.ts";
12
- import { executeAiSdkModel, executeOpenAICompatible } from "./aiSdkTransport.ts";
25
+ import type { JSONValue } from "ai";
26
+ import { MAX_PROVIDER_TIMEOUT_MS } from "./env.ts";
27
+ import type { Reasoning, ReasoningResponseStyle } from "./env.ts";
28
+ import {
29
+ executeAiSdkModel,
30
+ executeOpenAICompatible,
31
+ transportFailureEvidence,
32
+ } from "./aiSdkTransport.ts";
13
33
  import type { LanguageModel } from "ai";
14
- import { toProviderError, ProviderError } from "./errors.ts";
15
- import { attributeUnitemizedReasoning } from "./usage.ts";
34
+ import { prepareRetries } from "ai/internal";
35
+ import { toProviderError, ProviderError, ProviderTimeoutError } from "./errors.ts";
16
36
  import type { ProviderNotice } from "./notices.ts";
17
37
  import { validateGbnf } from "@plurnk/gbnf";
18
38
  import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.ts";
19
39
  import { emitWarningOnce } from "./warnings.ts";
20
40
  import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
21
- import { validateAuthoritativeCharge } from "./cost.ts";
41
+ import { resolveProviderCost } from "./cost.ts";
42
+ import { validateProviderRequestAccounting } from "./accounting.ts";
43
+ import { validateProviderUsage } from "./usage.ts";
44
+ import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.ts";
22
45
 
23
46
  export type ProviderFetch = typeof globalThis.fetch;
24
47
 
@@ -30,30 +53,49 @@ export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" |
30
53
  // service-managed constrained sampling; endpoint-owned settings are not inferred.
31
54
  export type GrammarStyle = "none" | "llamacpp";
32
55
 
56
+ export type CacheAffinity =
57
+ | { readonly target: "header" | "body"; readonly name: string }
58
+ | { readonly target: "provider-option"; readonly provider: string; readonly name: string };
59
+
60
+ export type AiSdkProviderOptions = Record<string, Record<string, JSONValue | undefined>>;
61
+
33
62
  export type AiSdkProviderConfig = {
34
63
  model: string;
35
64
  url?: string; // OpenAI-compatible chat-completions URL
36
65
  languageModel?: LanguageModel; // native AI SDK provider model
37
66
  attributions?: (context: PluginAttributionContext) => PluginAttribution;
38
- fetchTimeoutMs: number;
39
- streamIdleTimeoutMs?: number; // streamed body inter-chunk deadline; zero/unset disables
67
+ fetchTimeoutMs: number; // one physical generation attempt; zero disables
68
+ operationTimeoutMs: number; // complete logical call across retries/backoff; zero disables
69
+ firstContentTimeoutMs: number; // first semantic streamed content; zero disables
70
+ streamIdleTimeoutMs?: number; // semantic streamed-content idle deadline; zero/unset disables
40
71
  headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
41
72
  fetch?: ProviderFetch; // per-instance request executor; default globalThis.fetch
42
73
  contextWindow?: number | null; // default null; caller resolves-or-fails, narrows to required with the interface
74
+ maxInputTokens?: number | null;
75
+ maxOutputTokens?: number | null;
76
+ outputBudget?: number | null;
77
+ reasoningBudget?: number | null;
78
+ // Native Anthropic and Bedrock SDKs interpret generic maxOutputTokens as
79
+ // visible output and add an explicit provider reasoning budget. This marker lets the
80
+ // adapter subtract that subset so the resulting wire cap remains PLURNK's
81
+ // one total output budget.
82
+ additiveReasoningProvider?: "anthropic" | "bedrock";
43
83
  reasoningStyle?: ReasoningStyle; // default "none"
44
84
  reasoningResponseStyle?: ReasoningResponseStyle; // {§provider-tagged-reasoning}; default "verbatim"
45
85
  countPromptTokens?: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
46
- calculateCost?: (usage: ProviderUsage) => number; // default () => 0
47
- calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }>;
48
- normalizeCharge?: AuthoritativeChargeNormalizer;
86
+ estimateCost?: (usage: ProviderUsage | undefined) => ProviderCost;
87
+ normalizeCost?: ProviderCostNormalizer;
49
88
  source?: string; // notice/problem source, e.g. "provider:openai"; default "provider"
50
89
  grammarStyle?: GrammarStyle; // how a GBNF grammar is carried; default "none" (not sent)
51
- // Send the OpenAI-standard `prompt_cache_key` set to workerId, so a
52
- // serverless backend with REPLICA-LOCAL prompt caching (fireworks, verified)
53
- // pins a worker's turns to one replica and claims its stable prefix. Default
54
- // false -- a backend that strict-validates unknown fields 400s, so enable only
55
- // where the field is accepted. Same identity that already drives slot affinity.
56
- promptCacheKey?: boolean;
90
+ // {§provider-cache-affinity} Provider routes own the exact documented
91
+ // projection; the common transport only applies it as managed request state.
92
+ cacheAffinity?: CacheAffinity;
93
+ // {§provider-cache-write-policy} Already policy-gated by provider construction.
94
+ // The transport attaches it to only the final leading system instruction.
95
+ systemCacheProviderOptions?: AiSdkProviderOptions;
96
+ // {§provider-readable-reasoning} Route-owned native option needed to expose
97
+ // readable reasoning. Applied only when the effective posture is not off.
98
+ reasoningResponseProviderOptions?: AiSdkProviderOptions;
57
99
  // Optional provider-configured service tier. Unlike caller sampling, this is
58
100
  // a fixed deployment choice and therefore wins on every request.
59
101
  serviceTier?: string;
@@ -78,14 +120,14 @@ export type AiSdkProviderConfig = {
78
120
  // when no probe ran or it read no row.
79
121
  servedModel?: string;
80
122
  // Backend decodes unbounded without a caller cap (llama-server n_predict
81
- // to the wall) — surfaced as Provider.requiresMaxTokens so consumers can
123
+ // to the wall) — surfaced as Provider.requiresOutputBudget so consumers can
82
124
  // boot-refuse an envelope-less local alias. Default unset (no claim).
83
- requiresMaxTokens?: boolean;
125
+ requiresOutputBudget?: boolean;
84
126
  // The side-channel reasoning intent — REQUIRED, no in-code default
85
127
  // (PLURNK_PROVIDERS_REASONING + _BUDGET, read via reasoningFromEnv):
86
- // { mode: off|adaptive|on, budget: iff on }. The provider maps it to the
87
- // backend's mechanism via reasoningStyle; budget is only ever a magnitude,
88
- // never a hidden activation flag.
128
+ // { mode: off|adaptive|on, budget: optional when on }. The provider maps it
129
+ // to the backend's mechanism via reasoningStyle; budget is only ever an
130
+ // explicit magnitude, never a hidden activation flag.
89
131
  reasoning: Reasoning;
90
132
  // Decode tuning: no in-code defaults; the canonical measured values (0.2 /
91
133
  // 1.15) ship as the floor in .env.defaults (alias-scopable). `temperature` is the
@@ -126,20 +168,32 @@ export type AiSdkProviderConfig = {
126
168
  // gated per-alias.
127
169
  topLogprobs?: number | null;
128
170
  rawBody?: boolean;
129
- // {§provider-generation-envelope} The generation-envelope reserves, env-read via
130
- // envelopeFromEnv — a percentage of the DETECTED window or an absolute token
171
+ // {§provider-generation-envelope} The generation budgets, env-read via the
172
+ // common envelope parser — a percentage of the detected window or an absolute token
131
173
  // count. Optional so an out-of-date sibling keeps constructing (no claim);
132
174
  // the standard factory always supplies them. Resolved against contextWindow
133
175
  // at read time (getters), so a probe that lands after config assembly still
134
176
  // derives correctly.
135
- reasoningReserve?: ReserveSpec;
136
- completionReserve?: ReserveSpec;
137
177
  // The plurnk.ai router owns tuning — false suppresses the
138
178
  // client-side temperature/penalty FLOORS on this provider (caller `sampling`
139
179
  // still passes through verbatim). Default true (floors ride).
140
180
  tuningFloors?: boolean;
141
181
  };
142
182
 
183
+ class ProviderRequestObserverError extends Error {
184
+ constructor(cause: unknown) {
185
+ super("provider request accounting could not be durably settled", { cause });
186
+ this.name = "ProviderRequestObserverError";
187
+ }
188
+ }
189
+
190
+ class ProviderRequestAccountingError extends Error {
191
+ constructor(cause: unknown) {
192
+ super("provider request accounting could not be normalized", { cause });
193
+ this.name = "ProviderRequestAccountingError";
194
+ }
195
+ }
196
+
143
197
  // Drop trailing occurrences of a server-rendered EOG marker. llama-server
144
198
  // under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
145
199
  // trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
@@ -197,12 +251,19 @@ const projectTaggedReasoning = (
197
251
  ? projectLeadingReasoning(content, structuredReasoning, "<think>", "</think>")
198
252
  : { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
199
253
 
200
- // llama-server's template reasoning parser can project this leading channel out
201
- // of the OpenAI-compatible response. Grammar evidence needs the sentence before
202
- // that lossy projection, so constrained template turns request it verbatim and
203
- // split the observed enclosure here.
204
- const projectTemplateReasoning = (content: string): TaggedReasoningProjection =>
205
- projectLeadingReasoning(content, "", "<|channel>thought\n", "<channel|>");
254
+ // llama-server's template reasoning parser can project either supported leading
255
+ // reasoning envelope out of the OpenAI-compatible response. Grammar evidence
256
+ // needs the sentence before that lossy projection, so constrained template turns
257
+ // request it verbatim and split the observed enclosure here.
258
+ const projectTemplateReasoning = (content: string): TaggedReasoningProjection => {
259
+ for (const [opening, closing] of [
260
+ ["<|channel>thought\n", "<channel|>"],
261
+ ["<think>\n", "</think>"],
262
+ ] as const) {
263
+ if (content.startsWith(opening)) return projectLeadingReasoning(content, "", opening, closing);
264
+ }
265
+ return { content, reasoning: "", projected: false, contentStart: 0 };
266
+ };
206
267
 
207
268
  // Shared budget→effort breakpoints (xai and google had identical copies).
208
269
  export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
@@ -211,6 +272,13 @@ export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
211
272
  return "high";
212
273
  };
213
274
 
275
+ // AI SDK's portable reasoning control has no boolean-enabled value. `medium`
276
+ // is the neutral activation projection for an explicit, unqualified `on`; it
277
+ // changes no PLURNK output budget. An operator reasoning subset, when present, remains
278
+ // the only input to the existing magnitude-to-tier projection.
279
+ const effortFromReasoning = (reasoning: Reasoning): "low" | "medium" | "high" =>
280
+ reasoning.budget === null ? "medium" : effortFromBudget(reasoning.budget);
281
+
214
282
  // Body keys the provider owns — a caller's `sampling` passthrough may not set
215
283
  // these. Two families:
216
284
  // transport/managed — grammar transport, the stream/JSON choice, slot pinning,
@@ -219,7 +287,7 @@ export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
219
287
  // response; n>1 = paid, dropped output), the tool-calling family (tools-in-
220
288
  // body doctrine, §2: native tool_calls return null content = a broken turn),
221
289
  // modalities/audio (text-only contract), prediction (decode semantics, not
222
- // sampling), and the token caps (the envelope is the managed maxTokens —
290
+ // sampling), and the token caps (the envelope is the managed maxOutputTokens —
223
291
  // sampling must not bypass the consumer's cap).
224
292
  // Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
225
293
  // platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
@@ -238,6 +306,8 @@ export default class AiSdkProvider implements Provider {
238
306
  #url: string | undefined;
239
307
  #languageModel: LanguageModel | undefined;
240
308
  #fetchTimeoutMs: number;
309
+ #operationTimeoutMs: number;
310
+ #firstContentTimeoutMs: number;
241
311
  #streamIdleTimeoutMs: number | undefined;
242
312
  #headers: Record<string, string>;
243
313
  #fetch: ProviderFetch;
@@ -245,6 +315,11 @@ export default class AiSdkProvider implements Provider {
245
315
  #apiKeyRejectedMessage: string | undefined;
246
316
  #eosText: string | undefined;
247
317
  #contextWindow: number | null;
318
+ #maxInputTokens: number | null;
319
+ #maxOutputTokens: number | null;
320
+ #outputBudget: number | null;
321
+ #reasoningBudget: number | null;
322
+ #additiveReasoningProvider: "anthropic" | "bedrock" | undefined;
248
323
  #reasoning: Reasoning;
249
324
  #temperature: number;
250
325
  #repeatPenalty: number;
@@ -257,12 +332,13 @@ export default class AiSdkProvider implements Provider {
257
332
  #reasoningResponseStyle: ReasoningResponseStyle;
258
333
  #countPromptTokens: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
259
334
  #promptTokensUrl: string | undefined;
260
- #calculateCost: (usage: ProviderUsage) => number;
261
- #calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }>;
262
- #normalizeCharge?: AuthoritativeChargeNormalizer;
335
+ #estimateCost: (usage: ProviderUsage | undefined) => ProviderCost;
336
+ #normalizeCost?: ProviderCostNormalizer;
263
337
  #source: string;
264
338
  #grammarStyle: GrammarStyle;
265
- #promptCacheKey: boolean;
339
+ #cacheAffinity: CacheAffinity | undefined;
340
+ #systemCacheProviderOptions: AiSdkProviderOptions | undefined;
341
+ #reasoningResponseProviderOptions: AiSdkProviderOptions | undefined;
266
342
  #serviceTier: string | undefined;
267
343
  #gbnfDebug: boolean;
268
344
  #streaming: boolean;
@@ -272,12 +348,10 @@ export default class AiSdkProvider implements Provider {
272
348
  #retryAttempts: number;
273
349
  #errorDetailLimit: number | undefined;
274
350
  #topLogprobs: number | null;
275
- #reasoningReserve: ReserveSpec | undefined;
276
- #completionReserve: ReserveSpec | undefined;
277
351
  #tuningFloors: boolean;
278
352
  #rawBody: boolean;
279
353
  #servedModel: string | undefined;
280
- #requiresMaxTokens: boolean | undefined;
354
+ #requiresOutputBudget: boolean | undefined;
281
355
  readonly attributions?: (context: PluginAttributionContext) => PluginAttribution;
282
356
 
283
357
  // Optional capability ({§provider-local-capabilities}): exact tokenization served by the backend's
@@ -293,11 +367,28 @@ export default class AiSdkProvider implements Provider {
293
367
  if ((this.#url === undefined) === (this.#languageModel === undefined)) {
294
368
  throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
295
369
  }
370
+ for (const [name, value] of [
371
+ ["fetchTimeoutMs", config.fetchTimeoutMs],
372
+ ["operationTimeoutMs", config.operationTimeoutMs],
373
+ ["firstContentTimeoutMs", config.firstContentTimeoutMs],
374
+ ["streamIdleTimeoutMs", config.streamIdleTimeoutMs],
375
+ ] as const) {
376
+ if (value !== undefined && (!Number.isInteger(value) || value < 0 || value > MAX_PROVIDER_TIMEOUT_MS)) {
377
+ throw new Error(`${config.source ?? "provider"}: ${name} must be an integer from 0 through ${MAX_PROVIDER_TIMEOUT_MS} milliseconds`);
378
+ }
379
+ }
296
380
  this.#fetchTimeoutMs = config.fetchTimeoutMs;
381
+ this.#operationTimeoutMs = config.operationTimeoutMs;
382
+ this.#firstContentTimeoutMs = config.firstContentTimeoutMs;
297
383
  this.#streamIdleTimeoutMs = config.streamIdleTimeoutMs;
298
384
  this.#headers = config.headers ?? {};
299
385
  this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
300
386
  this.#contextWindow = config.contextWindow ?? null;
387
+ this.#maxInputTokens = config.maxInputTokens ?? null;
388
+ this.#maxOutputTokens = config.maxOutputTokens ?? null;
389
+ this.#outputBudget = config.outputBudget ?? null;
390
+ this.#reasoningBudget = config.reasoningBudget ?? null;
391
+ this.#additiveReasoningProvider = config.additiveReasoningProvider;
301
392
  this.#reasoning = config.reasoning;
302
393
  // Loud guard: an out-of-date consumer (stale plugin dist) omitting the
303
394
  // required tuning fields must fail at construction, not silently send
@@ -321,12 +412,36 @@ export default class AiSdkProvider implements Provider {
321
412
  }
322
413
  this.#countPromptTokens = config.countPromptTokens ?? ((messages) => estimatePromptTokens(messages));
323
414
  this.#promptTokensUrl = config.promptTokensUrl;
324
- this.#calculateCost = config.calculateCost ?? (() => 0);
325
- this.#calculateCharge = config.calculateCharge;
326
- this.#normalizeCharge = config.normalizeCharge;
415
+ this.#estimateCost = config.estimateCost
416
+ ?? (() => ({
417
+ kind: "unknown",
418
+ reason: "the request reported no direct cost and no model rate is configured",
419
+ }));
420
+ this.#normalizeCost = config.normalizeCost;
327
421
  this.#source = config.source ?? "provider";
328
422
  this.#grammarStyle = config.grammarStyle ?? "none";
329
- this.#promptCacheKey = config.promptCacheKey ?? false;
423
+ this.#cacheAffinity = config.cacheAffinity;
424
+ this.#systemCacheProviderOptions = config.systemCacheProviderOptions;
425
+ this.#reasoningResponseProviderOptions = config.reasoningResponseProviderOptions;
426
+ if (this.#languageModel === undefined && this.#cacheAffinity?.target === "provider-option") {
427
+ throw new Error(`${this.#source}: native provider-option cache affinity requires an AI SDK model`);
428
+ }
429
+ if (this.#languageModel !== undefined && this.#cacheAffinity?.target === "body") {
430
+ throw new Error(`${this.#source}: body cache affinity requires an OpenAI-compatible URL`);
431
+ }
432
+ if (this.#languageModel === undefined && this.#systemCacheProviderOptions !== undefined) {
433
+ throw new Error(`${this.#source}: system cache provider options require an AI SDK model`);
434
+ }
435
+ if (this.#languageModel === undefined && this.#reasoningResponseProviderOptions !== undefined) {
436
+ throw new Error(`${this.#source}: reasoning response provider options require an AI SDK model`);
437
+ }
438
+ if (this.#cacheAffinity?.target === "provider-option"
439
+ && Object.hasOwn(
440
+ this.#reasoningResponseProviderOptions?.[this.#cacheAffinity.provider] ?? {},
441
+ this.#cacheAffinity.name,
442
+ )) {
443
+ throw new Error(`${this.#source}: reasoning response options conflict with cache-affinity option ${this.#cacheAffinity.provider}.${this.#cacheAffinity.name}`);
444
+ }
330
445
  this.#serviceTier = config.serviceTier;
331
446
  this.#gbnfDebug = config.gbnfDebug ?? false;
332
447
  this.#streaming = config.streaming ?? true;
@@ -337,18 +452,42 @@ export default class AiSdkProvider implements Provider {
337
452
  this.#supportsSlotPinning = config.supportsSlotPinning ?? false;
338
453
  this.#slotCount = config.slotCount ?? null;
339
454
  this.#topLogprobs = config.topLogprobs ?? null;
340
- this.#reasoningReserve = config.reasoningReserve;
341
- this.#completionReserve = config.completionReserve;
342
455
  this.#tuningFloors = config.tuningFloors ?? true;
343
456
  this.#rawBody = config.rawBody ?? false;
344
457
  this.#servedModel = config.servedModel;
345
- this.#requiresMaxTokens = config.requiresMaxTokens;
346
- const reasoningReserve = this.reasoningReserve;
347
- if (this.#reasoningStyle === "template"
458
+ this.#requiresOutputBudget = config.requiresOutputBudget;
459
+ for (const [name, value] of [
460
+ ["contextWindow", this.#contextWindow],
461
+ ["maxInputTokens", this.#maxInputTokens],
462
+ ["maxOutputTokens", this.#maxOutputTokens],
463
+ ["outputBudget", this.#outputBudget],
464
+ ["reasoningBudget", this.#reasoningBudget],
465
+ ] as const) {
466
+ if (value !== null && (!Number.isSafeInteger(value) || value <= 0)) {
467
+ throw new Error(`${this.#source}: ${name} must be a positive safe integer or null`);
468
+ }
469
+ }
470
+ if (this.#reasoningBudget !== null
471
+ && this.#outputBudget !== null
472
+ && this.#reasoningBudget >= this.#outputBudget) {
473
+ throw new Error(`${this.#source}: reasoningBudget must be smaller than the total outputBudget`);
474
+ }
475
+ if (this.#reasoning.budget !== this.#reasoningBudget) {
476
+ throw new Error(`${this.#source}: reasoning intent and generation envelope disagree on reasoningBudget`);
477
+ }
478
+ if (this.#reasoningStyle === "anthropic"
479
+ && this.#reasoning.mode === "on"
480
+ && this.#reasoning.budget === null
481
+ && this.#reasoningBudget === null) {
482
+ throw new Error(`${this.#source}: explicit Anthropic reasoning requires PLURNK_PROVIDERS_REASONING_BUDGET`);
483
+ }
484
+ if (this.#additiveReasoningProvider !== undefined
348
485
  && this.#reasoning.mode === "on"
349
- && reasoningReserve !== null
350
- && this.#reasoning.budget! > reasoningReserve) {
351
- throw new Error(`${this.#source}: PLURNK_PROVIDERS_REASONING_BUDGET (${this.#reasoning.budget}) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE (${reasoningReserve})`);
486
+ && this.#reasoningBudget === null) {
487
+ throw new Error(`${this.#source}: explicit ${this.#additiveReasoningProvider} reasoning requires PLURNK_PROVIDERS_REASONING_BUDGET so the total output budget remains bounded`);
488
+ }
489
+ if (this.#requiresOutputBudget === true && this.#outputBudget === null) {
490
+ throw new Error(`${this.#source}: this backend requires a resolved PLURNK_PROVIDERS_OUTPUT_BUDGET`);
352
491
  }
353
492
  const { tokenizeUrl } = config;
354
493
  if (tokenizeUrl !== undefined) {
@@ -357,7 +496,9 @@ export default class AiSdkProvider implements Provider {
357
496
  method: "POST",
358
497
  headers: { "Content-Type": "application/json", ...this.#headers },
359
498
  body: JSON.stringify({ content: text }),
360
- signal: AbortSignal.timeout(this.#fetchTimeoutMs),
499
+ ...(this.#fetchTimeoutMs > 0
500
+ ? { signal: AbortSignal.timeout(this.#fetchTimeoutMs) }
501
+ : {}),
361
502
  });
362
503
  if (!res.ok) throw new Error(`${this.#source}: tokenize endpoint returned ${res.status}`);
363
504
  const { tokens } = (await res.json()) as { tokens?: unknown };
@@ -370,20 +511,22 @@ export default class AiSdkProvider implements Provider {
370
511
  }
371
512
 
372
513
  get contextWindow(): number | null { return this.#contextWindow; }
373
- // {§provider-generation-envelope} Envelope reserves — absolute pins stand alone; percentages need the
374
- // detected window; null = underivable (no claim for core's no-cap path).
375
- #resolveReserve(spec: ReserveSpec | undefined): number | null {
376
- if (spec === undefined) return null;
377
- if ("tokens" in spec) return spec.tokens;
378
- return this.#contextWindow === null ? null : Math.round(spec.percent * this.#contextWindow);
514
+ get maxInputTokens(): number | null { return this.#maxInputTokens; }
515
+ get maxOutputTokens(): number | null { return this.#maxOutputTokens; }
516
+ get outputBudget(): number | null { return this.#outputBudget; }
517
+ get reasoningBudget(): number | null { return this.#reasoningBudget; }
518
+ get inputCapacity(): number | null {
519
+ return effectiveInputCapacity({
520
+ contextWindow: this.#contextWindow,
521
+ maxInputTokens: this.#maxInputTokens,
522
+ outputBudget: this.#outputBudget,
523
+ });
379
524
  }
380
- get reasoningReserve(): number | null { return this.#resolveReserve(this.#reasoningReserve); }
381
- get completionReserve(): number | null { return this.#resolveReserve(this.#completionReserve); }
382
525
  get model(): string { return this.#model; }
383
526
  // Backend's self-reported served id; undefined when unprobed/unknown.
384
527
  get servedModel(): string | undefined { return this.#servedModel; }
385
528
  // Resolved "decodes unbounded without a cap" fact; undefined = no claim.
386
- get requiresMaxTokens(): boolean | undefined { return this.#requiresMaxTokens; }
529
+ get requiresOutputBudget(): boolean | undefined { return this.#requiresOutputBudget; }
387
530
  // Resolved capability: will a transported grammar actually constrain
388
531
  // this backend's decode? Introspectable so a consumer can verify the rails
389
532
  // are LIVE without spending a generation on a forcing-grammar probe.
@@ -402,7 +545,14 @@ export default class AiSdkProvider implements Provider {
402
545
 
403
546
  signal?.throwIfAborted();
404
547
  try {
405
- const timeout = AbortSignal.timeout(this.#fetchTimeoutMs);
548
+ const timeout = this.#fetchTimeoutMs > 0
549
+ ? AbortSignal.timeout(this.#fetchTimeoutMs)
550
+ : undefined;
551
+ const requestSignal = signal === undefined
552
+ ? timeout
553
+ : timeout === undefined
554
+ ? signal
555
+ : AbortSignal.any([signal, timeout]);
406
556
  const response = await this.#fetch(this.#promptTokensUrl, {
407
557
  method: "POST",
408
558
  headers: { "Content-Type": "application/json", ...this.#headers },
@@ -411,7 +561,7 @@ export default class AiSdkProvider implements Provider {
411
561
  messages,
412
562
  ...this.#reasoningBody(),
413
563
  }),
414
- signal: signal === undefined ? timeout : AbortSignal.any([signal, timeout]),
564
+ ...(requestSignal === undefined ? {} : { signal: requestSignal }),
415
565
  });
416
566
  if (!response.ok) {
417
567
  return estimatePromptTokens(
@@ -439,23 +589,46 @@ export default class AiSdkProvider implements Provider {
439
589
  );
440
590
  }
441
591
  }
442
- calculateCost(usage: ProviderUsage): number { return this.#calculateCost(usage); }
443
- calculateCharge(usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }> {
444
- return this.#calculateCharge?.(usage)
445
- ?? { kind: "unknown", reason: "no provider rate or settled charge is available" };
446
- }
447
592
 
593
+ async assessRequestCapacity(
594
+ messages: readonly ChatMessage[],
595
+ maxOutputTokens?: number,
596
+ signal?: AbortSignal,
597
+ ): Promise<ProviderRequestCapacity> {
598
+ const outputBudget = effectiveOutputBudget({
599
+ requested: maxOutputTokens,
600
+ configured: this.#outputBudget,
601
+ maxOutputTokens: this.#maxOutputTokens,
602
+ contextWindow: this.#contextWindow,
603
+ });
604
+ const reasoningBudget = effectiveReasoningBudget({
605
+ configured: this.#reasoningBudget,
606
+ outputBudget,
607
+ });
608
+ return assessRequestCapacity({
609
+ contextWindow: this.#contextWindow,
610
+ maxInputTokens: this.#maxInputTokens,
611
+ maxOutputTokens: this.#maxOutputTokens,
612
+ outputBudget,
613
+ reasoningBudget,
614
+ measurement: await this.countPromptTokens(messages, signal),
615
+ });
616
+ }
448
617
  // Reasoning activation and allowance are independent of grammar transport;
449
618
  // only the response representation becomes lossless when evidence is needed.
450
619
  // The llama-server template mapping is owned by {§llama-reasoning-request}.
451
- #reasoningBody(preserveGrammarSentence = false): Record<string, unknown> {
452
- const { mode, budget } = this.#reasoning;
620
+ #reasoningBody(
621
+ preserveGrammarSentence = false,
622
+ reasoningBudget = this.#reasoningBudget,
623
+ ): Record<string, unknown> {
624
+ const { mode } = this.#reasoning;
625
+ const budget = reasoningBudget;
453
626
  const on = mode !== "off";
454
627
  switch (this.#reasoningStyle) {
455
628
  case "template": {
456
629
  const allowance = mode === "off"
457
630
  ? 0
458
- : mode === "on" ? budget : this.reasoningReserve;
631
+ : mode === "on" && budget !== null ? budget : this.#reasoningBudget;
459
632
  return {
460
633
  chat_template_kwargs: { enable_thinking: on },
461
634
  reasoning_format: preserveGrammarSentence ? "none" : "auto",
@@ -464,9 +637,9 @@ export default class AiSdkProvider implements Provider {
464
637
  }
465
638
  case "think": return on ? { think: true } : {};
466
639
  case "include_reasoning": return on ? { include_reasoning: true } : {};
467
- // effort tiers from the budget; off/adaptive omit the field (the
468
- // API's default depth is its adaptive).
469
- case "effort": return mode === "on" ? { reasoning_effort: effortFromBudget(budget!) } : {};
640
+ // Explicit on uses the portable enabled posture or a tier derived
641
+ // from an explicit budget; off/adaptive omit the field.
642
+ case "effort": return mode === "on" ? { reasoning_effort: effortFromReasoning({ mode, budget }) } : {};
470
643
  // Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
471
644
  // reason-by-default model (DeepSeek V4: default 'high') reasoning.
472
645
  // ADAPTIVE omits the field: the backend's own default posture IS the
@@ -476,19 +649,25 @@ export default class AiSdkProvider implements Provider {
476
649
  // efforts 400.
477
650
  case "effort_explicit": return mode === "off"
478
651
  ? { reasoning_effort: "none" }
479
- : mode === "on" ? { reasoning_effort: effortFromBudget(budget!) } : {};
652
+ : mode === "on" ? { reasoning_effort: effortFromReasoning({ mode, budget }) } : {};
480
653
  // {§deepseek-reasoning-request}
481
654
  case "thinking_effort": return mode === "off"
482
655
  ? { thinking: { type: "disabled" } }
483
656
  : mode === "on" ? {
484
657
  thinking: { type: "enabled" },
485
- reasoning_effort: effortFromBudget(budget!),
658
+ ...(budget === null ? {} : { reasoning_effort: effortFromBudget(budget) }),
486
659
  } : {};
487
660
  // Anthropic compat: explicit thinking object. off → disabled; on →
488
- // enabled with budget_tokens; adaptive → omit (the API default).
661
+ // enabled with the explicit reasoning subset; adaptive →
662
+ // omit (the API default).
489
663
  case "anthropic": return mode === "off"
490
664
  ? { thinking: { type: "disabled" } }
491
- : mode === "on" ? { thinking: { type: "enabled", budget_tokens: budget } } : {};
665
+ : mode === "on" ? {
666
+ thinking: {
667
+ type: "enabled",
668
+ budget_tokens: budget!,
669
+ },
670
+ } : {};
492
671
  case "none": return {};
493
672
  }
494
673
  }
@@ -555,7 +734,7 @@ export default class AiSdkProvider implements Provider {
555
734
  }
556
735
  }
557
736
 
558
- // First-party telemetry headers ({§provider-request-authority}): forwarded only when the spec
737
+ // First-party telemetry headers ({§provider-request-authority} {§provider-call-kind}): forwarded only when the spec
559
738
  // opted in (the plurnk endpoint). The gate is here, not at the call site, so
560
739
  // attributions/client/strikes can never reach a third-party backend even if
561
740
  // the consumer passes them to the wrong provider. Empty values emit no header
@@ -563,7 +742,7 @@ export default class AiSdkProvider implements Provider {
563
742
  // absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
564
743
  // ride HTTP headers only — the packet never carries them (the model must
565
744
  // never see strike state; engine accounting is not a metric to game).
566
- #metadataHeaders(attributions: string[] | undefined, client: string | undefined, strikes: number | undefined, workerId: string, primaryWorkerId: string | undefined, workspaceId: string | undefined, loop: number | undefined, turn: number | undefined): Record<string, string> {
745
+ #metadataHeaders(attributions: string[] | undefined, client: string | undefined, strikes: number | undefined, workerId: string, primaryWorkerId: string | undefined, workspaceId: string | undefined, loop: number | undefined, turn: number | undefined, callKind: ProviderCallKind | undefined): Record<string, string> {
567
746
  if (!this.#firstPartyMetadata) return {};
568
747
  const h: Record<string, string> = {};
569
748
  if (attributions !== undefined && attributions.length > 0) h["Plurnk-Attribution"] = JSON.stringify(attributions);
@@ -588,6 +767,7 @@ export default class AiSdkProvider implements Provider {
588
767
  if (workspaceId !== undefined && workspaceId.length > 0) h["Plurnk-Workspace-Id"] = workspaceId;
589
768
  if (loop !== undefined && Number.isInteger(loop) && loop >= 1) h["Plurnk-Loop"] = String(loop);
590
769
  if (turn !== undefined && Number.isInteger(turn) && turn >= 1) h["Plurnk-Turn"] = String(turn);
770
+ if (callKind !== undefined) h["Plurnk-Call-Kind"] = callKind;
591
771
  return h;
592
772
  }
593
773
 
@@ -627,9 +807,69 @@ export default class AiSdkProvider implements Provider {
627
807
  return out;
628
808
  }
629
809
 
630
- async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }: { messages: ChatMessage[]; workerId: string; primaryWorkerId?: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse> {
810
+ #requestProviderOptions(
811
+ workerId: string,
812
+ reasoningBudget: number | null,
813
+ ): AiSdkProviderOptions | undefined {
814
+ const responseOptions = this.#reasoning.mode === "off"
815
+ ? undefined
816
+ : this.#reasoningResponseProviderOptions;
817
+ const nativeReasoning = this.#reasoning.mode === "on" && reasoningBudget !== null
818
+ ? this.#additiveReasoningProvider === "anthropic"
819
+ ? { anthropic: { thinking: { type: "enabled", budgetTokens: reasoningBudget } } }
820
+ : this.#additiveReasoningProvider === "bedrock"
821
+ ? { bedrock: { reasoningConfig: { type: "enabled", budgetTokens: reasoningBudget } } }
822
+ : undefined
823
+ : undefined;
824
+ const options: AiSdkProviderOptions = {};
825
+ for (const part of [responseOptions, nativeReasoning]) {
826
+ for (const [provider, values] of Object.entries(part ?? {})) {
827
+ options[provider] = { ...options[provider], ...values };
828
+ }
829
+ }
830
+ if (this.#cacheAffinity?.target === "provider-option") {
831
+ const { provider, name } = this.#cacheAffinity;
832
+ options[provider] = { ...options[provider], [name]: workerId };
833
+ }
834
+ return Object.keys(options).length === 0 ? undefined : options;
835
+ }
836
+
837
+ #nativeMaxOutputTokens(
838
+ outputBudget: number | null,
839
+ reasoningBudget: number | null,
840
+ ): number | undefined {
841
+ if (outputBudget === null) return undefined;
842
+ return this.#additiveReasoningProvider !== undefined
843
+ && this.#reasoning.mode === "on"
844
+ && reasoningBudget !== null
845
+ ? outputBudget - reasoningBudget
846
+ : outputBudget;
847
+ }
848
+
849
+ #accounting(
850
+ outcome: ProviderRequestAccounting["outcome"],
851
+ usage: ProviderUsage | undefined,
852
+ evidence: Parameters<ProviderCostNormalizer>[0],
853
+ status?: number,
854
+ ): ProviderRequestAccounting {
855
+ const knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
856
+ const direct = this.#normalizeCost?.(evidence);
857
+ return validateProviderRequestAccounting({
858
+ provider: this.#source,
859
+ model: this.#model,
860
+ outcome,
861
+ ...(status === undefined ? {} : { status }),
862
+ ...(knownUsage === undefined ? {} : { usage: knownUsage }),
863
+ cost: resolveProviderCost(direct, this.#estimateCost(knownUsage)),
864
+ });
865
+ }
866
+
867
+ async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxOutputTokens, attributions, client, strikes, workspaceId, loop, turn, sampling, observeRequest, callKind }: ProviderGenerateArgs): Promise<ProviderResponse> {
631
868
  // {§provider-interface} The worker identity is required.
632
869
  if (workerId === undefined || workerId.length === 0) throw new Error("generate: workerId is required — the worker's stable, opaque identity");
870
+ if (callKind !== undefined && callKind !== "emission" && callKind !== "bare") {
871
+ throw new Error(`generate: unsupported callKind ${JSON.stringify(callKind)}`);
872
+ }
633
873
  // Reject before any wire call when already aborted
634
874
  // ({§provider-failure-normalization}).
635
875
  signal?.throwIfAborted();
@@ -642,6 +882,20 @@ export default class AiSdkProvider implements Provider {
642
882
  const preserveGrammarSentence = wantGrammar
643
883
  && this.#reasoningStyle === "template";
644
884
 
885
+ const capacity = await this.assessRequestCapacity(messages, maxOutputTokens, signal);
886
+ if (capacity.decision === "reject") {
887
+ if (capacity.prompt.kind !== "exact") {
888
+ throw new TypeError(`${this.#source}: only an exact prompt measurement may reject capacity`);
889
+ }
890
+ throw new ProviderError(
891
+ this.#source,
892
+ "capacity_exceeded",
893
+ `The exact provider request uses ${capacity.prompt.tokens} input tokens, exceeding its ${capacity.inputCapacity} token input capacity.`,
894
+ { capacity, extensions: { capacityStage: "preflight", capacity } },
895
+ );
896
+ }
897
+ const effectiveMaxOutputTokens = capacity.outputBudget ?? undefined;
898
+
645
899
  // Assembly order = precedence: the family's sampling DEFAULTS
646
900
  // (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
647
901
  // paths and the name promises every request) < the caller's `sampling`
@@ -654,79 +908,195 @@ export default class AiSdkProvider implements Provider {
654
908
  ...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
655
909
  model: this.#model,
656
910
  messages,
657
- ...this.#reasoningBody(preserveGrammarSentence),
911
+ ...this.#reasoningBody(preserveGrammarSentence, capacity.reasoningBudget),
658
912
  ...this.#grammarBody(sendGrammar),
659
- ...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
913
+ ...(effectiveMaxOutputTokens !== undefined ? { max_tokens: effectiveMaxOutputTokens } : {}),
660
914
  // Request per-token logprobs only when enabled (managed field —
661
915
  // reserved from caller sampling; the env flag is the single control).
662
916
  ...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
663
917
  ...this.#slotBody(workerId),
664
- // Prompt-cache affinity -- workerId as the OpenAI-standard
665
- // prompt_cache_key routes a worker's turns to one serverless replica so
666
- // its stable prefix caches (managed; reserved from caller sampling).
667
- ...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
918
+ ...(this.#cacheAffinity?.target === "body"
919
+ ? { [this.#cacheAffinity.name]: workerId }
920
+ : {}),
668
921
  };
669
922
 
670
923
  // Per-request headers = static auth/routing + any first-party telemetry.
671
- const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
672
- const headers = Object.keys(metaHeaders).length === 0
673
- ? this.#headers
674
- : { ...this.#headers, ...metaHeaders };
675
- let raw;
676
- try {
677
- raw = this.#languageModel === undefined
678
- ? await executeOpenAICompatible({
679
- url: this.#url!,
924
+ const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind);
925
+ const headers = new Headers(this.#headers);
926
+ if (this.#cacheAffinity?.target === "header") {
927
+ headers.set(this.#cacheAffinity.name, workerId);
928
+ }
929
+ for (const [name, value] of Object.entries(metaHeaders)) headers.set(name, value);
930
+ const requestHeaders = Object.fromEntries(headers.entries());
931
+ const accounting: ProviderRequestAccounting[] = [];
932
+ const operationTimeout = this.#operationTimeoutMs > 0
933
+ ? AbortSignal.timeout(this.#operationTimeoutMs)
934
+ : undefined;
935
+ const operationSignal = signal === undefined
936
+ ? operationTimeout
937
+ : operationTimeout === undefined
938
+ ? signal
939
+ : AbortSignal.any([signal, operationTimeout]);
940
+ const executeRequest = async () => {
941
+ let settle: ProviderRequestSettlement | undefined;
942
+ try {
943
+ settle = await observeRequest?.({
944
+ provider: this.#source,
680
945
  model: this.#model,
681
- headers,
682
- body,
683
- messages,
684
- signal,
685
- fetch: this.#fetch,
686
- fetchTimeoutMs: this.#fetchTimeoutMs,
687
- streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
688
- retryAttempts: this.#retryAttempts,
689
- streaming: this.#streaming,
690
- captureRawBody: this.#rawBody,
691
- })
692
- : await executeAiSdkModel({
693
- languageModel: this.#languageModel,
694
- headers,
695
- messages,
696
- signal,
697
- fetchTimeoutMs: this.#fetchTimeoutMs,
698
- streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
699
- retryAttempts: this.#retryAttempts,
700
- streaming: this.#streaming,
701
- captureRawBody: this.#rawBody,
702
- temperature: this.#tuningFloors
703
- ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
704
- : typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
705
- topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
706
- topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
707
- presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
708
- frequencyPenalty: typeof sampling?.frequency_penalty === "number"
709
- ? sampling.frequency_penalty
710
- : this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
711
- stopSequences: typeof sampling?.stop === "string"
712
- ? [sampling.stop]
713
- : Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
714
- ? sampling.stop
715
- : undefined,
716
- seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
717
- maxOutputTokens: maxTokens,
718
- reasoning: this.#reasoning.mode === "off"
719
- ? "none"
720
- : this.#reasoning.mode === "adaptive"
721
- ? "provider-default"
722
- : effortFromBudget(this.#reasoning.budget!),
723
946
  });
947
+ } catch (cause) {
948
+ throw new ProviderRequestObserverError(cause);
949
+ }
950
+ const settleAccounting = async (
951
+ outcome: ProviderRequestAccounting["outcome"],
952
+ usage: ProviderUsage | undefined,
953
+ evidence: Parameters<ProviderCostNormalizer>[0],
954
+ status?: number,
955
+ ): Promise<ProviderRequestAccounting> => {
956
+ let requestAccounting: ProviderRequestAccounting;
957
+ let normalizationFailure: { cause: unknown } | undefined;
958
+ try {
959
+ requestAccounting = this.#accounting(outcome, usage, evidence, status);
960
+ } catch (cause) {
961
+ normalizationFailure = { cause };
962
+ let knownUsage: ProviderUsage | undefined;
963
+ try {
964
+ knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
965
+ } catch {
966
+ knownUsage = undefined;
967
+ }
968
+ requestAccounting = validateProviderRequestAccounting({
969
+ provider: this.#source,
970
+ model: this.#model,
971
+ outcome,
972
+ ...(status === undefined ? {} : { status }),
973
+ ...(knownUsage === undefined ? {} : { usage: knownUsage }),
974
+ cost: {
975
+ kind: "unknown",
976
+ reason: "provider request accounting could not be normalized after physical I/O",
977
+ },
978
+ });
979
+ }
980
+ accounting.push(requestAccounting);
981
+ try {
982
+ await settle?.(requestAccounting);
983
+ } catch (cause) {
984
+ throw new ProviderRequestObserverError(cause);
985
+ }
986
+ if (normalizationFailure !== undefined) {
987
+ throw new ProviderRequestAccountingError(normalizationFailure.cause);
988
+ }
989
+ return requestAccounting;
990
+ };
991
+ let response;
992
+ try {
993
+ response = this.#languageModel === undefined
994
+ ? await executeOpenAICompatible({
995
+ url: this.#url!,
996
+ model: this.#model,
997
+ headers: requestHeaders,
998
+ body,
999
+ messages,
1000
+ signal: operationSignal,
1001
+ fetch: this.#fetch,
1002
+ fetchTimeoutMs: this.#fetchTimeoutMs,
1003
+ firstContentTimeoutMs: this.#firstContentTimeoutMs,
1004
+ streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
1005
+ streaming: this.#streaming,
1006
+ captureRawBody: this.#rawBody,
1007
+ })
1008
+ : await executeAiSdkModel({
1009
+ languageModel: this.#languageModel,
1010
+ headers: requestHeaders,
1011
+ providerOptions: this.#requestProviderOptions(workerId, capacity.reasoningBudget),
1012
+ systemProviderOptions: this.#systemCacheProviderOptions,
1013
+ messages,
1014
+ signal: operationSignal,
1015
+ fetchTimeoutMs: this.#fetchTimeoutMs,
1016
+ firstContentTimeoutMs: this.#firstContentTimeoutMs,
1017
+ streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
1018
+ streaming: this.#streaming,
1019
+ captureRawBody: this.#rawBody,
1020
+ temperature: this.#tuningFloors
1021
+ ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
1022
+ : typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
1023
+ topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
1024
+ topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
1025
+ presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
1026
+ frequencyPenalty: typeof sampling?.frequency_penalty === "number"
1027
+ ? sampling.frequency_penalty
1028
+ : this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
1029
+ stopSequences: typeof sampling?.stop === "string"
1030
+ ? [sampling.stop]
1031
+ : Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
1032
+ ? sampling.stop
1033
+ : undefined,
1034
+ seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
1035
+ maxOutputTokens: this.#nativeMaxOutputTokens(capacity.outputBudget, capacity.reasoningBudget),
1036
+ reasoning: this.#reasoning.mode === "off"
1037
+ ? "none"
1038
+ : this.#reasoning.mode === "adaptive"
1039
+ ? "provider-default"
1040
+ : this.#additiveReasoningProvider !== undefined && capacity.reasoningBudget !== null
1041
+ ? "provider-default"
1042
+ : effortFromReasoning({
1043
+ mode: this.#reasoning.mode,
1044
+ budget: capacity.reasoningBudget,
1045
+ }),
1046
+ });
1047
+ } catch (error) {
1048
+ const failure = transportFailureEvidence(error);
1049
+ await settleAccounting(
1050
+ "error",
1051
+ failure.usage,
1052
+ failure.chargeEvidence,
1053
+ failure.status,
1054
+ );
1055
+ throw error;
1056
+ }
1057
+ await settleAccounting(
1058
+ "response",
1059
+ response.usage,
1060
+ response.chargeEvidence,
1061
+ );
1062
+ return response;
1063
+ };
1064
+
1065
+ let raw;
1066
+ try {
1067
+ const { retry } = prepareRetries({
1068
+ maxRetries: this.#retryAttempts,
1069
+ abortSignal: operationSignal,
1070
+ });
1071
+ raw = await retry(executeRequest);
724
1072
  } catch (err) {
1073
+ if (err instanceof ProviderRequestObserverError
1074
+ || err instanceof ProviderRequestAccountingError) throw err.cause;
725
1075
  if (signal?.aborted) throw err;
726
- const pe = toProviderError(err, this.#source, this.#errorDetailLimit);
1076
+ if (operationTimeout?.aborted) {
1077
+ const timeout = new ProviderTimeoutError("operation", this.#operationTimeoutMs, err);
1078
+ throw new ProviderError(this.#source, "deadline_exceeded", timeout.message, {
1079
+ status: 504,
1080
+ cause: timeout,
1081
+ retryable: false,
1082
+ extensions: {
1083
+ timeoutPhase: timeout.phase,
1084
+ timeoutMs: timeout.timeoutMs,
1085
+ },
1086
+ accounting,
1087
+ capacity,
1088
+ });
1089
+ }
1090
+ const pe = toProviderError(err, this.#source, this.#errorDetailLimit, capacity);
727
1091
  if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
728
- throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
1092
+ throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, {
1093
+ status: pe.status,
1094
+ cause: err,
1095
+ accounting,
1096
+ capacity,
1097
+ });
729
1098
  }
1099
+ pe.prependAccounting(accounting);
730
1100
  throw pe;
731
1101
  }
732
1102
 
@@ -747,10 +1117,10 @@ export default class AiSdkProvider implements Provider {
747
1117
  this.#reasoningResponseStyle,
748
1118
  );
749
1119
 
750
- // Preserve the exact sentence seen at the grammar boundary. Constrained
751
- // template turns request `reasoning_format: "none"`, so even an empty
752
- // channel remains observable. An unexpectedly projected response cannot
753
- // supply independent pre-projection evidence.
1120
+ // Preserve the exact pre-projection response. Constrained template turns
1121
+ // request `reasoning_format: "none"`, so even an empty channel and any
1122
+ // template-provided opener remain observable. An unexpectedly projected
1123
+ // response cannot supply independent evidence.
754
1124
  let grammarEvidence: GrammarEvidence | undefined;
755
1125
  if (wantGrammar) {
756
1126
  if (preserveGrammarSentence) {
@@ -779,12 +1149,13 @@ export default class AiSdkProvider implements Provider {
779
1149
  if (projectedReasoning.projected) {
780
1150
  raw.content = projectedReasoning.content;
781
1151
  raw.reasoning = projectedReasoning.reasoning;
782
- raw.usage = attributeUnitemizedReasoning(raw.usage, raw.reasoning, raw.content);
783
1152
  }
784
1153
 
785
1154
  let notices: ProviderNotice[] | undefined;
786
1155
  const usage = raw.usage;
787
- if (sendGrammar !== undefined && this.tokenize !== undefined) {
1156
+ if (sendGrammar !== undefined
1157
+ && this.tokenize !== undefined
1158
+ && usage?.outputTokens !== undefined) {
788
1159
  // Channel-escape detector: completion tokens
789
1160
  // billed far beyond every visible channel mean the decode ESCAPED into
790
1161
  // a server-discarded reasoning block mid-emission. This diagnostic
@@ -795,12 +1166,12 @@ export default class AiSdkProvider implements Provider {
795
1166
  this.tokenize(raw.reasoning),
796
1167
  ]);
797
1168
  const visible = contentTokens.length + reasoningTokens.length;
798
- if (usage.completion > visible + 64) {
1169
+ if (usage.outputTokens > visible + 64) {
799
1170
  (notices ??= []).push({
800
1171
  source: this.#source,
801
1172
  kind: "grammar_unenforced",
802
1173
  level: "warn",
803
- message: `decode escaped the grammar: ${usage.completion} completion tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
1174
+ message: `decode escaped the grammar: ${usage.outputTokens} output tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
804
1175
  position: [...raw.content].length,
805
1176
  });
806
1177
  }
@@ -824,22 +1195,40 @@ export default class AiSdkProvider implements Provider {
824
1195
  ...(raw.reasoningEncrypted.length > 0
825
1196
  ? { reasoningEncrypted: raw.reasoningEncrypted }
826
1197
  : {}),
827
- usage,
828
1198
  model: raw.model,
829
1199
  ...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
830
1200
  };
831
- const normalizedCharge = this.#normalizeCharge?.(raw.chargeEvidence);
832
- const charge = normalizedCharge === undefined
833
- ? undefined
834
- : validateAuthoritativeCharge(normalizedCharge);
835
1201
  const evidence = {
836
1202
  assistantRaw: raw,
837
- ...(charge === undefined ? {} : { charge }),
1203
+ accounting,
1204
+ capacity,
838
1205
  ...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
839
1206
  ...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
840
1207
  ...(meta !== undefined ? { meta } : {}),
841
1208
  ...(notices !== undefined ? { notices } : {}),
842
1209
  };
1210
+ if (capacity.outputBudget !== null
1211
+ && usage?.outputTokens !== undefined
1212
+ && usage.outputTokens > capacity.outputBudget) {
1213
+ const attempt: ProviderAttempt = {
1214
+ assistant: { ...assistant, finishReason: raw.finishReason },
1215
+ ...evidence,
1216
+ };
1217
+ throw new ProviderError(
1218
+ this.#source,
1219
+ "invalid_response",
1220
+ `The provider reported ${usage.outputTokens} output tokens after receiving a total output budget of ${capacity.outputBudget}.`,
1221
+ {
1222
+ attempt,
1223
+ accounting,
1224
+ extensions: {
1225
+ stage: "provider-response",
1226
+ outputBudget: capacity.outputBudget,
1227
+ reportedOutputTokens: usage.outputTokens,
1228
+ },
1229
+ },
1230
+ );
1231
+ }
843
1232
  if (raw.finishReason === "resource_interrupted") {
844
1233
  const attempt: ProviderResponse<"resource_interrupted"> = {
845
1234
  assistant: { ...assistant, finishReason: raw.finishReason },
@@ -851,6 +1240,7 @@ export default class AiSdkProvider implements Provider {
851
1240
  "The provider interrupted generation because inference resources were unavailable.",
852
1241
  {
853
1242
  attempt,
1243
+ accounting,
854
1244
  extensions: {
855
1245
  stage: "provider-response",
856
1246
  finishReason: "resource_interrupted",