@plurnk/plurnk-providers 1.4.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. package/.env.defaults +40 -34
  2. package/README.md +3 -0
  3. package/SPEC.md +153 -62
  4. package/dist/AiSdkProvider.d.ts +19 -25
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +353 -120
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +7 -13
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +36 -8
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +2 -21
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +19 -14
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/accounting.d.ts +6 -0
  17. package/dist/accounting.d.ts.map +1 -0
  18. package/dist/accounting.js +168 -0
  19. package/dist/accounting.js.map +1 -0
  20. package/dist/aiSdkTransport.d.ts +11 -3
  21. package/dist/aiSdkTransport.d.ts.map +1 -1
  22. package/dist/aiSdkTransport.js +198 -29
  23. package/dist/aiSdkTransport.js.map +1 -1
  24. package/dist/catalogProvider.d.ts +7 -2
  25. package/dist/catalogProvider.d.ts.map +1 -1
  26. package/dist/catalogProvider.js +32 -26
  27. package/dist/catalogProvider.js.map +1 -1
  28. package/dist/compatibleProvider.d.ts.map +1 -1
  29. package/dist/compatibleProvider.js +18 -7
  30. package/dist/compatibleProvider.js.map +1 -1
  31. package/dist/cost.d.ts +10 -10
  32. package/dist/cost.d.ts.map +1 -1
  33. package/dist/cost.js +88 -43
  34. package/dist/cost.js.map +1 -1
  35. package/dist/env.d.ts +5 -7
  36. package/dist/env.d.ts.map +1 -1
  37. package/dist/env.js +30 -32
  38. package/dist/env.js.map +1 -1
  39. package/dist/errors.d.ts +14 -2
  40. package/dist/errors.d.ts.map +1 -1
  41. package/dist/errors.js +60 -2
  42. package/dist/errors.js.map +1 -1
  43. package/dist/index.d.ts +4 -4
  44. package/dist/index.d.ts.map +1 -1
  45. package/dist/index.js +3 -2
  46. package/dist/index.js.map +1 -1
  47. package/dist/ollama.js +3 -3
  48. package/dist/ollama.js.map +1 -1
  49. package/dist/sdkModels.d.ts +6 -0
  50. package/dist/sdkModels.d.ts.map +1 -1
  51. package/dist/sdkModels.js +46 -3
  52. package/dist/sdkModels.js.map +1 -1
  53. package/dist/types.d.ts +40 -29
  54. package/dist/types.d.ts.map +1 -1
  55. package/dist/usage.d.ts +21 -4
  56. package/dist/usage.d.ts.map +1 -1
  57. package/dist/usage.js +188 -74
  58. package/dist/usage.js.map +1 -1
  59. package/package.json +9 -7
  60. package/src/AiSdkProvider.test.ts +1039 -182
  61. package/src/AiSdkProvider.ts +428 -141
  62. package/src/Mock.test.ts +37 -12
  63. package/src/Mock.ts +46 -12
  64. package/src/Pool.test.ts +19 -6
  65. package/src/Pool.ts +20 -16
  66. package/src/ProviderRegistry.test.ts +16 -11
  67. package/src/accounting.test.ts +94 -0
  68. package/src/accounting.ts +190 -0
  69. package/src/aiSdkTransport.test.ts +42 -49
  70. package/src/aiSdkTransport.ts +218 -32
  71. package/src/boundaries.test.ts +2 -0
  72. package/src/catalogProvider.test.ts +271 -24
  73. package/src/catalogProvider.ts +44 -28
  74. package/src/compatibleProvider.test.ts +6 -3
  75. package/src/compatibleProvider.ts +20 -7
  76. package/src/cost.test.ts +55 -35
  77. package/src/cost.ts +110 -54
  78. package/src/defaults.test.ts +13 -3
  79. package/src/env.test.ts +50 -26
  80. package/src/env.ts +43 -42
  81. package/src/errors.test.ts +47 -2
  82. package/src/errors.ts +68 -3
  83. package/src/index.ts +21 -5
  84. package/src/ollama.test.ts +4 -1
  85. package/src/ollama.ts +3 -3
  86. package/src/sdkModels.test.ts +94 -3
  87. package/src/sdkModels.ts +53 -3
  88. package/src/types.ts +91 -33
  89. package/src/usage.test.ts +112 -108
  90. package/src/usage.ts +233 -84
@@ -6,18 +6,40 @@
6
6
  // ordinary vendor protocol. The compatible URL path remains only for PLURNK
7
7
  // extensions and local endpoint probes the SDK cannot represent.
8
8
 
9
- import type { ChatMessage, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
9
+ import type {
10
+ ChatMessage,
11
+ GrammarEvidence,
12
+ PromptTokenMeasurement,
13
+ Provider,
14
+ ProviderCostNormalizer,
15
+ ProviderCallKind,
16
+ ProviderGenerateArgs,
17
+ ProviderRequestAccounting,
18
+ ProviderRequestObserver,
19
+ ProviderRequestSettlement,
20
+ ProviderResponse,
21
+ ProviderUsage,
22
+ } from "./types.ts";
10
23
  import type { ProviderCost } from "@plurnk/plurnk-contracts";
24
+ import type { JSONValue } from "ai";
25
+ import { MAX_PROVIDER_TIMEOUT_MS } from "./env.ts";
11
26
  import type { Reasoning, ReasoningResponseStyle, ReserveSpec } from "./env.ts";
12
- import { executeAiSdkModel, executeOpenAICompatible } from "./aiSdkTransport.ts";
27
+ import {
28
+ executeAiSdkModel,
29
+ executeOpenAICompatible,
30
+ transportFailureEvidence,
31
+ } from "./aiSdkTransport.ts";
13
32
  import type { LanguageModel } from "ai";
14
- import { toProviderError, ProviderError } from "./errors.ts";
15
- import { attributeUnitemizedReasoning } from "./usage.ts";
33
+ import { prepareRetries } from "ai/internal";
34
+ import { toProviderError, ProviderError, ProviderTimeoutError } from "./errors.ts";
16
35
  import type { ProviderNotice } from "./notices.ts";
17
36
  import { validateGbnf } from "@plurnk/gbnf";
18
37
  import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.ts";
19
38
  import { emitWarningOnce } from "./warnings.ts";
20
39
  import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
40
+ import { resolveProviderCost } from "./cost.ts";
41
+ import { validateProviderRequestAccounting } from "./accounting.ts";
42
+ import { validateProviderUsage } from "./usage.ts";
21
43
 
22
44
  export type ProviderFetch = typeof globalThis.fetch;
23
45
 
@@ -29,29 +51,40 @@ export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" |
29
51
  // service-managed constrained sampling; endpoint-owned settings are not inferred.
30
52
  export type GrammarStyle = "none" | "llamacpp";
31
53
 
54
+ export type CacheAffinity =
55
+ | { readonly target: "header" | "body"; readonly name: string }
56
+ | { readonly target: "provider-option"; readonly provider: string; readonly name: string };
57
+
58
+ export type AiSdkProviderOptions = Record<string, Record<string, JSONValue | undefined>>;
59
+
32
60
  export type AiSdkProviderConfig = {
33
61
  model: string;
34
62
  url?: string; // OpenAI-compatible chat-completions URL
35
63
  languageModel?: LanguageModel; // native AI SDK provider model
36
64
  attributions?: (context: PluginAttributionContext) => PluginAttribution;
37
- fetchTimeoutMs: number;
38
- streamIdleTimeoutMs?: number; // streamed body inter-chunk deadline; zero/unset disables
65
+ fetchTimeoutMs: number; // one physical generation attempt; zero disables
66
+ operationTimeoutMs: number; // complete logical call across retries/backoff; zero disables
67
+ firstContentTimeoutMs: number; // first semantic streamed content; zero disables
68
+ streamIdleTimeoutMs?: number; // semantic streamed-content idle deadline; zero/unset disables
39
69
  headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
40
70
  fetch?: ProviderFetch; // per-instance request executor; default globalThis.fetch
41
71
  contextWindow?: number | null; // default null; caller resolves-or-fails, narrows to required with the interface
42
72
  reasoningStyle?: ReasoningStyle; // default "none"
43
73
  reasoningResponseStyle?: ReasoningResponseStyle; // {§provider-tagged-reasoning}; default "verbatim"
44
74
  countPromptTokens?: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
45
- calculateCost?: (usage: ProviderUsage) => number; // default () => 0
46
- calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }>;
75
+ estimateCost?: (usage: ProviderUsage | undefined) => ProviderCost;
76
+ normalizeCost?: ProviderCostNormalizer;
47
77
  source?: string; // notice/problem source, e.g. "provider:openai"; default "provider"
48
78
  grammarStyle?: GrammarStyle; // how a GBNF grammar is carried; default "none" (not sent)
49
- // Send the OpenAI-standard `prompt_cache_key` set to workerId, so a
50
- // serverless backend with REPLICA-LOCAL prompt caching (fireworks, verified)
51
- // pins a worker's turns to one replica and claims its stable prefix. Default
52
- // false -- a backend that strict-validates unknown fields 400s, so enable only
53
- // where the field is accepted. Same identity that already drives slot affinity.
54
- promptCacheKey?: boolean;
79
+ // {§provider-cache-affinity} Provider routes own the exact documented
80
+ // projection; the common transport only applies it as managed request state.
81
+ cacheAffinity?: CacheAffinity;
82
+ // {§provider-cache-write-policy} Already policy-gated by provider construction.
83
+ // The transport attaches it to only the final leading system instruction.
84
+ systemCacheProviderOptions?: AiSdkProviderOptions;
85
+ // {§provider-readable-reasoning} Route-owned native option needed to expose
86
+ // readable reasoning. Applied only when the effective posture is not off.
87
+ reasoningResponseProviderOptions?: AiSdkProviderOptions;
55
88
  // Optional provider-configured service tier. Unlike caller sampling, this is
56
89
  // a fixed deployment choice and therefore wins on every request.
57
90
  serviceTier?: string;
@@ -81,9 +114,9 @@ export type AiSdkProviderConfig = {
81
114
  requiresMaxTokens?: boolean;
82
115
  // The side-channel reasoning intent — REQUIRED, no in-code default
83
116
  // (PLURNK_PROVIDERS_REASONING + _BUDGET, read via reasoningFromEnv):
84
- // { mode: off|adaptive|on, budget: iff on }. The provider maps it to the
85
- // backend's mechanism via reasoningStyle; budget is only ever a magnitude,
86
- // never a hidden activation flag.
117
+ // { mode: off|adaptive|on, budget: optional when on }. The provider maps it
118
+ // to the backend's mechanism via reasoningStyle; budget is only ever an
119
+ // explicit magnitude, never a hidden activation flag.
87
120
  reasoning: Reasoning;
88
121
  // Decode tuning: no in-code defaults; the canonical measured values (0.2 /
89
122
  // 1.15) ship as the floor in .env.defaults (alias-scopable). `temperature` is the
@@ -138,6 +171,20 @@ export type AiSdkProviderConfig = {
138
171
  tuningFloors?: boolean;
139
172
  };
140
173
 
174
+ class ProviderRequestObserverError extends Error {
175
+ constructor(cause: unknown) {
176
+ super("provider request accounting could not be durably settled", { cause });
177
+ this.name = "ProviderRequestObserverError";
178
+ }
179
+ }
180
+
181
+ class ProviderRequestAccountingError extends Error {
182
+ constructor(cause: unknown) {
183
+ super("provider request accounting could not be normalized", { cause });
184
+ this.name = "ProviderRequestAccountingError";
185
+ }
186
+ }
187
+
141
188
  // Drop trailing occurrences of a server-rendered EOG marker. llama-server
142
189
  // under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
143
190
  // trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
@@ -157,19 +204,15 @@ type TaggedReasoningProjection = {
157
204
  readonly contentStart: number;
158
205
  };
159
206
 
160
- // {§provider-tagged-reasoning} Only the model-contract position is structural:
161
- // one exact leading envelope. Parsing after stream assembly keeps SSE and JSON
162
- // on one path and leaves later literal tags in the visible suffix untouched.
163
- const projectTaggedReasoning = (
207
+ const projectLeadingReasoning = (
164
208
  content: string,
165
209
  structuredReasoning: string,
166
- style: ReasoningResponseStyle,
210
+ opening: string,
211
+ closing: string,
167
212
  ): TaggedReasoningProjection => {
168
- const opening = "<think>";
169
- if (style !== "think-tags" || structuredReasoning.length > 0 || !content.startsWith(opening)) {
213
+ if (structuredReasoning.length > 0 || !content.startsWith(opening)) {
170
214
  return { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
171
215
  }
172
- const closing = "</think>";
173
216
  const closingIndex = content.indexOf(closing, opening.length);
174
217
  if (closingIndex === -1) {
175
218
  return {
@@ -188,6 +231,31 @@ const projectTaggedReasoning = (
188
231
  };
189
232
  };
190
233
 
234
+ // {§provider-tagged-reasoning} Only the model-contract position is structural:
235
+ // one exact leading envelope. Parsing after stream assembly keeps SSE and JSON
236
+ // on one path and leaves later literal tags in the visible suffix untouched.
237
+ const projectTaggedReasoning = (
238
+ content: string,
239
+ structuredReasoning: string,
240
+ style: ReasoningResponseStyle,
241
+ ): TaggedReasoningProjection => style === "think-tags"
242
+ ? projectLeadingReasoning(content, structuredReasoning, "<think>", "</think>")
243
+ : { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
244
+
245
+ // llama-server's template reasoning parser can project either supported leading
246
+ // reasoning envelope out of the OpenAI-compatible response. Grammar evidence
247
+ // needs the sentence before that lossy projection, so constrained template turns
248
+ // request it verbatim and split the observed enclosure here.
249
+ const projectTemplateReasoning = (content: string): TaggedReasoningProjection => {
250
+ for (const [opening, closing] of [
251
+ ["<|channel>thought\n", "<channel|>"],
252
+ ["<think>\n", "</think>"],
253
+ ] as const) {
254
+ if (content.startsWith(opening)) return projectLeadingReasoning(content, "", opening, closing);
255
+ }
256
+ return { content, reasoning: "", projected: false, contentStart: 0 };
257
+ };
258
+
191
259
  // Shared budget→effort breakpoints (xai and google had identical copies).
192
260
  export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
193
261
  if (budget <= 1000) return "low";
@@ -195,6 +263,13 @@ export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
195
263
  return "high";
196
264
  };
197
265
 
266
+ // AI SDK's portable reasoning control has no boolean-enabled value. `medium`
267
+ // is the neutral activation projection for an explicit, unqualified `on`; it
268
+ // changes no PLURNK token reserve. An operator budget, when present, remains
269
+ // the only input to the existing magnitude-to-tier projection.
270
+ const effortFromReasoning = (reasoning: Reasoning): "low" | "medium" | "high" =>
271
+ reasoning.budget === null ? "medium" : effortFromBudget(reasoning.budget);
272
+
198
273
  // Body keys the provider owns — a caller's `sampling` passthrough may not set
199
274
  // these. Two families:
200
275
  // transport/managed — grammar transport, the stream/JSON choice, slot pinning,
@@ -222,6 +297,8 @@ export default class AiSdkProvider implements Provider {
222
297
  #url: string | undefined;
223
298
  #languageModel: LanguageModel | undefined;
224
299
  #fetchTimeoutMs: number;
300
+ #operationTimeoutMs: number;
301
+ #firstContentTimeoutMs: number;
225
302
  #streamIdleTimeoutMs: number | undefined;
226
303
  #headers: Record<string, string>;
227
304
  #fetch: ProviderFetch;
@@ -241,11 +318,13 @@ export default class AiSdkProvider implements Provider {
241
318
  #reasoningResponseStyle: ReasoningResponseStyle;
242
319
  #countPromptTokens: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
243
320
  #promptTokensUrl: string | undefined;
244
- #calculateCost: (usage: ProviderUsage) => number;
245
- #calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }>;
321
+ #estimateCost: (usage: ProviderUsage | undefined) => ProviderCost;
322
+ #normalizeCost?: ProviderCostNormalizer;
246
323
  #source: string;
247
324
  #grammarStyle: GrammarStyle;
248
- #promptCacheKey: boolean;
325
+ #cacheAffinity: CacheAffinity | undefined;
326
+ #systemCacheProviderOptions: AiSdkProviderOptions | undefined;
327
+ #reasoningResponseProviderOptions: AiSdkProviderOptions | undefined;
249
328
  #serviceTier: string | undefined;
250
329
  #gbnfDebug: boolean;
251
330
  #streaming: boolean;
@@ -268,7 +347,6 @@ export default class AiSdkProvider implements Provider {
268
347
  // tokenizeUrl (llama-server), so `provider.tokenize === undefined` remains
269
348
  // the honest capability signal for every other backend.
270
349
  tokenize?: (text: string) => Promise<number[]>;
271
-
272
350
  constructor(config: AiSdkProviderConfig) {
273
351
  this.#model = config.model;
274
352
  this.#url = config.url;
@@ -277,7 +355,19 @@ export default class AiSdkProvider implements Provider {
277
355
  if ((this.#url === undefined) === (this.#languageModel === undefined)) {
278
356
  throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
279
357
  }
358
+ for (const [name, value] of [
359
+ ["fetchTimeoutMs", config.fetchTimeoutMs],
360
+ ["operationTimeoutMs", config.operationTimeoutMs],
361
+ ["firstContentTimeoutMs", config.firstContentTimeoutMs],
362
+ ["streamIdleTimeoutMs", config.streamIdleTimeoutMs],
363
+ ] as const) {
364
+ if (value !== undefined && (!Number.isInteger(value) || value < 0 || value > MAX_PROVIDER_TIMEOUT_MS)) {
365
+ throw new Error(`${config.source ?? "provider"}: ${name} must be an integer from 0 through ${MAX_PROVIDER_TIMEOUT_MS} milliseconds`);
366
+ }
367
+ }
280
368
  this.#fetchTimeoutMs = config.fetchTimeoutMs;
369
+ this.#operationTimeoutMs = config.operationTimeoutMs;
370
+ this.#firstContentTimeoutMs = config.firstContentTimeoutMs;
281
371
  this.#streamIdleTimeoutMs = config.streamIdleTimeoutMs;
282
372
  this.#headers = config.headers ?? {};
283
373
  this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
@@ -305,11 +395,36 @@ export default class AiSdkProvider implements Provider {
305
395
  }
306
396
  this.#countPromptTokens = config.countPromptTokens ?? ((messages) => estimatePromptTokens(messages));
307
397
  this.#promptTokensUrl = config.promptTokensUrl;
308
- this.#calculateCost = config.calculateCost ?? (() => 0);
309
- this.#calculateCharge = config.calculateCharge;
398
+ this.#estimateCost = config.estimateCost
399
+ ?? (() => ({
400
+ kind: "unknown",
401
+ reason: "the request reported no direct cost and no model rate is configured",
402
+ }));
403
+ this.#normalizeCost = config.normalizeCost;
310
404
  this.#source = config.source ?? "provider";
311
405
  this.#grammarStyle = config.grammarStyle ?? "none";
312
- this.#promptCacheKey = config.promptCacheKey ?? false;
406
+ this.#cacheAffinity = config.cacheAffinity;
407
+ this.#systemCacheProviderOptions = config.systemCacheProviderOptions;
408
+ this.#reasoningResponseProviderOptions = config.reasoningResponseProviderOptions;
409
+ if (this.#languageModel === undefined && this.#cacheAffinity?.target === "provider-option") {
410
+ throw new Error(`${this.#source}: native provider-option cache affinity requires an AI SDK model`);
411
+ }
412
+ if (this.#languageModel !== undefined && this.#cacheAffinity?.target === "body") {
413
+ throw new Error(`${this.#source}: body cache affinity requires an OpenAI-compatible URL`);
414
+ }
415
+ if (this.#languageModel === undefined && this.#systemCacheProviderOptions !== undefined) {
416
+ throw new Error(`${this.#source}: system cache provider options require an AI SDK model`);
417
+ }
418
+ if (this.#languageModel === undefined && this.#reasoningResponseProviderOptions !== undefined) {
419
+ throw new Error(`${this.#source}: reasoning response provider options require an AI SDK model`);
420
+ }
421
+ if (this.#cacheAffinity?.target === "provider-option"
422
+ && Object.hasOwn(
423
+ this.#reasoningResponseProviderOptions?.[this.#cacheAffinity.provider] ?? {},
424
+ this.#cacheAffinity.name,
425
+ )) {
426
+ throw new Error(`${this.#source}: reasoning response options conflict with cache-affinity option ${this.#cacheAffinity.provider}.${this.#cacheAffinity.name}`);
427
+ }
313
428
  this.#serviceTier = config.serviceTier;
314
429
  this.#gbnfDebug = config.gbnfDebug ?? false;
315
430
  this.#streaming = config.streaming ?? true;
@@ -329,10 +444,17 @@ export default class AiSdkProvider implements Provider {
329
444
  const reasoningReserve = this.reasoningReserve;
330
445
  if (this.#reasoningStyle === "template"
331
446
  && this.#reasoning.mode === "on"
447
+ && this.#reasoning.budget !== null
332
448
  && reasoningReserve !== null
333
- && this.#reasoning.budget! > reasoningReserve) {
449
+ && this.#reasoning.budget > reasoningReserve) {
334
450
  throw new Error(`${this.#source}: PLURNK_PROVIDERS_REASONING_BUDGET (${this.#reasoning.budget}) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE (${reasoningReserve})`);
335
451
  }
452
+ if (this.#reasoningStyle === "anthropic"
453
+ && this.#reasoning.mode === "on"
454
+ && this.#reasoning.budget === null
455
+ && reasoningReserve === null) {
456
+ throw new Error(`${this.#source}: explicit Anthropic reasoning requires a resolved reasoning reserve or PLURNK_PROVIDERS_REASONING_BUDGET`);
457
+ }
336
458
  const { tokenizeUrl } = config;
337
459
  if (tokenizeUrl !== undefined) {
338
460
  this.tokenize = async (text: string): Promise<number[]> => {
@@ -340,7 +462,9 @@ export default class AiSdkProvider implements Provider {
340
462
  method: "POST",
341
463
  headers: { "Content-Type": "application/json", ...this.#headers },
342
464
  body: JSON.stringify({ content: text }),
343
- signal: AbortSignal.timeout(this.#fetchTimeoutMs),
465
+ ...(this.#fetchTimeoutMs > 0
466
+ ? { signal: AbortSignal.timeout(this.#fetchTimeoutMs) }
467
+ : {}),
344
468
  });
345
469
  if (!res.ok) throw new Error(`${this.#source}: tokenize endpoint returned ${res.status}`);
346
470
  const { tokens } = (await res.json()) as { tokens?: unknown };
@@ -385,7 +509,14 @@ export default class AiSdkProvider implements Provider {
385
509
 
386
510
  signal?.throwIfAborted();
387
511
  try {
388
- const timeout = AbortSignal.timeout(this.#fetchTimeoutMs);
512
+ const timeout = this.#fetchTimeoutMs > 0
513
+ ? AbortSignal.timeout(this.#fetchTimeoutMs)
514
+ : undefined;
515
+ const requestSignal = signal === undefined
516
+ ? timeout
517
+ : timeout === undefined
518
+ ? signal
519
+ : AbortSignal.any([signal, timeout]);
389
520
  const response = await this.#fetch(this.#promptTokensUrl, {
390
521
  method: "POST",
391
522
  headers: { "Content-Type": "application/json", ...this.#headers },
@@ -394,7 +525,7 @@ export default class AiSdkProvider implements Provider {
394
525
  messages,
395
526
  ...this.#reasoningBody(),
396
527
  }),
397
- signal: signal === undefined ? timeout : AbortSignal.any([signal, timeout]),
528
+ ...(requestSignal === undefined ? {} : { signal: requestSignal }),
398
529
  });
399
530
  if (!response.ok) {
400
531
  return estimatePromptTokens(
@@ -422,33 +553,28 @@ export default class AiSdkProvider implements Provider {
422
553
  );
423
554
  }
424
555
  }
425
- calculateCost(usage: ProviderUsage): number { return this.#calculateCost(usage); }
426
- calculateCharge(usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }> {
427
- return this.#calculateCharge?.(usage)
428
- ?? { kind: "unknown", reason: "no provider rate or settled charge is available" };
429
- }
430
-
431
- // Reasoning intent maps independently of grammar transport. The llama-server
432
- // template mapping is owned by {§llama-reasoning-request}.
433
- #reasoningBody(): Record<string, unknown> {
556
+ // Reasoning activation and allowance are independent of grammar transport;
557
+ // only the response representation becomes lossless when evidence is needed.
558
+ // The llama-server template mapping is owned by {§llama-reasoning-request}.
559
+ #reasoningBody(preserveGrammarSentence = false): Record<string, unknown> {
434
560
  const { mode, budget } = this.#reasoning;
435
561
  const on = mode !== "off";
436
562
  switch (this.#reasoningStyle) {
437
563
  case "template": {
438
564
  const allowance = mode === "off"
439
565
  ? 0
440
- : mode === "on" ? budget : this.reasoningReserve;
566
+ : mode === "on" && budget !== null ? budget : this.reasoningReserve;
441
567
  return {
442
568
  chat_template_kwargs: { enable_thinking: on },
443
- reasoning_format: "auto",
569
+ reasoning_format: preserveGrammarSentence ? "none" : "auto",
444
570
  ...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
445
571
  };
446
572
  }
447
573
  case "think": return on ? { think: true } : {};
448
574
  case "include_reasoning": return on ? { include_reasoning: true } : {};
449
- // effort tiers from the budget; off/adaptive omit the field (the
450
- // API's default depth is its adaptive).
451
- case "effort": return mode === "on" ? { reasoning_effort: effortFromBudget(budget!) } : {};
575
+ // Explicit on uses the portable enabled posture or a tier derived
576
+ // from an explicit budget; off/adaptive omit the field.
577
+ case "effort": return mode === "on" ? { reasoning_effort: effortFromReasoning(this.#reasoning) } : {};
452
578
  // Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
453
579
  // reason-by-default model (DeepSeek V4: default 'high') reasoning.
454
580
  // ADAPTIVE omits the field: the backend's own default posture IS the
@@ -458,19 +584,25 @@ export default class AiSdkProvider implements Provider {
458
584
  // efforts 400.
459
585
  case "effort_explicit": return mode === "off"
460
586
  ? { reasoning_effort: "none" }
461
- : mode === "on" ? { reasoning_effort: effortFromBudget(budget!) } : {};
587
+ : mode === "on" ? { reasoning_effort: effortFromReasoning(this.#reasoning) } : {};
462
588
  // {§deepseek-reasoning-request}
463
589
  case "thinking_effort": return mode === "off"
464
590
  ? { thinking: { type: "disabled" } }
465
591
  : mode === "on" ? {
466
592
  thinking: { type: "enabled" },
467
- reasoning_effort: effortFromBudget(budget!),
593
+ ...(budget === null ? {} : { reasoning_effort: effortFromBudget(budget) }),
468
594
  } : {};
469
595
  // Anthropic compat: explicit thinking object. off → disabled; on →
470
- // enabled with budget_tokens; adaptive → omit (the API default).
596
+ // enabled with the explicit budget or resolved reserve; adaptive →
597
+ // omit (the API default).
471
598
  case "anthropic": return mode === "off"
472
599
  ? { thinking: { type: "disabled" } }
473
- : mode === "on" ? { thinking: { type: "enabled", budget_tokens: budget } } : {};
600
+ : mode === "on" ? {
601
+ thinking: {
602
+ type: "enabled",
603
+ budget_tokens: budget ?? this.reasoningReserve!,
604
+ },
605
+ } : {};
474
606
  case "none": return {};
475
607
  }
476
608
  }
@@ -537,7 +669,7 @@ export default class AiSdkProvider implements Provider {
537
669
  }
538
670
  }
539
671
 
540
- // First-party telemetry headers ({§provider-request-authority}): forwarded only when the spec
672
+ // First-party telemetry headers ({§provider-request-authority} {§provider-call-kind}): forwarded only when the spec
541
673
  // opted in (the plurnk endpoint). The gate is here, not at the call site, so
542
674
  // attributions/client/strikes can never reach a third-party backend even if
543
675
  // the consumer passes them to the wrong provider. Empty values emit no header
@@ -545,7 +677,7 @@ export default class AiSdkProvider implements Provider {
545
677
  // absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
546
678
  // ride HTTP headers only — the packet never carries them (the model must
547
679
  // never see strike state; engine accounting is not a metric to game).
548
- #metadataHeaders(attributions: string[] | undefined, client: string | undefined, strikes: number | undefined, workerId: string, primaryWorkerId: string | undefined, workspaceId: string | undefined, loop: number | undefined, turn: number | undefined): Record<string, string> {
680
+ #metadataHeaders(attributions: string[] | undefined, client: string | undefined, strikes: number | undefined, workerId: string, primaryWorkerId: string | undefined, workspaceId: string | undefined, loop: number | undefined, turn: number | undefined, callKind: ProviderCallKind | undefined): Record<string, string> {
549
681
  if (!this.#firstPartyMetadata) return {};
550
682
  const h: Record<string, string> = {};
551
683
  if (attributions !== undefined && attributions.length > 0) h["Plurnk-Attribution"] = JSON.stringify(attributions);
@@ -570,6 +702,7 @@ export default class AiSdkProvider implements Provider {
570
702
  if (workspaceId !== undefined && workspaceId.length > 0) h["Plurnk-Workspace-Id"] = workspaceId;
571
703
  if (loop !== undefined && Number.isInteger(loop) && loop >= 1) h["Plurnk-Loop"] = String(loop);
572
704
  if (turn !== undefined && Number.isInteger(turn) && turn >= 1) h["Plurnk-Turn"] = String(turn);
705
+ if (callKind !== undefined) h["Plurnk-Call-Kind"] = callKind;
573
706
  return h;
574
707
  }
575
708
 
@@ -609,9 +742,45 @@ export default class AiSdkProvider implements Provider {
609
742
  return out;
610
743
  }
611
744
 
612
- async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }: { messages: ChatMessage[]; workerId: string; primaryWorkerId?: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse> {
745
+ #requestProviderOptions(workerId: string): AiSdkProviderOptions | undefined {
746
+ const reasoningOptions = this.#reasoning.mode === "off"
747
+ ? undefined
748
+ : this.#reasoningResponseProviderOptions;
749
+ if (this.#cacheAffinity?.target !== "provider-option") return reasoningOptions;
750
+ const { provider, name } = this.#cacheAffinity;
751
+ return {
752
+ ...reasoningOptions,
753
+ [provider]: {
754
+ ...reasoningOptions?.[provider],
755
+ [name]: workerId,
756
+ },
757
+ };
758
+ }
759
+
760
+ #accounting(
761
+ outcome: ProviderRequestAccounting["outcome"],
762
+ usage: ProviderUsage | undefined,
763
+ evidence: Parameters<ProviderCostNormalizer>[0],
764
+ status?: number,
765
+ ): ProviderRequestAccounting {
766
+ const knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
767
+ const direct = this.#normalizeCost?.(evidence);
768
+ return validateProviderRequestAccounting({
769
+ provider: this.#source,
770
+ model: this.#model,
771
+ outcome,
772
+ ...(status === undefined ? {} : { status }),
773
+ ...(knownUsage === undefined ? {} : { usage: knownUsage }),
774
+ cost: resolveProviderCost(direct, this.#estimateCost(knownUsage)),
775
+ });
776
+ }
777
+
778
+ async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling, observeRequest, callKind }: ProviderGenerateArgs): Promise<ProviderResponse> {
613
779
  // {§provider-interface} The worker identity is required.
614
780
  if (workerId === undefined || workerId.length === 0) throw new Error("generate: workerId is required — the worker's stable, opaque identity");
781
+ if (callKind !== undefined && callKind !== "emission" && callKind !== "bare") {
782
+ throw new Error(`generate: unsupported callKind ${JSON.stringify(callKind)}`);
783
+ }
615
784
  // Reject before any wire call when already aborted
616
785
  // ({§provider-failure-normalization}).
617
786
  signal?.throwIfAborted();
@@ -621,6 +790,8 @@ export default class AiSdkProvider implements Provider {
621
790
  const wantGrammar = grammar !== undefined && this.#grammarStyle !== "none";
622
791
  if (wantGrammar && this.#gbnfDebug) this.#assertGrammarValid(grammar!);
623
792
  const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
793
+ const preserveGrammarSentence = wantGrammar
794
+ && this.#reasoningStyle === "template";
624
795
 
625
796
  // Assembly order = precedence: the family's sampling DEFAULTS
626
797
  // (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
@@ -634,77 +805,188 @@ export default class AiSdkProvider implements Provider {
634
805
  ...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
635
806
  model: this.#model,
636
807
  messages,
637
- ...this.#reasoningBody(),
808
+ ...this.#reasoningBody(preserveGrammarSentence),
638
809
  ...this.#grammarBody(sendGrammar),
639
810
  ...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
640
811
  // Request per-token logprobs only when enabled (managed field —
641
812
  // reserved from caller sampling; the env flag is the single control).
642
813
  ...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
643
814
  ...this.#slotBody(workerId),
644
- // Prompt-cache affinity -- workerId as the OpenAI-standard
645
- // prompt_cache_key routes a worker's turns to one serverless replica so
646
- // its stable prefix caches (managed; reserved from caller sampling).
647
- ...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
815
+ ...(this.#cacheAffinity?.target === "body"
816
+ ? { [this.#cacheAffinity.name]: workerId }
817
+ : {}),
648
818
  };
649
819
 
650
820
  // Per-request headers = static auth/routing + any first-party telemetry.
651
- const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
652
- const headers = Object.keys(metaHeaders).length > 0 ? { ...this.#headers, ...metaHeaders } : this.#headers;
653
- let raw;
654
- try {
655
- raw = this.#languageModel === undefined
656
- ? await executeOpenAICompatible({
657
- url: this.#url!,
821
+ const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind);
822
+ const headers = new Headers(this.#headers);
823
+ if (this.#cacheAffinity?.target === "header") {
824
+ headers.set(this.#cacheAffinity.name, workerId);
825
+ }
826
+ for (const [name, value] of Object.entries(metaHeaders)) headers.set(name, value);
827
+ const requestHeaders = Object.fromEntries(headers.entries());
828
+ const accounting: ProviderRequestAccounting[] = [];
829
+ const operationTimeout = this.#operationTimeoutMs > 0
830
+ ? AbortSignal.timeout(this.#operationTimeoutMs)
831
+ : undefined;
832
+ const operationSignal = signal === undefined
833
+ ? operationTimeout
834
+ : operationTimeout === undefined
835
+ ? signal
836
+ : AbortSignal.any([signal, operationTimeout]);
837
+ const executeRequest = async () => {
838
+ let settle: ProviderRequestSettlement | undefined;
839
+ try {
840
+ settle = await observeRequest?.({
841
+ provider: this.#source,
658
842
  model: this.#model,
659
- headers,
660
- body,
661
- messages,
662
- signal,
663
- fetch: this.#fetch,
664
- fetchTimeoutMs: this.#fetchTimeoutMs,
665
- streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
666
- retryAttempts: this.#retryAttempts,
667
- streaming: this.#streaming,
668
- captureRawBody: this.#rawBody,
669
- })
670
- : await executeAiSdkModel({
671
- languageModel: this.#languageModel,
672
- headers,
673
- messages,
674
- signal,
675
- fetchTimeoutMs: this.#fetchTimeoutMs,
676
- streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
677
- retryAttempts: this.#retryAttempts,
678
- streaming: this.#streaming,
679
- captureRawBody: this.#rawBody,
680
- temperature: this.#tuningFloors
681
- ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
682
- : typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
683
- topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
684
- topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
685
- presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
686
- frequencyPenalty: typeof sampling?.frequency_penalty === "number"
687
- ? sampling.frequency_penalty
688
- : this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
689
- stopSequences: typeof sampling?.stop === "string"
690
- ? [sampling.stop]
691
- : Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
692
- ? sampling.stop
693
- : undefined,
694
- seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
695
- maxOutputTokens: maxTokens,
696
- reasoning: this.#reasoning.mode === "off"
697
- ? "none"
698
- : this.#reasoning.mode === "adaptive"
699
- ? "provider-default"
700
- : effortFromBudget(this.#reasoning.budget!),
701
843
  });
844
+ } catch (cause) {
845
+ throw new ProviderRequestObserverError(cause);
846
+ }
847
+ const settleAccounting = async (
848
+ outcome: ProviderRequestAccounting["outcome"],
849
+ usage: ProviderUsage | undefined,
850
+ evidence: Parameters<ProviderCostNormalizer>[0],
851
+ status?: number,
852
+ ): Promise<ProviderRequestAccounting> => {
853
+ let requestAccounting: ProviderRequestAccounting;
854
+ let normalizationFailure: { cause: unknown } | undefined;
855
+ try {
856
+ requestAccounting = this.#accounting(outcome, usage, evidence, status);
857
+ } catch (cause) {
858
+ normalizationFailure = { cause };
859
+ let knownUsage: ProviderUsage | undefined;
860
+ try {
861
+ knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
862
+ } catch {
863
+ knownUsage = undefined;
864
+ }
865
+ requestAccounting = validateProviderRequestAccounting({
866
+ provider: this.#source,
867
+ model: this.#model,
868
+ outcome,
869
+ ...(status === undefined ? {} : { status }),
870
+ ...(knownUsage === undefined ? {} : { usage: knownUsage }),
871
+ cost: {
872
+ kind: "unknown",
873
+ reason: "provider request accounting could not be normalized after physical I/O",
874
+ },
875
+ });
876
+ }
877
+ accounting.push(requestAccounting);
878
+ try {
879
+ await settle?.(requestAccounting);
880
+ } catch (cause) {
881
+ throw new ProviderRequestObserverError(cause);
882
+ }
883
+ if (normalizationFailure !== undefined) {
884
+ throw new ProviderRequestAccountingError(normalizationFailure.cause);
885
+ }
886
+ return requestAccounting;
887
+ };
888
+ let response;
889
+ try {
890
+ response = this.#languageModel === undefined
891
+ ? await executeOpenAICompatible({
892
+ url: this.#url!,
893
+ model: this.#model,
894
+ headers: requestHeaders,
895
+ body,
896
+ messages,
897
+ signal: operationSignal,
898
+ fetch: this.#fetch,
899
+ fetchTimeoutMs: this.#fetchTimeoutMs,
900
+ firstContentTimeoutMs: this.#firstContentTimeoutMs,
901
+ streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
902
+ streaming: this.#streaming,
903
+ captureRawBody: this.#rawBody,
904
+ })
905
+ : await executeAiSdkModel({
906
+ languageModel: this.#languageModel,
907
+ headers: requestHeaders,
908
+ providerOptions: this.#requestProviderOptions(workerId),
909
+ systemProviderOptions: this.#systemCacheProviderOptions,
910
+ messages,
911
+ signal: operationSignal,
912
+ fetchTimeoutMs: this.#fetchTimeoutMs,
913
+ firstContentTimeoutMs: this.#firstContentTimeoutMs,
914
+ streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
915
+ streaming: this.#streaming,
916
+ captureRawBody: this.#rawBody,
917
+ temperature: this.#tuningFloors
918
+ ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
919
+ : typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
920
+ topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
921
+ topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
922
+ presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
923
+ frequencyPenalty: typeof sampling?.frequency_penalty === "number"
924
+ ? sampling.frequency_penalty
925
+ : this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
926
+ stopSequences: typeof sampling?.stop === "string"
927
+ ? [sampling.stop]
928
+ : Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
929
+ ? sampling.stop
930
+ : undefined,
931
+ seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
932
+ maxOutputTokens: maxTokens,
933
+ reasoning: this.#reasoning.mode === "off"
934
+ ? "none"
935
+ : this.#reasoning.mode === "adaptive"
936
+ ? "provider-default"
937
+ : effortFromReasoning(this.#reasoning),
938
+ });
939
+ } catch (error) {
940
+ const failure = transportFailureEvidence(error);
941
+ await settleAccounting(
942
+ "error",
943
+ failure.usage,
944
+ failure.chargeEvidence,
945
+ failure.status,
946
+ );
947
+ throw error;
948
+ }
949
+ await settleAccounting(
950
+ "response",
951
+ response.usage,
952
+ response.chargeEvidence,
953
+ );
954
+ return response;
955
+ };
956
+
957
+ let raw;
958
+ try {
959
+ const { retry } = prepareRetries({
960
+ maxRetries: this.#retryAttempts,
961
+ abortSignal: operationSignal,
962
+ });
963
+ raw = await retry(executeRequest);
702
964
  } catch (err) {
965
+ if (err instanceof ProviderRequestObserverError
966
+ || err instanceof ProviderRequestAccountingError) throw err.cause;
703
967
  if (signal?.aborted) throw err;
968
+ if (operationTimeout?.aborted) {
969
+ const timeout = new ProviderTimeoutError("operation", this.#operationTimeoutMs, err);
970
+ throw new ProviderError(this.#source, "deadline_exceeded", timeout.message, {
971
+ status: 504,
972
+ cause: timeout,
973
+ retryable: false,
974
+ extensions: {
975
+ timeoutPhase: timeout.phase,
976
+ timeoutMs: timeout.timeoutMs,
977
+ },
978
+ accounting,
979
+ });
980
+ }
704
981
  const pe = toProviderError(err, this.#source, this.#errorDetailLimit);
705
982
  if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
706
- throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
983
+ throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, {
984
+ status: pe.status,
985
+ cause: err,
986
+ accounting,
987
+ });
707
988
  }
989
+ pe.prependAccounting(accounting);
708
990
  throw pe;
709
991
  }
710
992
 
@@ -716,51 +998,54 @@ export default class AiSdkProvider implements Provider {
716
998
  // wire text for forensics.
717
999
  if (this.#eosText !== undefined) raw.content = stripTrailingSpecial(raw.content, this.#eosText);
718
1000
 
719
- const taggedReasoning = projectTaggedReasoning(
720
- raw.content,
721
- raw.reasoning,
722
- this.#reasoningResponseStyle,
723
- );
1001
+ const grammarInput = raw.content;
1002
+ const projectedReasoning = preserveGrammarSentence && !raw.reasoningProjected
1003
+ ? projectTemplateReasoning(raw.content)
1004
+ : projectTaggedReasoning(
1005
+ raw.content,
1006
+ raw.reasoning,
1007
+ this.#reasoningResponseStyle,
1008
+ );
724
1009
 
725
- // Preserve the exact sentence seen at the grammar boundary. llama-server's
726
- // `reasoning_format: "auto"` projects one raw Harmony enclosure into the
727
- // reasoning/content fields; the wire field's presence is the proof that the
728
- // projection occurred. The provider represents this evidence and never grades it.
1010
+ // Preserve the exact pre-projection response. Constrained template turns
1011
+ // request `reasoning_format: "none"`, so even an empty channel and any
1012
+ // template-provided opener remain observable. An unexpectedly projected
1013
+ // response cannot supply independent evidence.
729
1014
  let grammarEvidence: GrammarEvidence | undefined;
730
1015
  if (wantGrammar) {
731
- if (taggedReasoning.projected) {
732
- grammarEvidence = {
733
- input: raw.content,
734
- contentStart: taggedReasoning.contentStart,
735
- transported: sendGrammar !== undefined,
736
- };
737
- } else if (this.#reasoningStyle === "template" && this.#reasoning.mode !== "off") {
738
- if (raw.reasoningProjected) {
739
- const prefix = `<|channel>thought\n${raw.reasoning}<channel|>`;
1016
+ if (preserveGrammarSentence) {
1017
+ if (!raw.reasoningProjected) {
740
1018
  grammarEvidence = {
741
- input: `${prefix}${raw.content}`,
742
- contentStart: [...prefix].length,
1019
+ input: grammarInput,
1020
+ contentStart: projectedReasoning.projected ? projectedReasoning.contentStart : 0,
743
1021
  transported: sendGrammar !== undefined,
744
1022
  };
745
1023
  }
1024
+ } else if (projectedReasoning.projected) {
1025
+ grammarEvidence = {
1026
+ input: grammarInput,
1027
+ contentStart: projectedReasoning.contentStart,
1028
+ transported: sendGrammar !== undefined,
1029
+ };
746
1030
  } else {
747
1031
  grammarEvidence = {
748
- input: raw.content,
1032
+ input: grammarInput,
749
1033
  contentStart: 0,
750
1034
  transported: sendGrammar !== undefined,
751
1035
  };
752
1036
  }
753
1037
  }
754
1038
 
755
- if (taggedReasoning.projected) {
756
- raw.content = taggedReasoning.content;
757
- raw.reasoning = taggedReasoning.reasoning;
758
- raw.usage = attributeUnitemizedReasoning(raw.usage, raw.reasoning, raw.content);
1039
+ if (projectedReasoning.projected) {
1040
+ raw.content = projectedReasoning.content;
1041
+ raw.reasoning = projectedReasoning.reasoning;
759
1042
  }
760
1043
 
761
1044
  let notices: ProviderNotice[] | undefined;
762
1045
  const usage = raw.usage;
763
- if (sendGrammar !== undefined && this.tokenize !== undefined) {
1046
+ if (sendGrammar !== undefined
1047
+ && this.tokenize !== undefined
1048
+ && usage?.outputTokens !== undefined) {
764
1049
  // Channel-escape detector: completion tokens
765
1050
  // billed far beyond every visible channel mean the decode ESCAPED into
766
1051
  // a server-discarded reasoning block mid-emission. This diagnostic
@@ -771,12 +1056,12 @@ export default class AiSdkProvider implements Provider {
771
1056
  this.tokenize(raw.reasoning),
772
1057
  ]);
773
1058
  const visible = contentTokens.length + reasoningTokens.length;
774
- if (usage.completion > visible + 64) {
1059
+ if (usage.outputTokens > visible + 64) {
775
1060
  (notices ??= []).push({
776
1061
  source: this.#source,
777
1062
  kind: "grammar_unenforced",
778
1063
  level: "warn",
779
- message: `decode escaped the grammar: ${usage.completion} completion tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
1064
+ message: `decode escaped the grammar: ${usage.outputTokens} output tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
780
1065
  position: [...raw.content].length,
781
1066
  });
782
1067
  }
@@ -800,12 +1085,12 @@ export default class AiSdkProvider implements Provider {
800
1085
  ...(raw.reasoningEncrypted.length > 0
801
1086
  ? { reasoningEncrypted: raw.reasoningEncrypted }
802
1087
  : {}),
803
- usage,
804
1088
  model: raw.model,
805
1089
  ...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
806
1090
  };
807
1091
  const evidence = {
808
1092
  assistantRaw: raw,
1093
+ accounting,
809
1094
  ...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
810
1095
  ...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
811
1096
  ...(meta !== undefined ? { meta } : {}),
@@ -822,6 +1107,7 @@ export default class AiSdkProvider implements Provider {
822
1107
  "The provider interrupted generation because inference resources were unavailable.",
823
1108
  {
824
1109
  attempt,
1110
+ accounting,
825
1111
  extensions: {
826
1112
  stage: "provider-response",
827
1113
  finishReason: "resource_interrupted",
@@ -837,4 +1123,5 @@ export default class AiSdkProvider implements Provider {
837
1123
  ...evidence,
838
1124
  };
839
1125
  }
1126
+
840
1127
  }