@plurnk/plurnk-providers 1.5.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. package/.env.defaults +36 -22
  2. package/SPEC.md +133 -59
  3. package/dist/AiSdkProvider.d.ts +19 -26
  4. package/dist/AiSdkProvider.d.ts.map +1 -1
  5. package/dist/AiSdkProvider.js +318 -106
  6. package/dist/AiSdkProvider.js.map +1 -1
  7. package/dist/Mock.d.ts +4 -9
  8. package/dist/Mock.d.ts.map +1 -1
  9. package/dist/Mock.js +36 -9
  10. package/dist/Mock.js.map +1 -1
  11. package/dist/Pool.d.ts +2 -21
  12. package/dist/Pool.d.ts.map +1 -1
  13. package/dist/Pool.js +19 -14
  14. package/dist/Pool.js.map +1 -1
  15. package/dist/accounting.d.ts +5 -2
  16. package/dist/accounting.d.ts.map +1 -1
  17. package/dist/accounting.js +100 -16
  18. package/dist/accounting.js.map +1 -1
  19. package/dist/aiSdkTransport.d.ts +9 -2
  20. package/dist/aiSdkTransport.d.ts.map +1 -1
  21. package/dist/aiSdkTransport.js +160 -62
  22. package/dist/aiSdkTransport.js.map +1 -1
  23. package/dist/catalogProvider.d.ts +7 -3
  24. package/dist/catalogProvider.d.ts.map +1 -1
  25. package/dist/catalogProvider.js +30 -24
  26. package/dist/catalogProvider.js.map +1 -1
  27. package/dist/compatibleProvider.d.ts.map +1 -1
  28. package/dist/compatibleProvider.js +18 -7
  29. package/dist/compatibleProvider.js.map +1 -1
  30. package/dist/cost.d.ts +10 -10
  31. package/dist/cost.d.ts.map +1 -1
  32. package/dist/cost.js +90 -42
  33. package/dist/cost.js.map +1 -1
  34. package/dist/env.d.ts +5 -1
  35. package/dist/env.d.ts.map +1 -1
  36. package/dist/env.js +30 -10
  37. package/dist/env.js.map +1 -1
  38. package/dist/errors.d.ts +14 -2
  39. package/dist/errors.d.ts.map +1 -1
  40. package/dist/errors.js +58 -2
  41. package/dist/errors.js.map +1 -1
  42. package/dist/index.d.ts +4 -4
  43. package/dist/index.d.ts.map +1 -1
  44. package/dist/index.js +3 -2
  45. package/dist/index.js.map +1 -1
  46. package/dist/ollama.js +3 -3
  47. package/dist/ollama.js.map +1 -1
  48. package/dist/sdkModels.d.ts +6 -2
  49. package/dist/sdkModels.d.ts.map +1 -1
  50. package/dist/sdkModels.js +38 -5
  51. package/dist/sdkModels.js.map +1 -1
  52. package/dist/types.d.ts +33 -31
  53. package/dist/types.d.ts.map +1 -1
  54. package/dist/usage.d.ts +21 -5
  55. package/dist/usage.d.ts.map +1 -1
  56. package/dist/usage.js +164 -83
  57. package/dist/usage.js.map +1 -1
  58. package/package.json +7 -6
  59. package/src/AiSdkProvider.test.ts +788 -191
  60. package/src/AiSdkProvider.ts +381 -124
  61. package/src/Mock.test.ts +37 -12
  62. package/src/Mock.ts +45 -14
  63. package/src/Pool.test.ts +19 -6
  64. package/src/Pool.ts +20 -16
  65. package/src/ProviderRegistry.test.ts +16 -11
  66. package/src/accounting.test.ts +58 -22
  67. package/src/accounting.ts +120 -18
  68. package/src/aiSdkTransport.test.ts +42 -49
  69. package/src/aiSdkTransport.ts +174 -62
  70. package/src/boundaries.test.ts +1 -0
  71. package/src/catalogProvider.test.ts +258 -22
  72. package/src/catalogProvider.ts +42 -27
  73. package/src/compatibleProvider.test.ts +6 -3
  74. package/src/compatibleProvider.ts +20 -7
  75. package/src/cost.test.ts +55 -36
  76. package/src/cost.ts +111 -50
  77. package/src/defaults.test.ts +13 -3
  78. package/src/env.test.ts +54 -5
  79. package/src/env.ts +43 -18
  80. package/src/errors.test.ts +47 -2
  81. package/src/errors.ts +67 -3
  82. package/src/index.ts +21 -5
  83. package/src/ollama.test.ts +4 -1
  84. package/src/ollama.ts +3 -3
  85. package/src/sdkModels.test.ts +76 -4
  86. package/src/sdkModels.ts +45 -7
  87. package/src/types.ts +77 -38
  88. package/src/usage.test.ts +112 -116
  89. package/src/usage.ts +209 -93
@@ -5,13 +5,28 @@
5
5
  // Composition, not inheritance: an official AI SDK language model supplies the
6
6
  // ordinary vendor protocol. The compatible URL path remains only for PLURNK
7
7
  // extensions and local endpoint probes the SDK cannot represent.
8
- import { executeAiSdkModel, executeOpenAICompatible } from "./aiSdkTransport.js";
9
- import { toProviderError, ProviderError } from "./errors.js";
10
- import { attributeUnitemizedReasoning } from "./usage.js";
8
+ import { MAX_PROVIDER_TIMEOUT_MS } from "./env.js";
9
+ import { executeAiSdkModel, executeOpenAICompatible, transportFailureEvidence, } from "./aiSdkTransport.js";
10
+ import { prepareRetries } from "ai/internal";
11
+ import { toProviderError, ProviderError, ProviderTimeoutError } from "./errors.js";
11
12
  import { validateGbnf } from "@plurnk/gbnf";
12
13
  import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.js";
13
14
  import { emitWarningOnce } from "./warnings.js";
14
- import { validateAuthoritativeCharge } from "./cost.js";
15
+ import { resolveProviderCost } from "./cost.js";
16
+ import { validateProviderRequestAccounting } from "./accounting.js";
17
+ import { validateProviderUsage } from "./usage.js";
18
+ class ProviderRequestObserverError extends Error {
19
+ constructor(cause) {
20
+ super("provider request accounting could not be durably settled", { cause });
21
+ this.name = "ProviderRequestObserverError";
22
+ }
23
+ }
24
+ class ProviderRequestAccountingError extends Error {
25
+ constructor(cause) {
26
+ super("provider request accounting could not be normalized", { cause });
27
+ this.name = "ProviderRequestAccountingError";
28
+ }
29
+ }
15
30
  // Drop trailing occurrences of a server-rendered EOG marker. llama-server
16
31
  // under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
17
32
  // trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
@@ -52,11 +67,20 @@ const projectLeadingReasoning = (content, structuredReasoning, opening, closing)
52
67
  const projectTaggedReasoning = (content, structuredReasoning, style) => style === "think-tags"
53
68
  ? projectLeadingReasoning(content, structuredReasoning, "<think>", "</think>")
54
69
  : { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
55
- // llama-server's template reasoning parser can project this leading channel out
56
- // of the OpenAI-compatible response. Grammar evidence needs the sentence before
57
- // that lossy projection, so constrained template turns request it verbatim and
58
- // split the observed enclosure here.
59
- const projectTemplateReasoning = (content) => projectLeadingReasoning(content, "", "<|channel>thought\n", "<channel|>");
70
+ // llama-server's template reasoning parser can project either supported leading
71
+ // reasoning envelope out of the OpenAI-compatible response. Grammar evidence
72
+ // needs the sentence before that lossy projection, so constrained template turns
73
+ // request it verbatim and split the observed enclosure here.
74
+ const projectTemplateReasoning = (content) => {
75
+ for (const [opening, closing] of [
76
+ ["<|channel>thought\n", "<channel|>"],
77
+ ["<think>\n", "</think>"],
78
+ ]) {
79
+ if (content.startsWith(opening))
80
+ return projectLeadingReasoning(content, "", opening, closing);
81
+ }
82
+ return { content, reasoning: "", projected: false, contentStart: 0 };
83
+ };
60
84
  // Shared budget→effort breakpoints (xai and google had identical copies).
61
85
  export const effortFromBudget = (budget) => {
62
86
  if (budget <= 1000)
@@ -65,6 +89,11 @@ export const effortFromBudget = (budget) => {
65
89
  return "medium";
66
90
  return "high";
67
91
  };
92
+ // AI SDK's portable reasoning control has no boolean-enabled value. `medium`
93
+ // is the neutral activation projection for an explicit, unqualified `on`; it
94
+ // changes no PLURNK token reserve. An operator budget, when present, remains
95
+ // the only input to the existing magnitude-to-tier projection.
96
+ const effortFromReasoning = (reasoning) => reasoning.budget === null ? "medium" : effortFromBudget(reasoning.budget);
68
97
  // Body keys the provider owns — a caller's `sampling` passthrough may not set
69
98
  // these. Two families:
70
99
  // transport/managed — grammar transport, the stream/JSON choice, slot pinning,
@@ -91,6 +120,8 @@ export default class AiSdkProvider {
91
120
  #url;
92
121
  #languageModel;
93
122
  #fetchTimeoutMs;
123
+ #operationTimeoutMs;
124
+ #firstContentTimeoutMs;
94
125
  #streamIdleTimeoutMs;
95
126
  #headers;
96
127
  #fetch;
@@ -110,12 +141,13 @@ export default class AiSdkProvider {
110
141
  #reasoningResponseStyle;
111
142
  #countPromptTokens;
112
143
  #promptTokensUrl;
113
- #calculateCost;
114
- #calculateCharge;
115
- #normalizeCharge;
144
+ #estimateCost;
145
+ #normalizeCost;
116
146
  #source;
117
147
  #grammarStyle;
118
- #promptCacheKey;
148
+ #cacheAffinity;
149
+ #systemCacheProviderOptions;
150
+ #reasoningResponseProviderOptions;
119
151
  #serviceTier;
120
152
  #gbnfDebug;
121
153
  #streaming;
@@ -145,7 +177,19 @@ export default class AiSdkProvider {
145
177
  if ((this.#url === undefined) === (this.#languageModel === undefined)) {
146
178
  throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
147
179
  }
180
+ for (const [name, value] of [
181
+ ["fetchTimeoutMs", config.fetchTimeoutMs],
182
+ ["operationTimeoutMs", config.operationTimeoutMs],
183
+ ["firstContentTimeoutMs", config.firstContentTimeoutMs],
184
+ ["streamIdleTimeoutMs", config.streamIdleTimeoutMs],
185
+ ]) {
186
+ if (value !== undefined && (!Number.isInteger(value) || value < 0 || value > MAX_PROVIDER_TIMEOUT_MS)) {
187
+ throw new Error(`${config.source ?? "provider"}: ${name} must be an integer from 0 through ${MAX_PROVIDER_TIMEOUT_MS} milliseconds`);
188
+ }
189
+ }
148
190
  this.#fetchTimeoutMs = config.fetchTimeoutMs;
191
+ this.#operationTimeoutMs = config.operationTimeoutMs;
192
+ this.#firstContentTimeoutMs = config.firstContentTimeoutMs;
149
193
  this.#streamIdleTimeoutMs = config.streamIdleTimeoutMs;
150
194
  this.#headers = config.headers ?? {};
151
195
  this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
@@ -173,12 +217,33 @@ export default class AiSdkProvider {
173
217
  }
174
218
  this.#countPromptTokens = config.countPromptTokens ?? ((messages) => estimatePromptTokens(messages));
175
219
  this.#promptTokensUrl = config.promptTokensUrl;
176
- this.#calculateCost = config.calculateCost ?? (() => 0);
177
- this.#calculateCharge = config.calculateCharge;
178
- this.#normalizeCharge = config.normalizeCharge;
220
+ this.#estimateCost = config.estimateCost
221
+ ?? (() => ({
222
+ kind: "unknown",
223
+ reason: "the request reported no direct cost and no model rate is configured",
224
+ }));
225
+ this.#normalizeCost = config.normalizeCost;
179
226
  this.#source = config.source ?? "provider";
180
227
  this.#grammarStyle = config.grammarStyle ?? "none";
181
- this.#promptCacheKey = config.promptCacheKey ?? false;
228
+ this.#cacheAffinity = config.cacheAffinity;
229
+ this.#systemCacheProviderOptions = config.systemCacheProviderOptions;
230
+ this.#reasoningResponseProviderOptions = config.reasoningResponseProviderOptions;
231
+ if (this.#languageModel === undefined && this.#cacheAffinity?.target === "provider-option") {
232
+ throw new Error(`${this.#source}: native provider-option cache affinity requires an AI SDK model`);
233
+ }
234
+ if (this.#languageModel !== undefined && this.#cacheAffinity?.target === "body") {
235
+ throw new Error(`${this.#source}: body cache affinity requires an OpenAI-compatible URL`);
236
+ }
237
+ if (this.#languageModel === undefined && this.#systemCacheProviderOptions !== undefined) {
238
+ throw new Error(`${this.#source}: system cache provider options require an AI SDK model`);
239
+ }
240
+ if (this.#languageModel === undefined && this.#reasoningResponseProviderOptions !== undefined) {
241
+ throw new Error(`${this.#source}: reasoning response provider options require an AI SDK model`);
242
+ }
243
+ if (this.#cacheAffinity?.target === "provider-option"
244
+ && Object.hasOwn(this.#reasoningResponseProviderOptions?.[this.#cacheAffinity.provider] ?? {}, this.#cacheAffinity.name)) {
245
+ throw new Error(`${this.#source}: reasoning response options conflict with cache-affinity option ${this.#cacheAffinity.provider}.${this.#cacheAffinity.name}`);
246
+ }
182
247
  this.#serviceTier = config.serviceTier;
183
248
  this.#gbnfDebug = config.gbnfDebug ?? false;
184
249
  this.#streaming = config.streaming ?? true;
@@ -198,10 +263,17 @@ export default class AiSdkProvider {
198
263
  const reasoningReserve = this.reasoningReserve;
199
264
  if (this.#reasoningStyle === "template"
200
265
  && this.#reasoning.mode === "on"
266
+ && this.#reasoning.budget !== null
201
267
  && reasoningReserve !== null
202
268
  && this.#reasoning.budget > reasoningReserve) {
203
269
  throw new Error(`${this.#source}: PLURNK_PROVIDERS_REASONING_BUDGET (${this.#reasoning.budget}) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE (${reasoningReserve})`);
204
270
  }
271
+ if (this.#reasoningStyle === "anthropic"
272
+ && this.#reasoning.mode === "on"
273
+ && this.#reasoning.budget === null
274
+ && reasoningReserve === null) {
275
+ throw new Error(`${this.#source}: explicit Anthropic reasoning requires a resolved reasoning reserve or PLURNK_PROVIDERS_REASONING_BUDGET`);
276
+ }
205
277
  const { tokenizeUrl } = config;
206
278
  if (tokenizeUrl !== undefined) {
207
279
  this.tokenize = async (text) => {
@@ -209,7 +281,9 @@ export default class AiSdkProvider {
209
281
  method: "POST",
210
282
  headers: { "Content-Type": "application/json", ...this.#headers },
211
283
  body: JSON.stringify({ content: text }),
212
- signal: AbortSignal.timeout(this.#fetchTimeoutMs),
284
+ ...(this.#fetchTimeoutMs > 0
285
+ ? { signal: AbortSignal.timeout(this.#fetchTimeoutMs) }
286
+ : {}),
213
287
  });
214
288
  if (!res.ok)
215
289
  throw new Error(`${this.#source}: tokenize endpoint returned ${res.status}`);
@@ -248,7 +322,14 @@ export default class AiSdkProvider {
248
322
  }
249
323
  signal?.throwIfAborted();
250
324
  try {
251
- const timeout = AbortSignal.timeout(this.#fetchTimeoutMs);
325
+ const timeout = this.#fetchTimeoutMs > 0
326
+ ? AbortSignal.timeout(this.#fetchTimeoutMs)
327
+ : undefined;
328
+ const requestSignal = signal === undefined
329
+ ? timeout
330
+ : timeout === undefined
331
+ ? signal
332
+ : AbortSignal.any([signal, timeout]);
252
333
  const response = await this.#fetch(this.#promptTokensUrl, {
253
334
  method: "POST",
254
335
  headers: { "Content-Type": "application/json", ...this.#headers },
@@ -257,7 +338,7 @@ export default class AiSdkProvider {
257
338
  messages,
258
339
  ...this.#reasoningBody(),
259
340
  }),
260
- signal: signal === undefined ? timeout : AbortSignal.any([signal, timeout]),
341
+ ...(requestSignal === undefined ? {} : { signal: requestSignal }),
261
342
  });
262
343
  if (!response.ok) {
263
344
  return estimatePromptTokens(messages, `llama-server input-token endpoint returned HTTP ${response.status}`);
@@ -277,11 +358,6 @@ export default class AiSdkProvider {
277
358
  return estimatePromptTokens(messages, `llama-server input-token measurement failed: ${cause instanceof Error ? cause.message : String(cause)}`);
278
359
  }
279
360
  }
280
- calculateCost(usage) { return this.#calculateCost(usage); }
281
- calculateCharge(usage) {
282
- return this.#calculateCharge?.(usage)
283
- ?? { kind: "unknown", reason: "no provider rate or settled charge is available" };
284
- }
285
361
  // Reasoning activation and allowance are independent of grammar transport;
286
362
  // only the response representation becomes lossless when evidence is needed.
287
363
  // The llama-server template mapping is owned by {§llama-reasoning-request}.
@@ -292,7 +368,7 @@ export default class AiSdkProvider {
292
368
  case "template": {
293
369
  const allowance = mode === "off"
294
370
  ? 0
295
- : mode === "on" ? budget : this.reasoningReserve;
371
+ : mode === "on" && budget !== null ? budget : this.reasoningReserve;
296
372
  return {
297
373
  chat_template_kwargs: { enable_thinking: on },
298
374
  reasoning_format: preserveGrammarSentence ? "none" : "auto",
@@ -301,9 +377,9 @@ export default class AiSdkProvider {
301
377
  }
302
378
  case "think": return on ? { think: true } : {};
303
379
  case "include_reasoning": return on ? { include_reasoning: true } : {};
304
- // effort tiers from the budget; off/adaptive omit the field (the
305
- // API's default depth is its adaptive).
306
- case "effort": return mode === "on" ? { reasoning_effort: effortFromBudget(budget) } : {};
380
+ // Explicit on uses the portable enabled posture or a tier derived
381
+ // from an explicit budget; off/adaptive omit the field.
382
+ case "effort": return mode === "on" ? { reasoning_effort: effortFromReasoning(this.#reasoning) } : {};
307
383
  // Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
308
384
  // reason-by-default model (DeepSeek V4: default 'high') reasoning.
309
385
  // ADAPTIVE omits the field: the backend's own default posture IS the
@@ -313,19 +389,25 @@ export default class AiSdkProvider {
313
389
  // efforts 400.
314
390
  case "effort_explicit": return mode === "off"
315
391
  ? { reasoning_effort: "none" }
316
- : mode === "on" ? { reasoning_effort: effortFromBudget(budget) } : {};
392
+ : mode === "on" ? { reasoning_effort: effortFromReasoning(this.#reasoning) } : {};
317
393
  // {§deepseek-reasoning-request}
318
394
  case "thinking_effort": return mode === "off"
319
395
  ? { thinking: { type: "disabled" } }
320
396
  : mode === "on" ? {
321
397
  thinking: { type: "enabled" },
322
- reasoning_effort: effortFromBudget(budget),
398
+ ...(budget === null ? {} : { reasoning_effort: effortFromBudget(budget) }),
323
399
  } : {};
324
400
  // Anthropic compat: explicit thinking object. off → disabled; on →
325
- // enabled with budget_tokens; adaptive → omit (the API default).
401
+ // enabled with the explicit budget or resolved reserve; adaptive →
402
+ // omit (the API default).
326
403
  case "anthropic": return mode === "off"
327
404
  ? { thinking: { type: "disabled" } }
328
- : mode === "on" ? { thinking: { type: "enabled", budget_tokens: budget } } : {};
405
+ : mode === "on" ? {
406
+ thinking: {
407
+ type: "enabled",
408
+ budget_tokens: budget ?? this.reasoningReserve,
409
+ },
410
+ } : {};
329
411
  case "none": return {};
330
412
  }
331
413
  }
@@ -390,7 +472,7 @@ export default class AiSdkProvider {
390
472
  case "none": return this.#frequencyPenalty > 0 ? { frequency_penalty: this.#frequencyPenalty } : {};
391
473
  }
392
474
  }
393
- // First-party telemetry headers ({§provider-request-authority}): forwarded only when the spec
475
+ // First-party telemetry headers ({§provider-request-authority} {§provider-call-kind}): forwarded only when the spec
394
476
  // opted in (the plurnk endpoint). The gate is here, not at the call site, so
395
477
  // attributions/client/strikes can never reach a third-party backend even if
396
478
  // the consumer passes them to the wrong provider. Empty values emit no header
@@ -398,7 +480,7 @@ export default class AiSdkProvider {
398
480
  // absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
399
481
  // ride HTTP headers only — the packet never carries them (the model must
400
482
  // never see strike state; engine accounting is not a metric to game).
401
- #metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn) {
483
+ #metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind) {
402
484
  if (!this.#firstPartyMetadata)
403
485
  return {};
404
486
  const h = {};
@@ -431,6 +513,8 @@ export default class AiSdkProvider {
431
513
  h["Plurnk-Loop"] = String(loop);
432
514
  if (turn !== undefined && Number.isInteger(turn) && turn >= 1)
433
515
  h["Plurnk-Turn"] = String(turn);
516
+ if (callKind !== undefined)
517
+ h["Plurnk-Call-Kind"] = callKind;
434
518
  return h;
435
519
  }
436
520
  // PLURNK_PROVIDERS_GBNF_DEBUG ({§gbnf-response-observation}): validate the supplied GBNF locally and fail
@@ -470,10 +554,40 @@ export default class AiSdkProvider {
470
554
  out[k] = v;
471
555
  return out;
472
556
  }
473
- async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }) {
557
+ #requestProviderOptions(workerId) {
558
+ const reasoningOptions = this.#reasoning.mode === "off"
559
+ ? undefined
560
+ : this.#reasoningResponseProviderOptions;
561
+ if (this.#cacheAffinity?.target !== "provider-option")
562
+ return reasoningOptions;
563
+ const { provider, name } = this.#cacheAffinity;
564
+ return {
565
+ ...reasoningOptions,
566
+ [provider]: {
567
+ ...reasoningOptions?.[provider],
568
+ [name]: workerId,
569
+ },
570
+ };
571
+ }
572
+ #accounting(outcome, usage, evidence, status) {
573
+ const knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
574
+ const direct = this.#normalizeCost?.(evidence);
575
+ return validateProviderRequestAccounting({
576
+ provider: this.#source,
577
+ model: this.#model,
578
+ outcome,
579
+ ...(status === undefined ? {} : { status }),
580
+ ...(knownUsage === undefined ? {} : { usage: knownUsage }),
581
+ cost: resolveProviderCost(direct, this.#estimateCost(knownUsage)),
582
+ });
583
+ }
584
+ async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling, observeRequest, callKind }) {
474
585
  // {§provider-interface} The worker identity is required.
475
586
  if (workerId === undefined || workerId.length === 0)
476
587
  throw new Error("generate: workerId is required — the worker's stable, opaque identity");
588
+ if (callKind !== undefined && callKind !== "emission" && callKind !== "bare") {
589
+ throw new Error(`generate: unsupported callKind ${JSON.stringify(callKind)}`);
590
+ }
477
591
  // Reject before any wire call when already aborted
478
592
  // ({§provider-failure-normalization}).
479
593
  signal?.throwIfAborted();
@@ -504,73 +618,174 @@ export default class AiSdkProvider {
504
618
  // reserved from caller sampling; the env flag is the single control).
505
619
  ...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
506
620
  ...this.#slotBody(workerId),
507
- // Prompt-cache affinity -- workerId as the OpenAI-standard
508
- // prompt_cache_key routes a worker's turns to one serverless replica so
509
- // its stable prefix caches (managed; reserved from caller sampling).
510
- ...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
621
+ ...(this.#cacheAffinity?.target === "body"
622
+ ? { [this.#cacheAffinity.name]: workerId }
623
+ : {}),
511
624
  };
512
625
  // Per-request headers = static auth/routing + any first-party telemetry.
513
- const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
514
- const headers = Object.keys(metaHeaders).length === 0
515
- ? this.#headers
516
- : { ...this.#headers, ...metaHeaders };
517
- let raw;
518
- try {
519
- raw = this.#languageModel === undefined
520
- ? await executeOpenAICompatible({
521
- url: this.#url,
626
+ const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind);
627
+ const headers = new Headers(this.#headers);
628
+ if (this.#cacheAffinity?.target === "header") {
629
+ headers.set(this.#cacheAffinity.name, workerId);
630
+ }
631
+ for (const [name, value] of Object.entries(metaHeaders))
632
+ headers.set(name, value);
633
+ const requestHeaders = Object.fromEntries(headers.entries());
634
+ const accounting = [];
635
+ const operationTimeout = this.#operationTimeoutMs > 0
636
+ ? AbortSignal.timeout(this.#operationTimeoutMs)
637
+ : undefined;
638
+ const operationSignal = signal === undefined
639
+ ? operationTimeout
640
+ : operationTimeout === undefined
641
+ ? signal
642
+ : AbortSignal.any([signal, operationTimeout]);
643
+ const executeRequest = async () => {
644
+ let settle;
645
+ try {
646
+ settle = await observeRequest?.({
647
+ provider: this.#source,
522
648
  model: this.#model,
523
- headers,
524
- body,
525
- messages,
526
- signal,
527
- fetch: this.#fetch,
528
- fetchTimeoutMs: this.#fetchTimeoutMs,
529
- streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
530
- retryAttempts: this.#retryAttempts,
531
- streaming: this.#streaming,
532
- captureRawBody: this.#rawBody,
533
- })
534
- : await executeAiSdkModel({
535
- languageModel: this.#languageModel,
536
- headers,
537
- messages,
538
- signal,
539
- fetchTimeoutMs: this.#fetchTimeoutMs,
540
- streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
541
- retryAttempts: this.#retryAttempts,
542
- streaming: this.#streaming,
543
- captureRawBody: this.#rawBody,
544
- temperature: this.#tuningFloors
545
- ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
546
- : typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
547
- topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
548
- topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
549
- presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
550
- frequencyPenalty: typeof sampling?.frequency_penalty === "number"
551
- ? sampling.frequency_penalty
552
- : this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
553
- stopSequences: typeof sampling?.stop === "string"
554
- ? [sampling.stop]
555
- : Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
556
- ? sampling.stop
557
- : undefined,
558
- seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
559
- maxOutputTokens: maxTokens,
560
- reasoning: this.#reasoning.mode === "off"
561
- ? "none"
562
- : this.#reasoning.mode === "adaptive"
563
- ? "provider-default"
564
- : effortFromBudget(this.#reasoning.budget),
565
649
  });
650
+ }
651
+ catch (cause) {
652
+ throw new ProviderRequestObserverError(cause);
653
+ }
654
+ const settleAccounting = async (outcome, usage, evidence, status) => {
655
+ let requestAccounting;
656
+ let normalizationFailure;
657
+ try {
658
+ requestAccounting = this.#accounting(outcome, usage, evidence, status);
659
+ }
660
+ catch (cause) {
661
+ normalizationFailure = { cause };
662
+ let knownUsage;
663
+ try {
664
+ knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
665
+ }
666
+ catch {
667
+ knownUsage = undefined;
668
+ }
669
+ requestAccounting = validateProviderRequestAccounting({
670
+ provider: this.#source,
671
+ model: this.#model,
672
+ outcome,
673
+ ...(status === undefined ? {} : { status }),
674
+ ...(knownUsage === undefined ? {} : { usage: knownUsage }),
675
+ cost: {
676
+ kind: "unknown",
677
+ reason: "provider request accounting could not be normalized after physical I/O",
678
+ },
679
+ });
680
+ }
681
+ accounting.push(requestAccounting);
682
+ try {
683
+ await settle?.(requestAccounting);
684
+ }
685
+ catch (cause) {
686
+ throw new ProviderRequestObserverError(cause);
687
+ }
688
+ if (normalizationFailure !== undefined) {
689
+ throw new ProviderRequestAccountingError(normalizationFailure.cause);
690
+ }
691
+ return requestAccounting;
692
+ };
693
+ let response;
694
+ try {
695
+ response = this.#languageModel === undefined
696
+ ? await executeOpenAICompatible({
697
+ url: this.#url,
698
+ model: this.#model,
699
+ headers: requestHeaders,
700
+ body,
701
+ messages,
702
+ signal: operationSignal,
703
+ fetch: this.#fetch,
704
+ fetchTimeoutMs: this.#fetchTimeoutMs,
705
+ firstContentTimeoutMs: this.#firstContentTimeoutMs,
706
+ streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
707
+ streaming: this.#streaming,
708
+ captureRawBody: this.#rawBody,
709
+ })
710
+ : await executeAiSdkModel({
711
+ languageModel: this.#languageModel,
712
+ headers: requestHeaders,
713
+ providerOptions: this.#requestProviderOptions(workerId),
714
+ systemProviderOptions: this.#systemCacheProviderOptions,
715
+ messages,
716
+ signal: operationSignal,
717
+ fetchTimeoutMs: this.#fetchTimeoutMs,
718
+ firstContentTimeoutMs: this.#firstContentTimeoutMs,
719
+ streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
720
+ streaming: this.#streaming,
721
+ captureRawBody: this.#rawBody,
722
+ temperature: this.#tuningFloors
723
+ ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
724
+ : typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
725
+ topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
726
+ topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
727
+ presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
728
+ frequencyPenalty: typeof sampling?.frequency_penalty === "number"
729
+ ? sampling.frequency_penalty
730
+ : this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
731
+ stopSequences: typeof sampling?.stop === "string"
732
+ ? [sampling.stop]
733
+ : Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
734
+ ? sampling.stop
735
+ : undefined,
736
+ seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
737
+ maxOutputTokens: maxTokens,
738
+ reasoning: this.#reasoning.mode === "off"
739
+ ? "none"
740
+ : this.#reasoning.mode === "adaptive"
741
+ ? "provider-default"
742
+ : effortFromReasoning(this.#reasoning),
743
+ });
744
+ }
745
+ catch (error) {
746
+ const failure = transportFailureEvidence(error);
747
+ await settleAccounting("error", failure.usage, failure.chargeEvidence, failure.status);
748
+ throw error;
749
+ }
750
+ await settleAccounting("response", response.usage, response.chargeEvidence);
751
+ return response;
752
+ };
753
+ let raw;
754
+ try {
755
+ const { retry } = prepareRetries({
756
+ maxRetries: this.#retryAttempts,
757
+ abortSignal: operationSignal,
758
+ });
759
+ raw = await retry(executeRequest);
566
760
  }
567
761
  catch (err) {
762
+ if (err instanceof ProviderRequestObserverError
763
+ || err instanceof ProviderRequestAccountingError)
764
+ throw err.cause;
568
765
  if (signal?.aborted)
569
766
  throw err;
767
+ if (operationTimeout?.aborted) {
768
+ const timeout = new ProviderTimeoutError("operation", this.#operationTimeoutMs, err);
769
+ throw new ProviderError(this.#source, "deadline_exceeded", timeout.message, {
770
+ status: 504,
771
+ cause: timeout,
772
+ retryable: false,
773
+ extensions: {
774
+ timeoutPhase: timeout.phase,
775
+ timeoutMs: timeout.timeoutMs,
776
+ },
777
+ accounting,
778
+ });
779
+ }
570
780
  const pe = toProviderError(err, this.#source, this.#errorDetailLimit);
571
781
  if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
572
- throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
782
+ throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, {
783
+ status: pe.status,
784
+ cause: err,
785
+ accounting,
786
+ });
573
787
  }
788
+ pe.prependAccounting(accounting);
574
789
  throw pe;
575
790
  }
576
791
  // llama-server --special renders EOG tokens as text, so a turn ending
@@ -585,10 +800,10 @@ export default class AiSdkProvider {
585
800
  const projectedReasoning = preserveGrammarSentence && !raw.reasoningProjected
586
801
  ? projectTemplateReasoning(raw.content)
587
802
  : projectTaggedReasoning(raw.content, raw.reasoning, this.#reasoningResponseStyle);
588
- // Preserve the exact sentence seen at the grammar boundary. Constrained
589
- // template turns request `reasoning_format: "none"`, so even an empty
590
- // channel remains observable. An unexpectedly projected response cannot
591
- // supply independent pre-projection evidence.
803
+ // Preserve the exact pre-projection response. Constrained template turns
804
+ // request `reasoning_format: "none"`, so even an empty channel and any
805
+ // template-provided opener remain observable. An unexpectedly projected
806
+ // response cannot supply independent evidence.
592
807
  let grammarEvidence;
593
808
  if (wantGrammar) {
594
809
  if (preserveGrammarSentence) {
@@ -618,11 +833,12 @@ export default class AiSdkProvider {
618
833
  if (projectedReasoning.projected) {
619
834
  raw.content = projectedReasoning.content;
620
835
  raw.reasoning = projectedReasoning.reasoning;
621
- raw.usage = attributeUnitemizedReasoning(raw.usage, raw.reasoning, raw.content);
622
836
  }
623
837
  let notices;
624
838
  const usage = raw.usage;
625
- if (sendGrammar !== undefined && this.tokenize !== undefined) {
839
+ if (sendGrammar !== undefined
840
+ && this.tokenize !== undefined
841
+ && usage?.outputTokens !== undefined) {
626
842
  // Channel-escape detector: completion tokens
627
843
  // billed far beyond every visible channel mean the decode ESCAPED into
628
844
  // a server-discarded reasoning block mid-emission. This diagnostic
@@ -633,12 +849,12 @@ export default class AiSdkProvider {
633
849
  this.tokenize(raw.reasoning),
634
850
  ]);
635
851
  const visible = contentTokens.length + reasoningTokens.length;
636
- if (usage.completion > visible + 64) {
852
+ if (usage.outputTokens > visible + 64) {
637
853
  (notices ??= []).push({
638
854
  source: this.#source,
639
855
  kind: "grammar_unenforced",
640
856
  level: "warn",
641
- message: `decode escaped the grammar: ${usage.completion} completion tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
857
+ message: `decode escaped the grammar: ${usage.outputTokens} output tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
642
858
  position: [...raw.content].length,
643
859
  });
644
860
  }
@@ -658,17 +874,12 @@ export default class AiSdkProvider {
658
874
  ...(raw.reasoningEncrypted.length > 0
659
875
  ? { reasoningEncrypted: raw.reasoningEncrypted }
660
876
  : {}),
661
- usage,
662
877
  model: raw.model,
663
878
  ...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
664
879
  };
665
- const normalizedCharge = this.#normalizeCharge?.(raw.chargeEvidence);
666
- const charge = normalizedCharge === undefined
667
- ? undefined
668
- : validateAuthoritativeCharge(normalizedCharge);
669
880
  const evidence = {
670
881
  assistantRaw: raw,
671
- ...(charge === undefined ? {} : { charge }),
882
+ accounting,
672
883
  ...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
673
884
  ...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
674
885
  ...(meta !== undefined ? { meta } : {}),
@@ -681,6 +892,7 @@ export default class AiSdkProvider {
681
892
  };
682
893
  throw new ProviderError(this.#source, "resource_interrupted", "The provider interrupted generation because inference resources were unavailable.", {
683
894
  attempt,
895
+ accounting,
684
896
  extensions: {
685
897
  stage: "provider-response",
686
898
  finishReason: "resource_interrupted",