@plurnk/plurnk-providers 1.5.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. package/.env.defaults +36 -22
  2. package/SPEC.md +133 -59
  3. package/dist/AiSdkProvider.d.ts +19 -26
  4. package/dist/AiSdkProvider.d.ts.map +1 -1
  5. package/dist/AiSdkProvider.js +318 -106
  6. package/dist/AiSdkProvider.js.map +1 -1
  7. package/dist/Mock.d.ts +4 -9
  8. package/dist/Mock.d.ts.map +1 -1
  9. package/dist/Mock.js +36 -9
  10. package/dist/Mock.js.map +1 -1
  11. package/dist/Pool.d.ts +2 -21
  12. package/dist/Pool.d.ts.map +1 -1
  13. package/dist/Pool.js +19 -14
  14. package/dist/Pool.js.map +1 -1
  15. package/dist/accounting.d.ts +5 -2
  16. package/dist/accounting.d.ts.map +1 -1
  17. package/dist/accounting.js +100 -16
  18. package/dist/accounting.js.map +1 -1
  19. package/dist/aiSdkTransport.d.ts +9 -2
  20. package/dist/aiSdkTransport.d.ts.map +1 -1
  21. package/dist/aiSdkTransport.js +160 -62
  22. package/dist/aiSdkTransport.js.map +1 -1
  23. package/dist/catalogProvider.d.ts +7 -3
  24. package/dist/catalogProvider.d.ts.map +1 -1
  25. package/dist/catalogProvider.js +30 -24
  26. package/dist/catalogProvider.js.map +1 -1
  27. package/dist/compatibleProvider.d.ts.map +1 -1
  28. package/dist/compatibleProvider.js +18 -7
  29. package/dist/compatibleProvider.js.map +1 -1
  30. package/dist/cost.d.ts +10 -10
  31. package/dist/cost.d.ts.map +1 -1
  32. package/dist/cost.js +90 -42
  33. package/dist/cost.js.map +1 -1
  34. package/dist/env.d.ts +5 -1
  35. package/dist/env.d.ts.map +1 -1
  36. package/dist/env.js +30 -10
  37. package/dist/env.js.map +1 -1
  38. package/dist/errors.d.ts +14 -2
  39. package/dist/errors.d.ts.map +1 -1
  40. package/dist/errors.js +58 -2
  41. package/dist/errors.js.map +1 -1
  42. package/dist/index.d.ts +4 -4
  43. package/dist/index.d.ts.map +1 -1
  44. package/dist/index.js +3 -2
  45. package/dist/index.js.map +1 -1
  46. package/dist/ollama.js +3 -3
  47. package/dist/ollama.js.map +1 -1
  48. package/dist/sdkModels.d.ts +6 -2
  49. package/dist/sdkModels.d.ts.map +1 -1
  50. package/dist/sdkModels.js +38 -5
  51. package/dist/sdkModels.js.map +1 -1
  52. package/dist/types.d.ts +33 -31
  53. package/dist/types.d.ts.map +1 -1
  54. package/dist/usage.d.ts +21 -5
  55. package/dist/usage.d.ts.map +1 -1
  56. package/dist/usage.js +164 -83
  57. package/dist/usage.js.map +1 -1
  58. package/package.json +7 -6
  59. package/src/AiSdkProvider.test.ts +788 -191
  60. package/src/AiSdkProvider.ts +381 -124
  61. package/src/Mock.test.ts +37 -12
  62. package/src/Mock.ts +45 -14
  63. package/src/Pool.test.ts +19 -6
  64. package/src/Pool.ts +20 -16
  65. package/src/ProviderRegistry.test.ts +16 -11
  66. package/src/accounting.test.ts +58 -22
  67. package/src/accounting.ts +120 -18
  68. package/src/aiSdkTransport.test.ts +42 -49
  69. package/src/aiSdkTransport.ts +174 -62
  70. package/src/boundaries.test.ts +1 -0
  71. package/src/catalogProvider.test.ts +258 -22
  72. package/src/catalogProvider.ts +42 -27
  73. package/src/compatibleProvider.test.ts +6 -3
  74. package/src/compatibleProvider.ts +20 -7
  75. package/src/cost.test.ts +55 -36
  76. package/src/cost.ts +111 -50
  77. package/src/defaults.test.ts +13 -3
  78. package/src/env.test.ts +54 -5
  79. package/src/env.ts +43 -18
  80. package/src/errors.test.ts +47 -2
  81. package/src/errors.ts +67 -3
  82. package/src/index.ts +21 -5
  83. package/src/ollama.test.ts +4 -1
  84. package/src/ollama.ts +3 -3
  85. package/src/sdkModels.test.ts +76 -4
  86. package/src/sdkModels.ts +45 -7
  87. package/src/types.ts +77 -38
  88. package/src/usage.test.ts +112 -116
  89. package/src/usage.ts +209 -93
@@ -6,19 +6,40 @@
6
6
  // ordinary vendor protocol. The compatible URL path remains only for PLURNK
7
7
  // extensions and local endpoint probes the SDK cannot represent.
8
8
 
9
- import type { AuthoritativeChargeNormalizer, ChatMessage, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
9
+ import type {
10
+ ChatMessage,
11
+ GrammarEvidence,
12
+ PromptTokenMeasurement,
13
+ Provider,
14
+ ProviderCostNormalizer,
15
+ ProviderCallKind,
16
+ ProviderGenerateArgs,
17
+ ProviderRequestAccounting,
18
+ ProviderRequestObserver,
19
+ ProviderRequestSettlement,
20
+ ProviderResponse,
21
+ ProviderUsage,
22
+ } from "./types.ts";
10
23
  import type { ProviderCost } from "@plurnk/plurnk-contracts";
24
+ import type { JSONValue } from "ai";
25
+ import { MAX_PROVIDER_TIMEOUT_MS } from "./env.ts";
11
26
  import type { Reasoning, ReasoningResponseStyle, ReserveSpec } from "./env.ts";
12
- import { executeAiSdkModel, executeOpenAICompatible } from "./aiSdkTransport.ts";
27
+ import {
28
+ executeAiSdkModel,
29
+ executeOpenAICompatible,
30
+ transportFailureEvidence,
31
+ } from "./aiSdkTransport.ts";
13
32
  import type { LanguageModel } from "ai";
14
- import { toProviderError, ProviderError } from "./errors.ts";
15
- import { attributeUnitemizedReasoning } from "./usage.ts";
33
+ import { prepareRetries } from "ai/internal";
34
+ import { toProviderError, ProviderError, ProviderTimeoutError } from "./errors.ts";
16
35
  import type { ProviderNotice } from "./notices.ts";
17
36
  import { validateGbnf } from "@plurnk/gbnf";
18
37
  import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.ts";
19
38
  import { emitWarningOnce } from "./warnings.ts";
20
39
  import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
21
- import { validateAuthoritativeCharge } from "./cost.ts";
40
+ import { resolveProviderCost } from "./cost.ts";
41
+ import { validateProviderRequestAccounting } from "./accounting.ts";
42
+ import { validateProviderUsage } from "./usage.ts";
22
43
 
23
44
  export type ProviderFetch = typeof globalThis.fetch;
24
45
 
@@ -30,30 +51,40 @@ export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" |
30
51
  // service-managed constrained sampling; endpoint-owned settings are not inferred.
31
52
  export type GrammarStyle = "none" | "llamacpp";
32
53
 
54
+ export type CacheAffinity =
55
+ | { readonly target: "header" | "body"; readonly name: string }
56
+ | { readonly target: "provider-option"; readonly provider: string; readonly name: string };
57
+
58
+ export type AiSdkProviderOptions = Record<string, Record<string, JSONValue | undefined>>;
59
+
33
60
  export type AiSdkProviderConfig = {
34
61
  model: string;
35
62
  url?: string; // OpenAI-compatible chat-completions URL
36
63
  languageModel?: LanguageModel; // native AI SDK provider model
37
64
  attributions?: (context: PluginAttributionContext) => PluginAttribution;
38
- fetchTimeoutMs: number;
39
- streamIdleTimeoutMs?: number; // streamed body inter-chunk deadline; zero/unset disables
65
+ fetchTimeoutMs: number; // one physical generation attempt; zero disables
66
+ operationTimeoutMs: number; // complete logical call across retries/backoff; zero disables
67
+ firstContentTimeoutMs: number; // first semantic streamed content; zero disables
68
+ streamIdleTimeoutMs?: number; // semantic streamed-content idle deadline; zero/unset disables
40
69
  headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
41
70
  fetch?: ProviderFetch; // per-instance request executor; default globalThis.fetch
42
71
  contextWindow?: number | null; // default null; caller resolves-or-fails, narrows to required with the interface
43
72
  reasoningStyle?: ReasoningStyle; // default "none"
44
73
  reasoningResponseStyle?: ReasoningResponseStyle; // {§provider-tagged-reasoning}; default "verbatim"
45
74
  countPromptTokens?: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
46
- calculateCost?: (usage: ProviderUsage) => number; // default () => 0
47
- calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }>;
48
- normalizeCharge?: AuthoritativeChargeNormalizer;
75
+ estimateCost?: (usage: ProviderUsage | undefined) => ProviderCost;
76
+ normalizeCost?: ProviderCostNormalizer;
49
77
  source?: string; // notice/problem source, e.g. "provider:openai"; default "provider"
50
78
  grammarStyle?: GrammarStyle; // how a GBNF grammar is carried; default "none" (not sent)
51
- // Send the OpenAI-standard `prompt_cache_key` set to workerId, so a
52
- // serverless backend with REPLICA-LOCAL prompt caching (fireworks, verified)
53
- // pins a worker's turns to one replica and claims its stable prefix. Default
54
- // false -- a backend that strict-validates unknown fields 400s, so enable only
55
- // where the field is accepted. Same identity that already drives slot affinity.
56
- promptCacheKey?: boolean;
79
+ // {§provider-cache-affinity} Provider routes own the exact documented
80
+ // projection; the common transport only applies it as managed request state.
81
+ cacheAffinity?: CacheAffinity;
82
+ // {§provider-cache-write-policy} Already policy-gated by provider construction.
83
+ // The transport attaches it to only the final leading system instruction.
84
+ systemCacheProviderOptions?: AiSdkProviderOptions;
85
+ // {§provider-readable-reasoning} Route-owned native option needed to expose
86
+ // readable reasoning. Applied only when the effective posture is not off.
87
+ reasoningResponseProviderOptions?: AiSdkProviderOptions;
57
88
  // Optional provider-configured service tier. Unlike caller sampling, this is
58
89
  // a fixed deployment choice and therefore wins on every request.
59
90
  serviceTier?: string;
@@ -83,9 +114,9 @@ export type AiSdkProviderConfig = {
83
114
  requiresMaxTokens?: boolean;
84
115
  // The side-channel reasoning intent — REQUIRED, no in-code default
85
116
  // (PLURNK_PROVIDERS_REASONING + _BUDGET, read via reasoningFromEnv):
86
- // { mode: off|adaptive|on, budget: iff on }. The provider maps it to the
87
- // backend's mechanism via reasoningStyle; budget is only ever a magnitude,
88
- // never a hidden activation flag.
117
+ // { mode: off|adaptive|on, budget: optional when on }. The provider maps it
118
+ // to the backend's mechanism via reasoningStyle; budget is only ever an
119
+ // explicit magnitude, never a hidden activation flag.
89
120
  reasoning: Reasoning;
90
121
  // Decode tuning: no in-code defaults; the canonical measured values (0.2 /
91
122
  // 1.15) ship as the floor in .env.defaults (alias-scopable). `temperature` is the
@@ -140,6 +171,20 @@ export type AiSdkProviderConfig = {
140
171
  tuningFloors?: boolean;
141
172
  };
142
173
 
174
+ class ProviderRequestObserverError extends Error {
175
+ constructor(cause: unknown) {
176
+ super("provider request accounting could not be durably settled", { cause });
177
+ this.name = "ProviderRequestObserverError";
178
+ }
179
+ }
180
+
181
+ class ProviderRequestAccountingError extends Error {
182
+ constructor(cause: unknown) {
183
+ super("provider request accounting could not be normalized", { cause });
184
+ this.name = "ProviderRequestAccountingError";
185
+ }
186
+ }
187
+
143
188
  // Drop trailing occurrences of a server-rendered EOG marker. llama-server
144
189
  // under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
145
190
  // trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
@@ -197,12 +242,19 @@ const projectTaggedReasoning = (
197
242
  ? projectLeadingReasoning(content, structuredReasoning, "<think>", "</think>")
198
243
  : { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
199
244
 
200
- // llama-server's template reasoning parser can project this leading channel out
201
- // of the OpenAI-compatible response. Grammar evidence needs the sentence before
202
- // that lossy projection, so constrained template turns request it verbatim and
203
- // split the observed enclosure here.
204
- const projectTemplateReasoning = (content: string): TaggedReasoningProjection =>
205
- projectLeadingReasoning(content, "", "<|channel>thought\n", "<channel|>");
245
+ // llama-server's template reasoning parser can project either supported leading
246
+ // reasoning envelope out of the OpenAI-compatible response. Grammar evidence
247
+ // needs the sentence before that lossy projection, so constrained template turns
248
+ // request it verbatim and split the observed enclosure here.
249
+ const projectTemplateReasoning = (content: string): TaggedReasoningProjection => {
250
+ for (const [opening, closing] of [
251
+ ["<|channel>thought\n", "<channel|>"],
252
+ ["<think>\n", "</think>"],
253
+ ] as const) {
254
+ if (content.startsWith(opening)) return projectLeadingReasoning(content, "", opening, closing);
255
+ }
256
+ return { content, reasoning: "", projected: false, contentStart: 0 };
257
+ };
206
258
 
207
259
  // Shared budget→effort breakpoints (xai and google had identical copies).
208
260
  export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
@@ -211,6 +263,13 @@ export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
211
263
  return "high";
212
264
  };
213
265
 
266
+ // AI SDK's portable reasoning control has no boolean-enabled value. `medium`
267
+ // is the neutral activation projection for an explicit, unqualified `on`; it
268
+ // changes no PLURNK token reserve. An operator budget, when present, remains
269
+ // the only input to the existing magnitude-to-tier projection.
270
+ const effortFromReasoning = (reasoning: Reasoning): "low" | "medium" | "high" =>
271
+ reasoning.budget === null ? "medium" : effortFromBudget(reasoning.budget);
272
+
214
273
  // Body keys the provider owns — a caller's `sampling` passthrough may not set
215
274
  // these. Two families:
216
275
  // transport/managed — grammar transport, the stream/JSON choice, slot pinning,
@@ -238,6 +297,8 @@ export default class AiSdkProvider implements Provider {
238
297
  #url: string | undefined;
239
298
  #languageModel: LanguageModel | undefined;
240
299
  #fetchTimeoutMs: number;
300
+ #operationTimeoutMs: number;
301
+ #firstContentTimeoutMs: number;
241
302
  #streamIdleTimeoutMs: number | undefined;
242
303
  #headers: Record<string, string>;
243
304
  #fetch: ProviderFetch;
@@ -257,12 +318,13 @@ export default class AiSdkProvider implements Provider {
257
318
  #reasoningResponseStyle: ReasoningResponseStyle;
258
319
  #countPromptTokens: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
259
320
  #promptTokensUrl: string | undefined;
260
- #calculateCost: (usage: ProviderUsage) => number;
261
- #calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }>;
262
- #normalizeCharge?: AuthoritativeChargeNormalizer;
321
+ #estimateCost: (usage: ProviderUsage | undefined) => ProviderCost;
322
+ #normalizeCost?: ProviderCostNormalizer;
263
323
  #source: string;
264
324
  #grammarStyle: GrammarStyle;
265
- #promptCacheKey: boolean;
325
+ #cacheAffinity: CacheAffinity | undefined;
326
+ #systemCacheProviderOptions: AiSdkProviderOptions | undefined;
327
+ #reasoningResponseProviderOptions: AiSdkProviderOptions | undefined;
266
328
  #serviceTier: string | undefined;
267
329
  #gbnfDebug: boolean;
268
330
  #streaming: boolean;
@@ -293,7 +355,19 @@ export default class AiSdkProvider implements Provider {
293
355
  if ((this.#url === undefined) === (this.#languageModel === undefined)) {
294
356
  throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
295
357
  }
358
+ for (const [name, value] of [
359
+ ["fetchTimeoutMs", config.fetchTimeoutMs],
360
+ ["operationTimeoutMs", config.operationTimeoutMs],
361
+ ["firstContentTimeoutMs", config.firstContentTimeoutMs],
362
+ ["streamIdleTimeoutMs", config.streamIdleTimeoutMs],
363
+ ] as const) {
364
+ if (value !== undefined && (!Number.isInteger(value) || value < 0 || value > MAX_PROVIDER_TIMEOUT_MS)) {
365
+ throw new Error(`${config.source ?? "provider"}: ${name} must be an integer from 0 through ${MAX_PROVIDER_TIMEOUT_MS} milliseconds`);
366
+ }
367
+ }
296
368
  this.#fetchTimeoutMs = config.fetchTimeoutMs;
369
+ this.#operationTimeoutMs = config.operationTimeoutMs;
370
+ this.#firstContentTimeoutMs = config.firstContentTimeoutMs;
297
371
  this.#streamIdleTimeoutMs = config.streamIdleTimeoutMs;
298
372
  this.#headers = config.headers ?? {};
299
373
  this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
@@ -321,12 +395,36 @@ export default class AiSdkProvider implements Provider {
321
395
  }
322
396
  this.#countPromptTokens = config.countPromptTokens ?? ((messages) => estimatePromptTokens(messages));
323
397
  this.#promptTokensUrl = config.promptTokensUrl;
324
- this.#calculateCost = config.calculateCost ?? (() => 0);
325
- this.#calculateCharge = config.calculateCharge;
326
- this.#normalizeCharge = config.normalizeCharge;
398
+ this.#estimateCost = config.estimateCost
399
+ ?? (() => ({
400
+ kind: "unknown",
401
+ reason: "the request reported no direct cost and no model rate is configured",
402
+ }));
403
+ this.#normalizeCost = config.normalizeCost;
327
404
  this.#source = config.source ?? "provider";
328
405
  this.#grammarStyle = config.grammarStyle ?? "none";
329
- this.#promptCacheKey = config.promptCacheKey ?? false;
406
+ this.#cacheAffinity = config.cacheAffinity;
407
+ this.#systemCacheProviderOptions = config.systemCacheProviderOptions;
408
+ this.#reasoningResponseProviderOptions = config.reasoningResponseProviderOptions;
409
+ if (this.#languageModel === undefined && this.#cacheAffinity?.target === "provider-option") {
410
+ throw new Error(`${this.#source}: native provider-option cache affinity requires an AI SDK model`);
411
+ }
412
+ if (this.#languageModel !== undefined && this.#cacheAffinity?.target === "body") {
413
+ throw new Error(`${this.#source}: body cache affinity requires an OpenAI-compatible URL`);
414
+ }
415
+ if (this.#languageModel === undefined && this.#systemCacheProviderOptions !== undefined) {
416
+ throw new Error(`${this.#source}: system cache provider options require an AI SDK model`);
417
+ }
418
+ if (this.#languageModel === undefined && this.#reasoningResponseProviderOptions !== undefined) {
419
+ throw new Error(`${this.#source}: reasoning response provider options require an AI SDK model`);
420
+ }
421
+ if (this.#cacheAffinity?.target === "provider-option"
422
+ && Object.hasOwn(
423
+ this.#reasoningResponseProviderOptions?.[this.#cacheAffinity.provider] ?? {},
424
+ this.#cacheAffinity.name,
425
+ )) {
426
+ throw new Error(`${this.#source}: reasoning response options conflict with cache-affinity option ${this.#cacheAffinity.provider}.${this.#cacheAffinity.name}`);
427
+ }
330
428
  this.#serviceTier = config.serviceTier;
331
429
  this.#gbnfDebug = config.gbnfDebug ?? false;
332
430
  this.#streaming = config.streaming ?? true;
@@ -346,10 +444,17 @@ export default class AiSdkProvider implements Provider {
346
444
  const reasoningReserve = this.reasoningReserve;
347
445
  if (this.#reasoningStyle === "template"
348
446
  && this.#reasoning.mode === "on"
447
+ && this.#reasoning.budget !== null
349
448
  && reasoningReserve !== null
350
- && this.#reasoning.budget! > reasoningReserve) {
449
+ && this.#reasoning.budget > reasoningReserve) {
351
450
  throw new Error(`${this.#source}: PLURNK_PROVIDERS_REASONING_BUDGET (${this.#reasoning.budget}) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE (${reasoningReserve})`);
352
451
  }
452
+ if (this.#reasoningStyle === "anthropic"
453
+ && this.#reasoning.mode === "on"
454
+ && this.#reasoning.budget === null
455
+ && reasoningReserve === null) {
456
+ throw new Error(`${this.#source}: explicit Anthropic reasoning requires a resolved reasoning reserve or PLURNK_PROVIDERS_REASONING_BUDGET`);
457
+ }
353
458
  const { tokenizeUrl } = config;
354
459
  if (tokenizeUrl !== undefined) {
355
460
  this.tokenize = async (text: string): Promise<number[]> => {
@@ -357,7 +462,9 @@ export default class AiSdkProvider implements Provider {
357
462
  method: "POST",
358
463
  headers: { "Content-Type": "application/json", ...this.#headers },
359
464
  body: JSON.stringify({ content: text }),
360
- signal: AbortSignal.timeout(this.#fetchTimeoutMs),
465
+ ...(this.#fetchTimeoutMs > 0
466
+ ? { signal: AbortSignal.timeout(this.#fetchTimeoutMs) }
467
+ : {}),
361
468
  });
362
469
  if (!res.ok) throw new Error(`${this.#source}: tokenize endpoint returned ${res.status}`);
363
470
  const { tokens } = (await res.json()) as { tokens?: unknown };
@@ -402,7 +509,14 @@ export default class AiSdkProvider implements Provider {
402
509
 
403
510
  signal?.throwIfAborted();
404
511
  try {
405
- const timeout = AbortSignal.timeout(this.#fetchTimeoutMs);
512
+ const timeout = this.#fetchTimeoutMs > 0
513
+ ? AbortSignal.timeout(this.#fetchTimeoutMs)
514
+ : undefined;
515
+ const requestSignal = signal === undefined
516
+ ? timeout
517
+ : timeout === undefined
518
+ ? signal
519
+ : AbortSignal.any([signal, timeout]);
406
520
  const response = await this.#fetch(this.#promptTokensUrl, {
407
521
  method: "POST",
408
522
  headers: { "Content-Type": "application/json", ...this.#headers },
@@ -411,7 +525,7 @@ export default class AiSdkProvider implements Provider {
411
525
  messages,
412
526
  ...this.#reasoningBody(),
413
527
  }),
414
- signal: signal === undefined ? timeout : AbortSignal.any([signal, timeout]),
528
+ ...(requestSignal === undefined ? {} : { signal: requestSignal }),
415
529
  });
416
530
  if (!response.ok) {
417
531
  return estimatePromptTokens(
@@ -439,12 +553,6 @@ export default class AiSdkProvider implements Provider {
439
553
  );
440
554
  }
441
555
  }
442
- calculateCost(usage: ProviderUsage): number { return this.#calculateCost(usage); }
443
- calculateCharge(usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }> {
444
- return this.#calculateCharge?.(usage)
445
- ?? { kind: "unknown", reason: "no provider rate or settled charge is available" };
446
- }
447
-
448
556
  // Reasoning activation and allowance are independent of grammar transport;
449
557
  // only the response representation becomes lossless when evidence is needed.
450
558
  // The llama-server template mapping is owned by {§llama-reasoning-request}.
@@ -455,7 +563,7 @@ export default class AiSdkProvider implements Provider {
455
563
  case "template": {
456
564
  const allowance = mode === "off"
457
565
  ? 0
458
- : mode === "on" ? budget : this.reasoningReserve;
566
+ : mode === "on" && budget !== null ? budget : this.reasoningReserve;
459
567
  return {
460
568
  chat_template_kwargs: { enable_thinking: on },
461
569
  reasoning_format: preserveGrammarSentence ? "none" : "auto",
@@ -464,9 +572,9 @@ export default class AiSdkProvider implements Provider {
464
572
  }
465
573
  case "think": return on ? { think: true } : {};
466
574
  case "include_reasoning": return on ? { include_reasoning: true } : {};
467
- // effort tiers from the budget; off/adaptive omit the field (the
468
- // API's default depth is its adaptive).
469
- case "effort": return mode === "on" ? { reasoning_effort: effortFromBudget(budget!) } : {};
575
+ // Explicit on uses the portable enabled posture or a tier derived
576
+ // from an explicit budget; off/adaptive omit the field.
577
+ case "effort": return mode === "on" ? { reasoning_effort: effortFromReasoning(this.#reasoning) } : {};
470
578
  // Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
471
579
  // reason-by-default model (DeepSeek V4: default 'high') reasoning.
472
580
  // ADAPTIVE omits the field: the backend's own default posture IS the
@@ -476,19 +584,25 @@ export default class AiSdkProvider implements Provider {
476
584
  // efforts 400.
477
585
  case "effort_explicit": return mode === "off"
478
586
  ? { reasoning_effort: "none" }
479
- : mode === "on" ? { reasoning_effort: effortFromBudget(budget!) } : {};
587
+ : mode === "on" ? { reasoning_effort: effortFromReasoning(this.#reasoning) } : {};
480
588
  // {§deepseek-reasoning-request}
481
589
  case "thinking_effort": return mode === "off"
482
590
  ? { thinking: { type: "disabled" } }
483
591
  : mode === "on" ? {
484
592
  thinking: { type: "enabled" },
485
- reasoning_effort: effortFromBudget(budget!),
593
+ ...(budget === null ? {} : { reasoning_effort: effortFromBudget(budget) }),
486
594
  } : {};
487
595
  // Anthropic compat: explicit thinking object. off → disabled; on →
488
- // enabled with budget_tokens; adaptive → omit (the API default).
596
+ // enabled with the explicit budget or resolved reserve; adaptive →
597
+ // omit (the API default).
489
598
  case "anthropic": return mode === "off"
490
599
  ? { thinking: { type: "disabled" } }
491
- : mode === "on" ? { thinking: { type: "enabled", budget_tokens: budget } } : {};
600
+ : mode === "on" ? {
601
+ thinking: {
602
+ type: "enabled",
603
+ budget_tokens: budget ?? this.reasoningReserve!,
604
+ },
605
+ } : {};
492
606
  case "none": return {};
493
607
  }
494
608
  }
@@ -555,7 +669,7 @@ export default class AiSdkProvider implements Provider {
555
669
  }
556
670
  }
557
671
 
558
- // First-party telemetry headers ({§provider-request-authority}): forwarded only when the spec
672
+ // First-party telemetry headers ({§provider-request-authority} {§provider-call-kind}): forwarded only when the spec
559
673
  // opted in (the plurnk endpoint). The gate is here, not at the call site, so
560
674
  // attributions/client/strikes can never reach a third-party backend even if
561
675
  // the consumer passes them to the wrong provider. Empty values emit no header
@@ -563,7 +677,7 @@ export default class AiSdkProvider implements Provider {
563
677
  // absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
564
678
  // ride HTTP headers only — the packet never carries them (the model must
565
679
  // never see strike state; engine accounting is not a metric to game).
566
- #metadataHeaders(attributions: string[] | undefined, client: string | undefined, strikes: number | undefined, workerId: string, primaryWorkerId: string | undefined, workspaceId: string | undefined, loop: number | undefined, turn: number | undefined): Record<string, string> {
680
+ #metadataHeaders(attributions: string[] | undefined, client: string | undefined, strikes: number | undefined, workerId: string, primaryWorkerId: string | undefined, workspaceId: string | undefined, loop: number | undefined, turn: number | undefined, callKind: ProviderCallKind | undefined): Record<string, string> {
567
681
  if (!this.#firstPartyMetadata) return {};
568
682
  const h: Record<string, string> = {};
569
683
  if (attributions !== undefined && attributions.length > 0) h["Plurnk-Attribution"] = JSON.stringify(attributions);
@@ -588,6 +702,7 @@ export default class AiSdkProvider implements Provider {
588
702
  if (workspaceId !== undefined && workspaceId.length > 0) h["Plurnk-Workspace-Id"] = workspaceId;
589
703
  if (loop !== undefined && Number.isInteger(loop) && loop >= 1) h["Plurnk-Loop"] = String(loop);
590
704
  if (turn !== undefined && Number.isInteger(turn) && turn >= 1) h["Plurnk-Turn"] = String(turn);
705
+ if (callKind !== undefined) h["Plurnk-Call-Kind"] = callKind;
591
706
  return h;
592
707
  }
593
708
 
@@ -627,9 +742,45 @@ export default class AiSdkProvider implements Provider {
627
742
  return out;
628
743
  }
629
744
 
630
- async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }: { messages: ChatMessage[]; workerId: string; primaryWorkerId?: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse> {
745
+ #requestProviderOptions(workerId: string): AiSdkProviderOptions | undefined {
746
+ const reasoningOptions = this.#reasoning.mode === "off"
747
+ ? undefined
748
+ : this.#reasoningResponseProviderOptions;
749
+ if (this.#cacheAffinity?.target !== "provider-option") return reasoningOptions;
750
+ const { provider, name } = this.#cacheAffinity;
751
+ return {
752
+ ...reasoningOptions,
753
+ [provider]: {
754
+ ...reasoningOptions?.[provider],
755
+ [name]: workerId,
756
+ },
757
+ };
758
+ }
759
+
760
+ #accounting(
761
+ outcome: ProviderRequestAccounting["outcome"],
762
+ usage: ProviderUsage | undefined,
763
+ evidence: Parameters<ProviderCostNormalizer>[0],
764
+ status?: number,
765
+ ): ProviderRequestAccounting {
766
+ const knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
767
+ const direct = this.#normalizeCost?.(evidence);
768
+ return validateProviderRequestAccounting({
769
+ provider: this.#source,
770
+ model: this.#model,
771
+ outcome,
772
+ ...(status === undefined ? {} : { status }),
773
+ ...(knownUsage === undefined ? {} : { usage: knownUsage }),
774
+ cost: resolveProviderCost(direct, this.#estimateCost(knownUsage)),
775
+ });
776
+ }
777
+
778
+ async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling, observeRequest, callKind }: ProviderGenerateArgs): Promise<ProviderResponse> {
631
779
  // {§provider-interface} The worker identity is required.
632
780
  if (workerId === undefined || workerId.length === 0) throw new Error("generate: workerId is required — the worker's stable, opaque identity");
781
+ if (callKind !== undefined && callKind !== "emission" && callKind !== "bare") {
782
+ throw new Error(`generate: unsupported callKind ${JSON.stringify(callKind)}`);
783
+ }
633
784
  // Reject before any wire call when already aborted
634
785
  // ({§provider-failure-normalization}).
635
786
  signal?.throwIfAborted();
@@ -661,72 +812,181 @@ export default class AiSdkProvider implements Provider {
661
812
  // reserved from caller sampling; the env flag is the single control).
662
813
  ...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
663
814
  ...this.#slotBody(workerId),
664
- // Prompt-cache affinity -- workerId as the OpenAI-standard
665
- // prompt_cache_key routes a worker's turns to one serverless replica so
666
- // its stable prefix caches (managed; reserved from caller sampling).
667
- ...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
815
+ ...(this.#cacheAffinity?.target === "body"
816
+ ? { [this.#cacheAffinity.name]: workerId }
817
+ : {}),
668
818
  };
669
819
 
670
820
  // Per-request headers = static auth/routing + any first-party telemetry.
671
- const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
672
- const headers = Object.keys(metaHeaders).length === 0
673
- ? this.#headers
674
- : { ...this.#headers, ...metaHeaders };
675
- let raw;
676
- try {
677
- raw = this.#languageModel === undefined
678
- ? await executeOpenAICompatible({
679
- url: this.#url!,
821
+ const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind);
822
+ const headers = new Headers(this.#headers);
823
+ if (this.#cacheAffinity?.target === "header") {
824
+ headers.set(this.#cacheAffinity.name, workerId);
825
+ }
826
+ for (const [name, value] of Object.entries(metaHeaders)) headers.set(name, value);
827
+ const requestHeaders = Object.fromEntries(headers.entries());
828
+ const accounting: ProviderRequestAccounting[] = [];
829
+ const operationTimeout = this.#operationTimeoutMs > 0
830
+ ? AbortSignal.timeout(this.#operationTimeoutMs)
831
+ : undefined;
832
+ const operationSignal = signal === undefined
833
+ ? operationTimeout
834
+ : operationTimeout === undefined
835
+ ? signal
836
+ : AbortSignal.any([signal, operationTimeout]);
837
+ const executeRequest = async () => {
838
+ let settle: ProviderRequestSettlement | undefined;
839
+ try {
840
+ settle = await observeRequest?.({
841
+ provider: this.#source,
680
842
  model: this.#model,
681
- headers,
682
- body,
683
- messages,
684
- signal,
685
- fetch: this.#fetch,
686
- fetchTimeoutMs: this.#fetchTimeoutMs,
687
- streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
688
- retryAttempts: this.#retryAttempts,
689
- streaming: this.#streaming,
690
- captureRawBody: this.#rawBody,
691
- })
692
- : await executeAiSdkModel({
693
- languageModel: this.#languageModel,
694
- headers,
695
- messages,
696
- signal,
697
- fetchTimeoutMs: this.#fetchTimeoutMs,
698
- streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
699
- retryAttempts: this.#retryAttempts,
700
- streaming: this.#streaming,
701
- captureRawBody: this.#rawBody,
702
- temperature: this.#tuningFloors
703
- ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
704
- : typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
705
- topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
706
- topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
707
- presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
708
- frequencyPenalty: typeof sampling?.frequency_penalty === "number"
709
- ? sampling.frequency_penalty
710
- : this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
711
- stopSequences: typeof sampling?.stop === "string"
712
- ? [sampling.stop]
713
- : Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
714
- ? sampling.stop
715
- : undefined,
716
- seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
717
- maxOutputTokens: maxTokens,
718
- reasoning: this.#reasoning.mode === "off"
719
- ? "none"
720
- : this.#reasoning.mode === "adaptive"
721
- ? "provider-default"
722
- : effortFromBudget(this.#reasoning.budget!),
723
843
  });
844
+ } catch (cause) {
845
+ throw new ProviderRequestObserverError(cause);
846
+ }
847
+ const settleAccounting = async (
848
+ outcome: ProviderRequestAccounting["outcome"],
849
+ usage: ProviderUsage | undefined,
850
+ evidence: Parameters<ProviderCostNormalizer>[0],
851
+ status?: number,
852
+ ): Promise<ProviderRequestAccounting> => {
853
+ let requestAccounting: ProviderRequestAccounting;
854
+ let normalizationFailure: { cause: unknown } | undefined;
855
+ try {
856
+ requestAccounting = this.#accounting(outcome, usage, evidence, status);
857
+ } catch (cause) {
858
+ normalizationFailure = { cause };
859
+ let knownUsage: ProviderUsage | undefined;
860
+ try {
861
+ knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
862
+ } catch {
863
+ knownUsage = undefined;
864
+ }
865
+ requestAccounting = validateProviderRequestAccounting({
866
+ provider: this.#source,
867
+ model: this.#model,
868
+ outcome,
869
+ ...(status === undefined ? {} : { status }),
870
+ ...(knownUsage === undefined ? {} : { usage: knownUsage }),
871
+ cost: {
872
+ kind: "unknown",
873
+ reason: "provider request accounting could not be normalized after physical I/O",
874
+ },
875
+ });
876
+ }
877
+ accounting.push(requestAccounting);
878
+ try {
879
+ await settle?.(requestAccounting);
880
+ } catch (cause) {
881
+ throw new ProviderRequestObserverError(cause);
882
+ }
883
+ if (normalizationFailure !== undefined) {
884
+ throw new ProviderRequestAccountingError(normalizationFailure.cause);
885
+ }
886
+ return requestAccounting;
887
+ };
888
+ let response;
889
+ try {
890
+ response = this.#languageModel === undefined
891
+ ? await executeOpenAICompatible({
892
+ url: this.#url!,
893
+ model: this.#model,
894
+ headers: requestHeaders,
895
+ body,
896
+ messages,
897
+ signal: operationSignal,
898
+ fetch: this.#fetch,
899
+ fetchTimeoutMs: this.#fetchTimeoutMs,
900
+ firstContentTimeoutMs: this.#firstContentTimeoutMs,
901
+ streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
902
+ streaming: this.#streaming,
903
+ captureRawBody: this.#rawBody,
904
+ })
905
+ : await executeAiSdkModel({
906
+ languageModel: this.#languageModel,
907
+ headers: requestHeaders,
908
+ providerOptions: this.#requestProviderOptions(workerId),
909
+ systemProviderOptions: this.#systemCacheProviderOptions,
910
+ messages,
911
+ signal: operationSignal,
912
+ fetchTimeoutMs: this.#fetchTimeoutMs,
913
+ firstContentTimeoutMs: this.#firstContentTimeoutMs,
914
+ streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
915
+ streaming: this.#streaming,
916
+ captureRawBody: this.#rawBody,
917
+ temperature: this.#tuningFloors
918
+ ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
919
+ : typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
920
+ topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
921
+ topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
922
+ presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
923
+ frequencyPenalty: typeof sampling?.frequency_penalty === "number"
924
+ ? sampling.frequency_penalty
925
+ : this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
926
+ stopSequences: typeof sampling?.stop === "string"
927
+ ? [sampling.stop]
928
+ : Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
929
+ ? sampling.stop
930
+ : undefined,
931
+ seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
932
+ maxOutputTokens: maxTokens,
933
+ reasoning: this.#reasoning.mode === "off"
934
+ ? "none"
935
+ : this.#reasoning.mode === "adaptive"
936
+ ? "provider-default"
937
+ : effortFromReasoning(this.#reasoning),
938
+ });
939
+ } catch (error) {
940
+ const failure = transportFailureEvidence(error);
941
+ await settleAccounting(
942
+ "error",
943
+ failure.usage,
944
+ failure.chargeEvidence,
945
+ failure.status,
946
+ );
947
+ throw error;
948
+ }
949
+ await settleAccounting(
950
+ "response",
951
+ response.usage,
952
+ response.chargeEvidence,
953
+ );
954
+ return response;
955
+ };
956
+
957
+ let raw;
958
+ try {
959
+ const { retry } = prepareRetries({
960
+ maxRetries: this.#retryAttempts,
961
+ abortSignal: operationSignal,
962
+ });
963
+ raw = await retry(executeRequest);
724
964
  } catch (err) {
965
+ if (err instanceof ProviderRequestObserverError
966
+ || err instanceof ProviderRequestAccountingError) throw err.cause;
725
967
  if (signal?.aborted) throw err;
968
+ if (operationTimeout?.aborted) {
969
+ const timeout = new ProviderTimeoutError("operation", this.#operationTimeoutMs, err);
970
+ throw new ProviderError(this.#source, "deadline_exceeded", timeout.message, {
971
+ status: 504,
972
+ cause: timeout,
973
+ retryable: false,
974
+ extensions: {
975
+ timeoutPhase: timeout.phase,
976
+ timeoutMs: timeout.timeoutMs,
977
+ },
978
+ accounting,
979
+ });
980
+ }
726
981
  const pe = toProviderError(err, this.#source, this.#errorDetailLimit);
727
982
  if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
728
- throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
983
+ throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, {
984
+ status: pe.status,
985
+ cause: err,
986
+ accounting,
987
+ });
729
988
  }
989
+ pe.prependAccounting(accounting);
730
990
  throw pe;
731
991
  }
732
992
 
@@ -747,10 +1007,10 @@ export default class AiSdkProvider implements Provider {
747
1007
  this.#reasoningResponseStyle,
748
1008
  );
749
1009
 
750
- // Preserve the exact sentence seen at the grammar boundary. Constrained
751
- // template turns request `reasoning_format: "none"`, so even an empty
752
- // channel remains observable. An unexpectedly projected response cannot
753
- // supply independent pre-projection evidence.
1010
+ // Preserve the exact pre-projection response. Constrained template turns
1011
+ // request `reasoning_format: "none"`, so even an empty channel and any
1012
+ // template-provided opener remain observable. An unexpectedly projected
1013
+ // response cannot supply independent evidence.
754
1014
  let grammarEvidence: GrammarEvidence | undefined;
755
1015
  if (wantGrammar) {
756
1016
  if (preserveGrammarSentence) {
@@ -779,12 +1039,13 @@ export default class AiSdkProvider implements Provider {
779
1039
  if (projectedReasoning.projected) {
780
1040
  raw.content = projectedReasoning.content;
781
1041
  raw.reasoning = projectedReasoning.reasoning;
782
- raw.usage = attributeUnitemizedReasoning(raw.usage, raw.reasoning, raw.content);
783
1042
  }
784
1043
 
785
1044
  let notices: ProviderNotice[] | undefined;
786
1045
  const usage = raw.usage;
787
- if (sendGrammar !== undefined && this.tokenize !== undefined) {
1046
+ if (sendGrammar !== undefined
1047
+ && this.tokenize !== undefined
1048
+ && usage?.outputTokens !== undefined) {
788
1049
  // Channel-escape detector: completion tokens
789
1050
  // billed far beyond every visible channel mean the decode ESCAPED into
790
1051
  // a server-discarded reasoning block mid-emission. This diagnostic
@@ -795,12 +1056,12 @@ export default class AiSdkProvider implements Provider {
795
1056
  this.tokenize(raw.reasoning),
796
1057
  ]);
797
1058
  const visible = contentTokens.length + reasoningTokens.length;
798
- if (usage.completion > visible + 64) {
1059
+ if (usage.outputTokens > visible + 64) {
799
1060
  (notices ??= []).push({
800
1061
  source: this.#source,
801
1062
  kind: "grammar_unenforced",
802
1063
  level: "warn",
803
- message: `decode escaped the grammar: ${usage.completion} completion tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
1064
+ message: `decode escaped the grammar: ${usage.outputTokens} output tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
804
1065
  position: [...raw.content].length,
805
1066
  });
806
1067
  }
@@ -824,17 +1085,12 @@ export default class AiSdkProvider implements Provider {
824
1085
  ...(raw.reasoningEncrypted.length > 0
825
1086
  ? { reasoningEncrypted: raw.reasoningEncrypted }
826
1087
  : {}),
827
- usage,
828
1088
  model: raw.model,
829
1089
  ...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
830
1090
  };
831
- const normalizedCharge = this.#normalizeCharge?.(raw.chargeEvidence);
832
- const charge = normalizedCharge === undefined
833
- ? undefined
834
- : validateAuthoritativeCharge(normalizedCharge);
835
1091
  const evidence = {
836
1092
  assistantRaw: raw,
837
- ...(charge === undefined ? {} : { charge }),
1093
+ accounting,
838
1094
  ...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
839
1095
  ...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
840
1096
  ...(meta !== undefined ? { meta } : {}),
@@ -851,6 +1107,7 @@ export default class AiSdkProvider implements Provider {
851
1107
  "The provider interrupted generation because inference resources were unavailable.",
852
1108
  {
853
1109
  attempt,
1110
+ accounting,
854
1111
  extensions: {
855
1112
  stage: "provider-response",
856
1113
  finishReason: "resource_interrupted",