@plurnk/plurnk-providers 1.4.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. package/.env.defaults +40 -34
  2. package/README.md +3 -0
  3. package/SPEC.md +153 -62
  4. package/dist/AiSdkProvider.d.ts +19 -25
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +353 -120
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +7 -13
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +36 -8
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +2 -21
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +19 -14
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/accounting.d.ts +6 -0
  17. package/dist/accounting.d.ts.map +1 -0
  18. package/dist/accounting.js +168 -0
  19. package/dist/accounting.js.map +1 -0
  20. package/dist/aiSdkTransport.d.ts +11 -3
  21. package/dist/aiSdkTransport.d.ts.map +1 -1
  22. package/dist/aiSdkTransport.js +198 -29
  23. package/dist/aiSdkTransport.js.map +1 -1
  24. package/dist/catalogProvider.d.ts +7 -2
  25. package/dist/catalogProvider.d.ts.map +1 -1
  26. package/dist/catalogProvider.js +32 -26
  27. package/dist/catalogProvider.js.map +1 -1
  28. package/dist/compatibleProvider.d.ts.map +1 -1
  29. package/dist/compatibleProvider.js +18 -7
  30. package/dist/compatibleProvider.js.map +1 -1
  31. package/dist/cost.d.ts +10 -10
  32. package/dist/cost.d.ts.map +1 -1
  33. package/dist/cost.js +88 -43
  34. package/dist/cost.js.map +1 -1
  35. package/dist/env.d.ts +5 -7
  36. package/dist/env.d.ts.map +1 -1
  37. package/dist/env.js +30 -32
  38. package/dist/env.js.map +1 -1
  39. package/dist/errors.d.ts +14 -2
  40. package/dist/errors.d.ts.map +1 -1
  41. package/dist/errors.js +60 -2
  42. package/dist/errors.js.map +1 -1
  43. package/dist/index.d.ts +4 -4
  44. package/dist/index.d.ts.map +1 -1
  45. package/dist/index.js +3 -2
  46. package/dist/index.js.map +1 -1
  47. package/dist/ollama.js +3 -3
  48. package/dist/ollama.js.map +1 -1
  49. package/dist/sdkModels.d.ts +6 -0
  50. package/dist/sdkModels.d.ts.map +1 -1
  51. package/dist/sdkModels.js +46 -3
  52. package/dist/sdkModels.js.map +1 -1
  53. package/dist/types.d.ts +40 -29
  54. package/dist/types.d.ts.map +1 -1
  55. package/dist/usage.d.ts +21 -4
  56. package/dist/usage.d.ts.map +1 -1
  57. package/dist/usage.js +188 -74
  58. package/dist/usage.js.map +1 -1
  59. package/package.json +9 -7
  60. package/src/AiSdkProvider.test.ts +1039 -182
  61. package/src/AiSdkProvider.ts +428 -141
  62. package/src/Mock.test.ts +37 -12
  63. package/src/Mock.ts +46 -12
  64. package/src/Pool.test.ts +19 -6
  65. package/src/Pool.ts +20 -16
  66. package/src/ProviderRegistry.test.ts +16 -11
  67. package/src/accounting.test.ts +94 -0
  68. package/src/accounting.ts +190 -0
  69. package/src/aiSdkTransport.test.ts +42 -49
  70. package/src/aiSdkTransport.ts +218 -32
  71. package/src/boundaries.test.ts +2 -0
  72. package/src/catalogProvider.test.ts +271 -24
  73. package/src/catalogProvider.ts +44 -28
  74. package/src/compatibleProvider.test.ts +6 -3
  75. package/src/compatibleProvider.ts +20 -7
  76. package/src/cost.test.ts +55 -35
  77. package/src/cost.ts +110 -54
  78. package/src/defaults.test.ts +13 -3
  79. package/src/env.test.ts +50 -26
  80. package/src/env.ts +43 -42
  81. package/src/errors.test.ts +47 -2
  82. package/src/errors.ts +68 -3
  83. package/src/index.ts +21 -5
  84. package/src/ollama.test.ts +4 -1
  85. package/src/ollama.ts +3 -3
  86. package/src/sdkModels.test.ts +94 -3
  87. package/src/sdkModels.ts +53 -3
  88. package/src/types.ts +91 -33
  89. package/src/usage.test.ts +112 -108
  90. package/src/usage.ts +233 -84
@@ -5,12 +5,28 @@
5
5
  // Composition, not inheritance: an official AI SDK language model supplies the
6
6
  // ordinary vendor protocol. The compatible URL path remains only for PLURNK
7
7
  // extensions and local endpoint probes the SDK cannot represent.
8
- import { executeAiSdkModel, executeOpenAICompatible } from "./aiSdkTransport.js";
9
- import { toProviderError, ProviderError } from "./errors.js";
10
- import { attributeUnitemizedReasoning } from "./usage.js";
8
+ import { MAX_PROVIDER_TIMEOUT_MS } from "./env.js";
9
+ import { executeAiSdkModel, executeOpenAICompatible, transportFailureEvidence, } from "./aiSdkTransport.js";
10
+ import { prepareRetries } from "ai/internal";
11
+ import { toProviderError, ProviderError, ProviderTimeoutError } from "./errors.js";
11
12
  import { validateGbnf } from "@plurnk/gbnf";
12
13
  import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.js";
13
14
  import { emitWarningOnce } from "./warnings.js";
15
+ import { resolveProviderCost } from "./cost.js";
16
+ import { validateProviderRequestAccounting } from "./accounting.js";
17
+ import { validateProviderUsage } from "./usage.js";
18
+ class ProviderRequestObserverError extends Error {
19
+ constructor(cause) {
20
+ super("provider request accounting could not be durably settled", { cause });
21
+ this.name = "ProviderRequestObserverError";
22
+ }
23
+ }
24
+ class ProviderRequestAccountingError extends Error {
25
+ constructor(cause) {
26
+ super("provider request accounting could not be normalized", { cause });
27
+ this.name = "ProviderRequestAccountingError";
28
+ }
29
+ }
14
30
  // Drop trailing occurrences of a server-rendered EOG marker. llama-server
15
31
  // under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
16
32
  // trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
@@ -24,15 +40,10 @@ const stripTrailingSpecial = (content, marker) => {
24
40
  out = out.slice(0, -marker.length);
25
41
  return out;
26
42
  };
27
- // {§provider-tagged-reasoning} Only the model-contract position is structural:
28
- // one exact leading envelope. Parsing after stream assembly keeps SSE and JSON
29
- // on one path and leaves later literal tags in the visible suffix untouched.
30
- const projectTaggedReasoning = (content, structuredReasoning, style) => {
31
- const opening = "<think>";
32
- if (style !== "think-tags" || structuredReasoning.length > 0 || !content.startsWith(opening)) {
43
+ const projectLeadingReasoning = (content, structuredReasoning, opening, closing) => {
44
+ if (structuredReasoning.length > 0 || !content.startsWith(opening)) {
33
45
  return { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
34
46
  }
35
- const closing = "</think>";
36
47
  const closingIndex = content.indexOf(closing, opening.length);
37
48
  if (closingIndex === -1) {
38
49
  return {
@@ -50,6 +61,26 @@ const projectTaggedReasoning = (content, structuredReasoning, style) => {
50
61
  contentStart: [...content.slice(0, suffixStart)].length,
51
62
  };
52
63
  };
64
+ // {§provider-tagged-reasoning} Only the model-contract position is structural:
65
+ // one exact leading envelope. Parsing after stream assembly keeps SSE and JSON
66
+ // on one path and leaves later literal tags in the visible suffix untouched.
67
+ const projectTaggedReasoning = (content, structuredReasoning, style) => style === "think-tags"
68
+ ? projectLeadingReasoning(content, structuredReasoning, "<think>", "</think>")
69
+ : { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
70
+ // llama-server's template reasoning parser can project either supported leading
71
+ // reasoning envelope out of the OpenAI-compatible response. Grammar evidence
72
+ // needs the sentence before that lossy projection, so constrained template turns
73
+ // request it verbatim and split the observed enclosure here.
74
+ const projectTemplateReasoning = (content) => {
75
+ for (const [opening, closing] of [
76
+ ["<|channel>thought\n", "<channel|>"],
77
+ ["<think>\n", "</think>"],
78
+ ]) {
79
+ if (content.startsWith(opening))
80
+ return projectLeadingReasoning(content, "", opening, closing);
81
+ }
82
+ return { content, reasoning: "", projected: false, contentStart: 0 };
83
+ };
53
84
  // Shared budget→effort breakpoints (xai and google had identical copies).
54
85
  export const effortFromBudget = (budget) => {
55
86
  if (budget <= 1000)
@@ -58,6 +89,11 @@ export const effortFromBudget = (budget) => {
58
89
  return "medium";
59
90
  return "high";
60
91
  };
92
+ // AI SDK's portable reasoning control has no boolean-enabled value. `medium`
93
+ // is the neutral activation projection for an explicit, unqualified `on`; it
94
+ // changes no PLURNK token reserve. An operator budget, when present, remains
95
+ // the only input to the existing magnitude-to-tier projection.
96
+ const effortFromReasoning = (reasoning) => reasoning.budget === null ? "medium" : effortFromBudget(reasoning.budget);
61
97
  // Body keys the provider owns — a caller's `sampling` passthrough may not set
62
98
  // these. Two families:
63
99
  // transport/managed — grammar transport, the stream/JSON choice, slot pinning,
@@ -84,6 +120,8 @@ export default class AiSdkProvider {
84
120
  #url;
85
121
  #languageModel;
86
122
  #fetchTimeoutMs;
123
+ #operationTimeoutMs;
124
+ #firstContentTimeoutMs;
87
125
  #streamIdleTimeoutMs;
88
126
  #headers;
89
127
  #fetch;
@@ -103,11 +141,13 @@ export default class AiSdkProvider {
103
141
  #reasoningResponseStyle;
104
142
  #countPromptTokens;
105
143
  #promptTokensUrl;
106
- #calculateCost;
107
- #calculateCharge;
144
+ #estimateCost;
145
+ #normalizeCost;
108
146
  #source;
109
147
  #grammarStyle;
110
- #promptCacheKey;
148
+ #cacheAffinity;
149
+ #systemCacheProviderOptions;
150
+ #reasoningResponseProviderOptions;
111
151
  #serviceTier;
112
152
  #gbnfDebug;
113
153
  #streaming;
@@ -137,7 +177,19 @@ export default class AiSdkProvider {
137
177
  if ((this.#url === undefined) === (this.#languageModel === undefined)) {
138
178
  throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
139
179
  }
180
+ for (const [name, value] of [
181
+ ["fetchTimeoutMs", config.fetchTimeoutMs],
182
+ ["operationTimeoutMs", config.operationTimeoutMs],
183
+ ["firstContentTimeoutMs", config.firstContentTimeoutMs],
184
+ ["streamIdleTimeoutMs", config.streamIdleTimeoutMs],
185
+ ]) {
186
+ if (value !== undefined && (!Number.isInteger(value) || value < 0 || value > MAX_PROVIDER_TIMEOUT_MS)) {
187
+ throw new Error(`${config.source ?? "provider"}: ${name} must be an integer from 0 through ${MAX_PROVIDER_TIMEOUT_MS} milliseconds`);
188
+ }
189
+ }
140
190
  this.#fetchTimeoutMs = config.fetchTimeoutMs;
191
+ this.#operationTimeoutMs = config.operationTimeoutMs;
192
+ this.#firstContentTimeoutMs = config.firstContentTimeoutMs;
141
193
  this.#streamIdleTimeoutMs = config.streamIdleTimeoutMs;
142
194
  this.#headers = config.headers ?? {};
143
195
  this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
@@ -165,11 +217,33 @@ export default class AiSdkProvider {
165
217
  }
166
218
  this.#countPromptTokens = config.countPromptTokens ?? ((messages) => estimatePromptTokens(messages));
167
219
  this.#promptTokensUrl = config.promptTokensUrl;
168
- this.#calculateCost = config.calculateCost ?? (() => 0);
169
- this.#calculateCharge = config.calculateCharge;
220
+ this.#estimateCost = config.estimateCost
221
+ ?? (() => ({
222
+ kind: "unknown",
223
+ reason: "the request reported no direct cost and no model rate is configured",
224
+ }));
225
+ this.#normalizeCost = config.normalizeCost;
170
226
  this.#source = config.source ?? "provider";
171
227
  this.#grammarStyle = config.grammarStyle ?? "none";
172
- this.#promptCacheKey = config.promptCacheKey ?? false;
228
+ this.#cacheAffinity = config.cacheAffinity;
229
+ this.#systemCacheProviderOptions = config.systemCacheProviderOptions;
230
+ this.#reasoningResponseProviderOptions = config.reasoningResponseProviderOptions;
231
+ if (this.#languageModel === undefined && this.#cacheAffinity?.target === "provider-option") {
232
+ throw new Error(`${this.#source}: native provider-option cache affinity requires an AI SDK model`);
233
+ }
234
+ if (this.#languageModel !== undefined && this.#cacheAffinity?.target === "body") {
235
+ throw new Error(`${this.#source}: body cache affinity requires an OpenAI-compatible URL`);
236
+ }
237
+ if (this.#languageModel === undefined && this.#systemCacheProviderOptions !== undefined) {
238
+ throw new Error(`${this.#source}: system cache provider options require an AI SDK model`);
239
+ }
240
+ if (this.#languageModel === undefined && this.#reasoningResponseProviderOptions !== undefined) {
241
+ throw new Error(`${this.#source}: reasoning response provider options require an AI SDK model`);
242
+ }
243
+ if (this.#cacheAffinity?.target === "provider-option"
244
+ && Object.hasOwn(this.#reasoningResponseProviderOptions?.[this.#cacheAffinity.provider] ?? {}, this.#cacheAffinity.name)) {
245
+ throw new Error(`${this.#source}: reasoning response options conflict with cache-affinity option ${this.#cacheAffinity.provider}.${this.#cacheAffinity.name}`);
246
+ }
173
247
  this.#serviceTier = config.serviceTier;
174
248
  this.#gbnfDebug = config.gbnfDebug ?? false;
175
249
  this.#streaming = config.streaming ?? true;
@@ -189,10 +263,17 @@ export default class AiSdkProvider {
189
263
  const reasoningReserve = this.reasoningReserve;
190
264
  if (this.#reasoningStyle === "template"
191
265
  && this.#reasoning.mode === "on"
266
+ && this.#reasoning.budget !== null
192
267
  && reasoningReserve !== null
193
268
  && this.#reasoning.budget > reasoningReserve) {
194
269
  throw new Error(`${this.#source}: PLURNK_PROVIDERS_REASONING_BUDGET (${this.#reasoning.budget}) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE (${reasoningReserve})`);
195
270
  }
271
+ if (this.#reasoningStyle === "anthropic"
272
+ && this.#reasoning.mode === "on"
273
+ && this.#reasoning.budget === null
274
+ && reasoningReserve === null) {
275
+ throw new Error(`${this.#source}: explicit Anthropic reasoning requires a resolved reasoning reserve or PLURNK_PROVIDERS_REASONING_BUDGET`);
276
+ }
196
277
  const { tokenizeUrl } = config;
197
278
  if (tokenizeUrl !== undefined) {
198
279
  this.tokenize = async (text) => {
@@ -200,7 +281,9 @@ export default class AiSdkProvider {
200
281
  method: "POST",
201
282
  headers: { "Content-Type": "application/json", ...this.#headers },
202
283
  body: JSON.stringify({ content: text }),
203
- signal: AbortSignal.timeout(this.#fetchTimeoutMs),
284
+ ...(this.#fetchTimeoutMs > 0
285
+ ? { signal: AbortSignal.timeout(this.#fetchTimeoutMs) }
286
+ : {}),
204
287
  });
205
288
  if (!res.ok)
206
289
  throw new Error(`${this.#source}: tokenize endpoint returned ${res.status}`);
@@ -239,7 +322,14 @@ export default class AiSdkProvider {
239
322
  }
240
323
  signal?.throwIfAborted();
241
324
  try {
242
- const timeout = AbortSignal.timeout(this.#fetchTimeoutMs);
325
+ const timeout = this.#fetchTimeoutMs > 0
326
+ ? AbortSignal.timeout(this.#fetchTimeoutMs)
327
+ : undefined;
328
+ const requestSignal = signal === undefined
329
+ ? timeout
330
+ : timeout === undefined
331
+ ? signal
332
+ : AbortSignal.any([signal, timeout]);
243
333
  const response = await this.#fetch(this.#promptTokensUrl, {
244
334
  method: "POST",
245
335
  headers: { "Content-Type": "application/json", ...this.#headers },
@@ -248,7 +338,7 @@ export default class AiSdkProvider {
248
338
  messages,
249
339
  ...this.#reasoningBody(),
250
340
  }),
251
- signal: signal === undefined ? timeout : AbortSignal.any([signal, timeout]),
341
+ ...(requestSignal === undefined ? {} : { signal: requestSignal }),
252
342
  });
253
343
  if (!response.ok) {
254
344
  return estimatePromptTokens(messages, `llama-server input-token endpoint returned HTTP ${response.status}`);
@@ -268,32 +358,28 @@ export default class AiSdkProvider {
268
358
  return estimatePromptTokens(messages, `llama-server input-token measurement failed: ${cause instanceof Error ? cause.message : String(cause)}`);
269
359
  }
270
360
  }
271
- calculateCost(usage) { return this.#calculateCost(usage); }
272
- calculateCharge(usage) {
273
- return this.#calculateCharge?.(usage)
274
- ?? { kind: "unknown", reason: "no provider rate or settled charge is available" };
275
- }
276
- // Reasoning intent maps independently of grammar transport. The llama-server
277
- // template mapping is owned by {§llama-reasoning-request}.
278
- #reasoningBody() {
361
+ // Reasoning activation and allowance are independent of grammar transport;
362
+ // only the response representation becomes lossless when evidence is needed.
363
+ // The llama-server template mapping is owned by {§llama-reasoning-request}.
364
+ #reasoningBody(preserveGrammarSentence = false) {
279
365
  const { mode, budget } = this.#reasoning;
280
366
  const on = mode !== "off";
281
367
  switch (this.#reasoningStyle) {
282
368
  case "template": {
283
369
  const allowance = mode === "off"
284
370
  ? 0
285
- : mode === "on" ? budget : this.reasoningReserve;
371
+ : mode === "on" && budget !== null ? budget : this.reasoningReserve;
286
372
  return {
287
373
  chat_template_kwargs: { enable_thinking: on },
288
- reasoning_format: "auto",
374
+ reasoning_format: preserveGrammarSentence ? "none" : "auto",
289
375
  ...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
290
376
  };
291
377
  }
292
378
  case "think": return on ? { think: true } : {};
293
379
  case "include_reasoning": return on ? { include_reasoning: true } : {};
294
- // effort tiers from the budget; off/adaptive omit the field (the
295
- // API's default depth is its adaptive).
296
- case "effort": return mode === "on" ? { reasoning_effort: effortFromBudget(budget) } : {};
380
+ // Explicit on uses the portable enabled posture or a tier derived
381
+ // from an explicit budget; off/adaptive omit the field.
382
+ case "effort": return mode === "on" ? { reasoning_effort: effortFromReasoning(this.#reasoning) } : {};
297
383
  // Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
298
384
  // reason-by-default model (DeepSeek V4: default 'high') reasoning.
299
385
  // ADAPTIVE omits the field: the backend's own default posture IS the
@@ -303,19 +389,25 @@ export default class AiSdkProvider {
303
389
  // efforts 400.
304
390
  case "effort_explicit": return mode === "off"
305
391
  ? { reasoning_effort: "none" }
306
- : mode === "on" ? { reasoning_effort: effortFromBudget(budget) } : {};
392
+ : mode === "on" ? { reasoning_effort: effortFromReasoning(this.#reasoning) } : {};
307
393
  // {§deepseek-reasoning-request}
308
394
  case "thinking_effort": return mode === "off"
309
395
  ? { thinking: { type: "disabled" } }
310
396
  : mode === "on" ? {
311
397
  thinking: { type: "enabled" },
312
- reasoning_effort: effortFromBudget(budget),
398
+ ...(budget === null ? {} : { reasoning_effort: effortFromBudget(budget) }),
313
399
  } : {};
314
400
  // Anthropic compat: explicit thinking object. off → disabled; on →
315
- // enabled with budget_tokens; adaptive → omit (the API default).
401
+ // enabled with the explicit budget or resolved reserve; adaptive →
402
+ // omit (the API default).
316
403
  case "anthropic": return mode === "off"
317
404
  ? { thinking: { type: "disabled" } }
318
- : mode === "on" ? { thinking: { type: "enabled", budget_tokens: budget } } : {};
405
+ : mode === "on" ? {
406
+ thinking: {
407
+ type: "enabled",
408
+ budget_tokens: budget ?? this.reasoningReserve,
409
+ },
410
+ } : {};
319
411
  case "none": return {};
320
412
  }
321
413
  }
@@ -380,7 +472,7 @@ export default class AiSdkProvider {
380
472
  case "none": return this.#frequencyPenalty > 0 ? { frequency_penalty: this.#frequencyPenalty } : {};
381
473
  }
382
474
  }
383
- // First-party telemetry headers ({§provider-request-authority}): forwarded only when the spec
475
+ // First-party telemetry headers ({§provider-request-authority} {§provider-call-kind}): forwarded only when the spec
384
476
  // opted in (the plurnk endpoint). The gate is here, not at the call site, so
385
477
  // attributions/client/strikes can never reach a third-party backend even if
386
478
  // the consumer passes them to the wrong provider. Empty values emit no header
@@ -388,7 +480,7 @@ export default class AiSdkProvider {
388
480
  // absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
389
481
  // ride HTTP headers only — the packet never carries them (the model must
390
482
  // never see strike state; engine accounting is not a metric to game).
391
- #metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn) {
483
+ #metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind) {
392
484
  if (!this.#firstPartyMetadata)
393
485
  return {};
394
486
  const h = {};
@@ -421,6 +513,8 @@ export default class AiSdkProvider {
421
513
  h["Plurnk-Loop"] = String(loop);
422
514
  if (turn !== undefined && Number.isInteger(turn) && turn >= 1)
423
515
  h["Plurnk-Turn"] = String(turn);
516
+ if (callKind !== undefined)
517
+ h["Plurnk-Call-Kind"] = callKind;
424
518
  return h;
425
519
  }
426
520
  // PLURNK_PROVIDERS_GBNF_DEBUG ({§gbnf-response-observation}): validate the supplied GBNF locally and fail
@@ -460,10 +554,40 @@ export default class AiSdkProvider {
460
554
  out[k] = v;
461
555
  return out;
462
556
  }
463
- async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }) {
557
+ #requestProviderOptions(workerId) {
558
+ const reasoningOptions = this.#reasoning.mode === "off"
559
+ ? undefined
560
+ : this.#reasoningResponseProviderOptions;
561
+ if (this.#cacheAffinity?.target !== "provider-option")
562
+ return reasoningOptions;
563
+ const { provider, name } = this.#cacheAffinity;
564
+ return {
565
+ ...reasoningOptions,
566
+ [provider]: {
567
+ ...reasoningOptions?.[provider],
568
+ [name]: workerId,
569
+ },
570
+ };
571
+ }
572
+ #accounting(outcome, usage, evidence, status) {
573
+ const knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
574
+ const direct = this.#normalizeCost?.(evidence);
575
+ return validateProviderRequestAccounting({
576
+ provider: this.#source,
577
+ model: this.#model,
578
+ outcome,
579
+ ...(status === undefined ? {} : { status }),
580
+ ...(knownUsage === undefined ? {} : { usage: knownUsage }),
581
+ cost: resolveProviderCost(direct, this.#estimateCost(knownUsage)),
582
+ });
583
+ }
584
+ async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling, observeRequest, callKind }) {
464
585
  // {§provider-interface} The worker identity is required.
465
586
  if (workerId === undefined || workerId.length === 0)
466
587
  throw new Error("generate: workerId is required — the worker's stable, opaque identity");
588
+ if (callKind !== undefined && callKind !== "emission" && callKind !== "bare") {
589
+ throw new Error(`generate: unsupported callKind ${JSON.stringify(callKind)}`);
590
+ }
467
591
  // Reject before any wire call when already aborted
468
592
  // ({§provider-failure-normalization}).
469
593
  signal?.throwIfAborted();
@@ -473,6 +597,8 @@ export default class AiSdkProvider {
473
597
  if (wantGrammar && this.#gbnfDebug)
474
598
  this.#assertGrammarValid(grammar);
475
599
  const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
600
+ const preserveGrammarSentence = wantGrammar
601
+ && this.#reasoningStyle === "template";
476
602
  // Assembly order = precedence: the family's sampling DEFAULTS
477
603
  // (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
478
604
  // paths and the name promises every request) < the caller's `sampling`
@@ -485,78 +611,181 @@ export default class AiSdkProvider {
485
611
  ...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
486
612
  model: this.#model,
487
613
  messages,
488
- ...this.#reasoningBody(),
614
+ ...this.#reasoningBody(preserveGrammarSentence),
489
615
  ...this.#grammarBody(sendGrammar),
490
616
  ...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
491
617
  // Request per-token logprobs only when enabled (managed field —
492
618
  // reserved from caller sampling; the env flag is the single control).
493
619
  ...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
494
620
  ...this.#slotBody(workerId),
495
- // Prompt-cache affinity -- workerId as the OpenAI-standard
496
- // prompt_cache_key routes a worker's turns to one serverless replica so
497
- // its stable prefix caches (managed; reserved from caller sampling).
498
- ...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
621
+ ...(this.#cacheAffinity?.target === "body"
622
+ ? { [this.#cacheAffinity.name]: workerId }
623
+ : {}),
499
624
  };
500
625
  // Per-request headers = static auth/routing + any first-party telemetry.
501
- const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
502
- const headers = Object.keys(metaHeaders).length > 0 ? { ...this.#headers, ...metaHeaders } : this.#headers;
503
- let raw;
504
- try {
505
- raw = this.#languageModel === undefined
506
- ? await executeOpenAICompatible({
507
- url: this.#url,
626
+ const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind);
627
+ const headers = new Headers(this.#headers);
628
+ if (this.#cacheAffinity?.target === "header") {
629
+ headers.set(this.#cacheAffinity.name, workerId);
630
+ }
631
+ for (const [name, value] of Object.entries(metaHeaders))
632
+ headers.set(name, value);
633
+ const requestHeaders = Object.fromEntries(headers.entries());
634
+ const accounting = [];
635
+ const operationTimeout = this.#operationTimeoutMs > 0
636
+ ? AbortSignal.timeout(this.#operationTimeoutMs)
637
+ : undefined;
638
+ const operationSignal = signal === undefined
639
+ ? operationTimeout
640
+ : operationTimeout === undefined
641
+ ? signal
642
+ : AbortSignal.any([signal, operationTimeout]);
643
+ const executeRequest = async () => {
644
+ let settle;
645
+ try {
646
+ settle = await observeRequest?.({
647
+ provider: this.#source,
508
648
  model: this.#model,
509
- headers,
510
- body,
511
- messages,
512
- signal,
513
- fetch: this.#fetch,
514
- fetchTimeoutMs: this.#fetchTimeoutMs,
515
- streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
516
- retryAttempts: this.#retryAttempts,
517
- streaming: this.#streaming,
518
- captureRawBody: this.#rawBody,
519
- })
520
- : await executeAiSdkModel({
521
- languageModel: this.#languageModel,
522
- headers,
523
- messages,
524
- signal,
525
- fetchTimeoutMs: this.#fetchTimeoutMs,
526
- streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
527
- retryAttempts: this.#retryAttempts,
528
- streaming: this.#streaming,
529
- captureRawBody: this.#rawBody,
530
- temperature: this.#tuningFloors
531
- ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
532
- : typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
533
- topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
534
- topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
535
- presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
536
- frequencyPenalty: typeof sampling?.frequency_penalty === "number"
537
- ? sampling.frequency_penalty
538
- : this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
539
- stopSequences: typeof sampling?.stop === "string"
540
- ? [sampling.stop]
541
- : Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
542
- ? sampling.stop
543
- : undefined,
544
- seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
545
- maxOutputTokens: maxTokens,
546
- reasoning: this.#reasoning.mode === "off"
547
- ? "none"
548
- : this.#reasoning.mode === "adaptive"
549
- ? "provider-default"
550
- : effortFromBudget(this.#reasoning.budget),
551
649
  });
650
+ }
651
+ catch (cause) {
652
+ throw new ProviderRequestObserverError(cause);
653
+ }
654
+ const settleAccounting = async (outcome, usage, evidence, status) => {
655
+ let requestAccounting;
656
+ let normalizationFailure;
657
+ try {
658
+ requestAccounting = this.#accounting(outcome, usage, evidence, status);
659
+ }
660
+ catch (cause) {
661
+ normalizationFailure = { cause };
662
+ let knownUsage;
663
+ try {
664
+ knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
665
+ }
666
+ catch {
667
+ knownUsage = undefined;
668
+ }
669
+ requestAccounting = validateProviderRequestAccounting({
670
+ provider: this.#source,
671
+ model: this.#model,
672
+ outcome,
673
+ ...(status === undefined ? {} : { status }),
674
+ ...(knownUsage === undefined ? {} : { usage: knownUsage }),
675
+ cost: {
676
+ kind: "unknown",
677
+ reason: "provider request accounting could not be normalized after physical I/O",
678
+ },
679
+ });
680
+ }
681
+ accounting.push(requestAccounting);
682
+ try {
683
+ await settle?.(requestAccounting);
684
+ }
685
+ catch (cause) {
686
+ throw new ProviderRequestObserverError(cause);
687
+ }
688
+ if (normalizationFailure !== undefined) {
689
+ throw new ProviderRequestAccountingError(normalizationFailure.cause);
690
+ }
691
+ return requestAccounting;
692
+ };
693
+ let response;
694
+ try {
695
+ response = this.#languageModel === undefined
696
+ ? await executeOpenAICompatible({
697
+ url: this.#url,
698
+ model: this.#model,
699
+ headers: requestHeaders,
700
+ body,
701
+ messages,
702
+ signal: operationSignal,
703
+ fetch: this.#fetch,
704
+ fetchTimeoutMs: this.#fetchTimeoutMs,
705
+ firstContentTimeoutMs: this.#firstContentTimeoutMs,
706
+ streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
707
+ streaming: this.#streaming,
708
+ captureRawBody: this.#rawBody,
709
+ })
710
+ : await executeAiSdkModel({
711
+ languageModel: this.#languageModel,
712
+ headers: requestHeaders,
713
+ providerOptions: this.#requestProviderOptions(workerId),
714
+ systemProviderOptions: this.#systemCacheProviderOptions,
715
+ messages,
716
+ signal: operationSignal,
717
+ fetchTimeoutMs: this.#fetchTimeoutMs,
718
+ firstContentTimeoutMs: this.#firstContentTimeoutMs,
719
+ streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
720
+ streaming: this.#streaming,
721
+ captureRawBody: this.#rawBody,
722
+ temperature: this.#tuningFloors
723
+ ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
724
+ : typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
725
+ topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
726
+ topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
727
+ presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
728
+ frequencyPenalty: typeof sampling?.frequency_penalty === "number"
729
+ ? sampling.frequency_penalty
730
+ : this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
731
+ stopSequences: typeof sampling?.stop === "string"
732
+ ? [sampling.stop]
733
+ : Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
734
+ ? sampling.stop
735
+ : undefined,
736
+ seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
737
+ maxOutputTokens: maxTokens,
738
+ reasoning: this.#reasoning.mode === "off"
739
+ ? "none"
740
+ : this.#reasoning.mode === "adaptive"
741
+ ? "provider-default"
742
+ : effortFromReasoning(this.#reasoning),
743
+ });
744
+ }
745
+ catch (error) {
746
+ const failure = transportFailureEvidence(error);
747
+ await settleAccounting("error", failure.usage, failure.chargeEvidence, failure.status);
748
+ throw error;
749
+ }
750
+ await settleAccounting("response", response.usage, response.chargeEvidence);
751
+ return response;
752
+ };
753
+ let raw;
754
+ try {
755
+ const { retry } = prepareRetries({
756
+ maxRetries: this.#retryAttempts,
757
+ abortSignal: operationSignal,
758
+ });
759
+ raw = await retry(executeRequest);
552
760
  }
553
761
  catch (err) {
762
+ if (err instanceof ProviderRequestObserverError
763
+ || err instanceof ProviderRequestAccountingError)
764
+ throw err.cause;
554
765
  if (signal?.aborted)
555
766
  throw err;
767
+ if (operationTimeout?.aborted) {
768
+ const timeout = new ProviderTimeoutError("operation", this.#operationTimeoutMs, err);
769
+ throw new ProviderError(this.#source, "deadline_exceeded", timeout.message, {
770
+ status: 504,
771
+ cause: timeout,
772
+ retryable: false,
773
+ extensions: {
774
+ timeoutPhase: timeout.phase,
775
+ timeoutMs: timeout.timeoutMs,
776
+ },
777
+ accounting,
778
+ });
779
+ }
556
780
  const pe = toProviderError(err, this.#source, this.#errorDetailLimit);
557
781
  if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
558
- throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
782
+ throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, {
783
+ status: pe.status,
784
+ cause: err,
785
+ accounting,
786
+ });
559
787
  }
788
+ pe.prependAccounting(accounting);
560
789
  throw pe;
561
790
  }
562
791
  // llama-server --special renders EOG tokens as text, so a turn ending
@@ -567,46 +796,49 @@ export default class AiSdkProvider {
567
796
  // wire text for forensics.
568
797
  if (this.#eosText !== undefined)
569
798
  raw.content = stripTrailingSpecial(raw.content, this.#eosText);
570
- const taggedReasoning = projectTaggedReasoning(raw.content, raw.reasoning, this.#reasoningResponseStyle);
571
- // Preserve the exact sentence seen at the grammar boundary. llama-server's
572
- // `reasoning_format: "auto"` projects one raw Harmony enclosure into the
573
- // reasoning/content fields; the wire field's presence is the proof that the
574
- // projection occurred. The provider represents this evidence and never grades it.
799
+ const grammarInput = raw.content;
800
+ const projectedReasoning = preserveGrammarSentence && !raw.reasoningProjected
801
+ ? projectTemplateReasoning(raw.content)
802
+ : projectTaggedReasoning(raw.content, raw.reasoning, this.#reasoningResponseStyle);
803
+ // Preserve the exact pre-projection response. Constrained template turns
804
+ // request `reasoning_format: "none"`, so even an empty channel and any
805
+ // template-provided opener remain observable. An unexpectedly projected
806
+ // response cannot supply independent evidence.
575
807
  let grammarEvidence;
576
808
  if (wantGrammar) {
577
- if (taggedReasoning.projected) {
578
- grammarEvidence = {
579
- input: raw.content,
580
- contentStart: taggedReasoning.contentStart,
581
- transported: sendGrammar !== undefined,
582
- };
583
- }
584
- else if (this.#reasoningStyle === "template" && this.#reasoning.mode !== "off") {
585
- if (raw.reasoningProjected) {
586
- const prefix = `<|channel>thought\n${raw.reasoning}<channel|>`;
809
+ if (preserveGrammarSentence) {
810
+ if (!raw.reasoningProjected) {
587
811
  grammarEvidence = {
588
- input: `${prefix}${raw.content}`,
589
- contentStart: [...prefix].length,
812
+ input: grammarInput,
813
+ contentStart: projectedReasoning.projected ? projectedReasoning.contentStart : 0,
590
814
  transported: sendGrammar !== undefined,
591
815
  };
592
816
  }
593
817
  }
818
+ else if (projectedReasoning.projected) {
819
+ grammarEvidence = {
820
+ input: grammarInput,
821
+ contentStart: projectedReasoning.contentStart,
822
+ transported: sendGrammar !== undefined,
823
+ };
824
+ }
594
825
  else {
595
826
  grammarEvidence = {
596
- input: raw.content,
827
+ input: grammarInput,
597
828
  contentStart: 0,
598
829
  transported: sendGrammar !== undefined,
599
830
  };
600
831
  }
601
832
  }
602
- if (taggedReasoning.projected) {
603
- raw.content = taggedReasoning.content;
604
- raw.reasoning = taggedReasoning.reasoning;
605
- raw.usage = attributeUnitemizedReasoning(raw.usage, raw.reasoning, raw.content);
833
+ if (projectedReasoning.projected) {
834
+ raw.content = projectedReasoning.content;
835
+ raw.reasoning = projectedReasoning.reasoning;
606
836
  }
607
837
  let notices;
608
838
  const usage = raw.usage;
609
- if (sendGrammar !== undefined && this.tokenize !== undefined) {
839
+ if (sendGrammar !== undefined
840
+ && this.tokenize !== undefined
841
+ && usage?.outputTokens !== undefined) {
610
842
  // Channel-escape detector: completion tokens
611
843
  // billed far beyond every visible channel mean the decode ESCAPED into
612
844
  // a server-discarded reasoning block mid-emission. This diagnostic
@@ -617,12 +849,12 @@ export default class AiSdkProvider {
617
849
  this.tokenize(raw.reasoning),
618
850
  ]);
619
851
  const visible = contentTokens.length + reasoningTokens.length;
620
- if (usage.completion > visible + 64) {
852
+ if (usage.outputTokens > visible + 64) {
621
853
  (notices ??= []).push({
622
854
  source: this.#source,
623
855
  kind: "grammar_unenforced",
624
856
  level: "warn",
625
- message: `decode escaped the grammar: ${usage.completion} completion tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
857
+ message: `decode escaped the grammar: ${usage.outputTokens} output tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
626
858
  position: [...raw.content].length,
627
859
  });
628
860
  }
@@ -642,12 +874,12 @@ export default class AiSdkProvider {
642
874
  ...(raw.reasoningEncrypted.length > 0
643
875
  ? { reasoningEncrypted: raw.reasoningEncrypted }
644
876
  : {}),
645
- usage,
646
877
  model: raw.model,
647
878
  ...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
648
879
  };
649
880
  const evidence = {
650
881
  assistantRaw: raw,
882
+ accounting,
651
883
  ...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
652
884
  ...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
653
885
  ...(meta !== undefined ? { meta } : {}),
@@ -660,6 +892,7 @@ export default class AiSdkProvider {
660
892
  };
661
893
  throw new ProviderError(this.#source, "resource_interrupted", "The provider interrupted generation because inference resources were unavailable.", {
662
894
  attempt,
895
+ accounting,
663
896
  extensions: {
664
897
  stage: "provider-response",
665
898
  finishReason: "resource_interrupted",