@plurnk/plurnk-providers 1.5.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +36 -22
- package/SPEC.md +133 -59
- package/dist/AiSdkProvider.d.ts +19 -26
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +318 -106
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +4 -9
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +36 -9
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +2 -21
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +19 -14
- package/dist/Pool.js.map +1 -1
- package/dist/accounting.d.ts +5 -2
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js +100 -16
- package/dist/accounting.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +9 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +160 -62
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +7 -3
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +30 -24
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +18 -7
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +10 -10
- package/dist/cost.d.ts.map +1 -1
- package/dist/cost.js +90 -42
- package/dist/cost.js.map +1 -1
- package/dist/env.d.ts +5 -1
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +30 -10
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +14 -2
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +58 -2
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +4 -4
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -2
- package/dist/index.js.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/sdkModels.d.ts +6 -2
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +38 -5
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +33 -31
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +21 -5
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +164 -83
- package/dist/usage.js.map +1 -1
- package/package.json +7 -6
- package/src/AiSdkProvider.test.ts +788 -191
- package/src/AiSdkProvider.ts +381 -124
- package/src/Mock.test.ts +37 -12
- package/src/Mock.ts +45 -14
- package/src/Pool.test.ts +19 -6
- package/src/Pool.ts +20 -16
- package/src/ProviderRegistry.test.ts +16 -11
- package/src/accounting.test.ts +58 -22
- package/src/accounting.ts +120 -18
- package/src/aiSdkTransport.test.ts +42 -49
- package/src/aiSdkTransport.ts +174 -62
- package/src/boundaries.test.ts +1 -0
- package/src/catalogProvider.test.ts +258 -22
- package/src/catalogProvider.ts +42 -27
- package/src/compatibleProvider.test.ts +6 -3
- package/src/compatibleProvider.ts +20 -7
- package/src/cost.test.ts +55 -36
- package/src/cost.ts +111 -50
- package/src/defaults.test.ts +13 -3
- package/src/env.test.ts +54 -5
- package/src/env.ts +43 -18
- package/src/errors.test.ts +47 -2
- package/src/errors.ts +67 -3
- package/src/index.ts +21 -5
- package/src/ollama.test.ts +4 -1
- package/src/ollama.ts +3 -3
- package/src/sdkModels.test.ts +76 -4
- package/src/sdkModels.ts +45 -7
- package/src/types.ts +77 -38
- package/src/usage.test.ts +112 -116
- package/src/usage.ts +209 -93
package/dist/AiSdkProvider.js
CHANGED
|
@@ -5,13 +5,28 @@
|
|
|
5
5
|
// Composition, not inheritance: an official AI SDK language model supplies the
|
|
6
6
|
// ordinary vendor protocol. The compatible URL path remains only for PLURNK
|
|
7
7
|
// extensions and local endpoint probes the SDK cannot represent.
|
|
8
|
-
import {
|
|
9
|
-
import {
|
|
10
|
-
import {
|
|
8
|
+
import { MAX_PROVIDER_TIMEOUT_MS } from "./env.js";
|
|
9
|
+
import { executeAiSdkModel, executeOpenAICompatible, transportFailureEvidence, } from "./aiSdkTransport.js";
|
|
10
|
+
import { prepareRetries } from "ai/internal";
|
|
11
|
+
import { toProviderError, ProviderError, ProviderTimeoutError } from "./errors.js";
|
|
11
12
|
import { validateGbnf } from "@plurnk/gbnf";
|
|
12
13
|
import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.js";
|
|
13
14
|
import { emitWarningOnce } from "./warnings.js";
|
|
14
|
-
import {
|
|
15
|
+
import { resolveProviderCost } from "./cost.js";
|
|
16
|
+
import { validateProviderRequestAccounting } from "./accounting.js";
|
|
17
|
+
import { validateProviderUsage } from "./usage.js";
|
|
18
|
+
class ProviderRequestObserverError extends Error {
|
|
19
|
+
constructor(cause) {
|
|
20
|
+
super("provider request accounting could not be durably settled", { cause });
|
|
21
|
+
this.name = "ProviderRequestObserverError";
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
class ProviderRequestAccountingError extends Error {
|
|
25
|
+
constructor(cause) {
|
|
26
|
+
super("provider request accounting could not be normalized", { cause });
|
|
27
|
+
this.name = "ProviderRequestAccountingError";
|
|
28
|
+
}
|
|
29
|
+
}
|
|
15
30
|
// Drop trailing occurrences of a server-rendered EOG marker. llama-server
|
|
16
31
|
// under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
|
|
17
32
|
// trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
|
|
@@ -52,11 +67,20 @@ const projectLeadingReasoning = (content, structuredReasoning, opening, closing)
|
|
|
52
67
|
const projectTaggedReasoning = (content, structuredReasoning, style) => style === "think-tags"
|
|
53
68
|
? projectLeadingReasoning(content, structuredReasoning, "<think>", "</think>")
|
|
54
69
|
: { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
|
|
55
|
-
// llama-server's template reasoning parser can project
|
|
56
|
-
// of the OpenAI-compatible response. Grammar evidence
|
|
57
|
-
// that lossy projection, so constrained template turns
|
|
58
|
-
// split the observed enclosure here.
|
|
59
|
-
const projectTemplateReasoning = (content) =>
|
|
70
|
+
// llama-server's template reasoning parser can project either supported leading
|
|
71
|
+
// reasoning envelope out of the OpenAI-compatible response. Grammar evidence
|
|
72
|
+
// needs the sentence before that lossy projection, so constrained template turns
|
|
73
|
+
// request it verbatim and split the observed enclosure here.
|
|
74
|
+
const projectTemplateReasoning = (content) => {
|
|
75
|
+
for (const [opening, closing] of [
|
|
76
|
+
["<|channel>thought\n", "<channel|>"],
|
|
77
|
+
["<think>\n", "</think>"],
|
|
78
|
+
]) {
|
|
79
|
+
if (content.startsWith(opening))
|
|
80
|
+
return projectLeadingReasoning(content, "", opening, closing);
|
|
81
|
+
}
|
|
82
|
+
return { content, reasoning: "", projected: false, contentStart: 0 };
|
|
83
|
+
};
|
|
60
84
|
// Shared budget→effort breakpoints (xai and google had identical copies).
|
|
61
85
|
export const effortFromBudget = (budget) => {
|
|
62
86
|
if (budget <= 1000)
|
|
@@ -65,6 +89,11 @@ export const effortFromBudget = (budget) => {
|
|
|
65
89
|
return "medium";
|
|
66
90
|
return "high";
|
|
67
91
|
};
|
|
92
|
+
// AI SDK's portable reasoning control has no boolean-enabled value. `medium`
|
|
93
|
+
// is the neutral activation projection for an explicit, unqualified `on`; it
|
|
94
|
+
// changes no PLURNK token reserve. An operator budget, when present, remains
|
|
95
|
+
// the only input to the existing magnitude-to-tier projection.
|
|
96
|
+
const effortFromReasoning = (reasoning) => reasoning.budget === null ? "medium" : effortFromBudget(reasoning.budget);
|
|
68
97
|
// Body keys the provider owns — a caller's `sampling` passthrough may not set
|
|
69
98
|
// these. Two families:
|
|
70
99
|
// transport/managed — grammar transport, the stream/JSON choice, slot pinning,
|
|
@@ -91,6 +120,8 @@ export default class AiSdkProvider {
|
|
|
91
120
|
#url;
|
|
92
121
|
#languageModel;
|
|
93
122
|
#fetchTimeoutMs;
|
|
123
|
+
#operationTimeoutMs;
|
|
124
|
+
#firstContentTimeoutMs;
|
|
94
125
|
#streamIdleTimeoutMs;
|
|
95
126
|
#headers;
|
|
96
127
|
#fetch;
|
|
@@ -110,12 +141,13 @@ export default class AiSdkProvider {
|
|
|
110
141
|
#reasoningResponseStyle;
|
|
111
142
|
#countPromptTokens;
|
|
112
143
|
#promptTokensUrl;
|
|
113
|
-
#
|
|
114
|
-
#
|
|
115
|
-
#normalizeCharge;
|
|
144
|
+
#estimateCost;
|
|
145
|
+
#normalizeCost;
|
|
116
146
|
#source;
|
|
117
147
|
#grammarStyle;
|
|
118
|
-
#
|
|
148
|
+
#cacheAffinity;
|
|
149
|
+
#systemCacheProviderOptions;
|
|
150
|
+
#reasoningResponseProviderOptions;
|
|
119
151
|
#serviceTier;
|
|
120
152
|
#gbnfDebug;
|
|
121
153
|
#streaming;
|
|
@@ -145,7 +177,19 @@ export default class AiSdkProvider {
|
|
|
145
177
|
if ((this.#url === undefined) === (this.#languageModel === undefined)) {
|
|
146
178
|
throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
|
|
147
179
|
}
|
|
180
|
+
for (const [name, value] of [
|
|
181
|
+
["fetchTimeoutMs", config.fetchTimeoutMs],
|
|
182
|
+
["operationTimeoutMs", config.operationTimeoutMs],
|
|
183
|
+
["firstContentTimeoutMs", config.firstContentTimeoutMs],
|
|
184
|
+
["streamIdleTimeoutMs", config.streamIdleTimeoutMs],
|
|
185
|
+
]) {
|
|
186
|
+
if (value !== undefined && (!Number.isInteger(value) || value < 0 || value > MAX_PROVIDER_TIMEOUT_MS)) {
|
|
187
|
+
throw new Error(`${config.source ?? "provider"}: ${name} must be an integer from 0 through ${MAX_PROVIDER_TIMEOUT_MS} milliseconds`);
|
|
188
|
+
}
|
|
189
|
+
}
|
|
148
190
|
this.#fetchTimeoutMs = config.fetchTimeoutMs;
|
|
191
|
+
this.#operationTimeoutMs = config.operationTimeoutMs;
|
|
192
|
+
this.#firstContentTimeoutMs = config.firstContentTimeoutMs;
|
|
149
193
|
this.#streamIdleTimeoutMs = config.streamIdleTimeoutMs;
|
|
150
194
|
this.#headers = config.headers ?? {};
|
|
151
195
|
this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
|
|
@@ -173,12 +217,33 @@ export default class AiSdkProvider {
|
|
|
173
217
|
}
|
|
174
218
|
this.#countPromptTokens = config.countPromptTokens ?? ((messages) => estimatePromptTokens(messages));
|
|
175
219
|
this.#promptTokensUrl = config.promptTokensUrl;
|
|
176
|
-
this.#
|
|
177
|
-
|
|
178
|
-
|
|
220
|
+
this.#estimateCost = config.estimateCost
|
|
221
|
+
?? (() => ({
|
|
222
|
+
kind: "unknown",
|
|
223
|
+
reason: "the request reported no direct cost and no model rate is configured",
|
|
224
|
+
}));
|
|
225
|
+
this.#normalizeCost = config.normalizeCost;
|
|
179
226
|
this.#source = config.source ?? "provider";
|
|
180
227
|
this.#grammarStyle = config.grammarStyle ?? "none";
|
|
181
|
-
this.#
|
|
228
|
+
this.#cacheAffinity = config.cacheAffinity;
|
|
229
|
+
this.#systemCacheProviderOptions = config.systemCacheProviderOptions;
|
|
230
|
+
this.#reasoningResponseProviderOptions = config.reasoningResponseProviderOptions;
|
|
231
|
+
if (this.#languageModel === undefined && this.#cacheAffinity?.target === "provider-option") {
|
|
232
|
+
throw new Error(`${this.#source}: native provider-option cache affinity requires an AI SDK model`);
|
|
233
|
+
}
|
|
234
|
+
if (this.#languageModel !== undefined && this.#cacheAffinity?.target === "body") {
|
|
235
|
+
throw new Error(`${this.#source}: body cache affinity requires an OpenAI-compatible URL`);
|
|
236
|
+
}
|
|
237
|
+
if (this.#languageModel === undefined && this.#systemCacheProviderOptions !== undefined) {
|
|
238
|
+
throw new Error(`${this.#source}: system cache provider options require an AI SDK model`);
|
|
239
|
+
}
|
|
240
|
+
if (this.#languageModel === undefined && this.#reasoningResponseProviderOptions !== undefined) {
|
|
241
|
+
throw new Error(`${this.#source}: reasoning response provider options require an AI SDK model`);
|
|
242
|
+
}
|
|
243
|
+
if (this.#cacheAffinity?.target === "provider-option"
|
|
244
|
+
&& Object.hasOwn(this.#reasoningResponseProviderOptions?.[this.#cacheAffinity.provider] ?? {}, this.#cacheAffinity.name)) {
|
|
245
|
+
throw new Error(`${this.#source}: reasoning response options conflict with cache-affinity option ${this.#cacheAffinity.provider}.${this.#cacheAffinity.name}`);
|
|
246
|
+
}
|
|
182
247
|
this.#serviceTier = config.serviceTier;
|
|
183
248
|
this.#gbnfDebug = config.gbnfDebug ?? false;
|
|
184
249
|
this.#streaming = config.streaming ?? true;
|
|
@@ -198,10 +263,17 @@ export default class AiSdkProvider {
|
|
|
198
263
|
const reasoningReserve = this.reasoningReserve;
|
|
199
264
|
if (this.#reasoningStyle === "template"
|
|
200
265
|
&& this.#reasoning.mode === "on"
|
|
266
|
+
&& this.#reasoning.budget !== null
|
|
201
267
|
&& reasoningReserve !== null
|
|
202
268
|
&& this.#reasoning.budget > reasoningReserve) {
|
|
203
269
|
throw new Error(`${this.#source}: PLURNK_PROVIDERS_REASONING_BUDGET (${this.#reasoning.budget}) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE (${reasoningReserve})`);
|
|
204
270
|
}
|
|
271
|
+
if (this.#reasoningStyle === "anthropic"
|
|
272
|
+
&& this.#reasoning.mode === "on"
|
|
273
|
+
&& this.#reasoning.budget === null
|
|
274
|
+
&& reasoningReserve === null) {
|
|
275
|
+
throw new Error(`${this.#source}: explicit Anthropic reasoning requires a resolved reasoning reserve or PLURNK_PROVIDERS_REASONING_BUDGET`);
|
|
276
|
+
}
|
|
205
277
|
const { tokenizeUrl } = config;
|
|
206
278
|
if (tokenizeUrl !== undefined) {
|
|
207
279
|
this.tokenize = async (text) => {
|
|
@@ -209,7 +281,9 @@ export default class AiSdkProvider {
|
|
|
209
281
|
method: "POST",
|
|
210
282
|
headers: { "Content-Type": "application/json", ...this.#headers },
|
|
211
283
|
body: JSON.stringify({ content: text }),
|
|
212
|
-
|
|
284
|
+
...(this.#fetchTimeoutMs > 0
|
|
285
|
+
? { signal: AbortSignal.timeout(this.#fetchTimeoutMs) }
|
|
286
|
+
: {}),
|
|
213
287
|
});
|
|
214
288
|
if (!res.ok)
|
|
215
289
|
throw new Error(`${this.#source}: tokenize endpoint returned ${res.status}`);
|
|
@@ -248,7 +322,14 @@ export default class AiSdkProvider {
|
|
|
248
322
|
}
|
|
249
323
|
signal?.throwIfAborted();
|
|
250
324
|
try {
|
|
251
|
-
const timeout =
|
|
325
|
+
const timeout = this.#fetchTimeoutMs > 0
|
|
326
|
+
? AbortSignal.timeout(this.#fetchTimeoutMs)
|
|
327
|
+
: undefined;
|
|
328
|
+
const requestSignal = signal === undefined
|
|
329
|
+
? timeout
|
|
330
|
+
: timeout === undefined
|
|
331
|
+
? signal
|
|
332
|
+
: AbortSignal.any([signal, timeout]);
|
|
252
333
|
const response = await this.#fetch(this.#promptTokensUrl, {
|
|
253
334
|
method: "POST",
|
|
254
335
|
headers: { "Content-Type": "application/json", ...this.#headers },
|
|
@@ -257,7 +338,7 @@ export default class AiSdkProvider {
|
|
|
257
338
|
messages,
|
|
258
339
|
...this.#reasoningBody(),
|
|
259
340
|
}),
|
|
260
|
-
|
|
341
|
+
...(requestSignal === undefined ? {} : { signal: requestSignal }),
|
|
261
342
|
});
|
|
262
343
|
if (!response.ok) {
|
|
263
344
|
return estimatePromptTokens(messages, `llama-server input-token endpoint returned HTTP ${response.status}`);
|
|
@@ -277,11 +358,6 @@ export default class AiSdkProvider {
|
|
|
277
358
|
return estimatePromptTokens(messages, `llama-server input-token measurement failed: ${cause instanceof Error ? cause.message : String(cause)}`);
|
|
278
359
|
}
|
|
279
360
|
}
|
|
280
|
-
calculateCost(usage) { return this.#calculateCost(usage); }
|
|
281
|
-
calculateCharge(usage) {
|
|
282
|
-
return this.#calculateCharge?.(usage)
|
|
283
|
-
?? { kind: "unknown", reason: "no provider rate or settled charge is available" };
|
|
284
|
-
}
|
|
285
361
|
// Reasoning activation and allowance are independent of grammar transport;
|
|
286
362
|
// only the response representation becomes lossless when evidence is needed.
|
|
287
363
|
// The llama-server template mapping is owned by {§llama-reasoning-request}.
|
|
@@ -292,7 +368,7 @@ export default class AiSdkProvider {
|
|
|
292
368
|
case "template": {
|
|
293
369
|
const allowance = mode === "off"
|
|
294
370
|
? 0
|
|
295
|
-
: mode === "on" ? budget : this.reasoningReserve;
|
|
371
|
+
: mode === "on" && budget !== null ? budget : this.reasoningReserve;
|
|
296
372
|
return {
|
|
297
373
|
chat_template_kwargs: { enable_thinking: on },
|
|
298
374
|
reasoning_format: preserveGrammarSentence ? "none" : "auto",
|
|
@@ -301,9 +377,9 @@ export default class AiSdkProvider {
|
|
|
301
377
|
}
|
|
302
378
|
case "think": return on ? { think: true } : {};
|
|
303
379
|
case "include_reasoning": return on ? { include_reasoning: true } : {};
|
|
304
|
-
//
|
|
305
|
-
//
|
|
306
|
-
case "effort": return mode === "on" ? { reasoning_effort:
|
|
380
|
+
// Explicit on uses the portable enabled posture or a tier derived
|
|
381
|
+
// from an explicit budget; off/adaptive omit the field.
|
|
382
|
+
case "effort": return mode === "on" ? { reasoning_effort: effortFromReasoning(this.#reasoning) } : {};
|
|
307
383
|
// Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
|
|
308
384
|
// reason-by-default model (DeepSeek V4: default 'high') reasoning.
|
|
309
385
|
// ADAPTIVE omits the field: the backend's own default posture IS the
|
|
@@ -313,19 +389,25 @@ export default class AiSdkProvider {
|
|
|
313
389
|
// efforts 400.
|
|
314
390
|
case "effort_explicit": return mode === "off"
|
|
315
391
|
? { reasoning_effort: "none" }
|
|
316
|
-
: mode === "on" ? { reasoning_effort:
|
|
392
|
+
: mode === "on" ? { reasoning_effort: effortFromReasoning(this.#reasoning) } : {};
|
|
317
393
|
// {§deepseek-reasoning-request}
|
|
318
394
|
case "thinking_effort": return mode === "off"
|
|
319
395
|
? { thinking: { type: "disabled" } }
|
|
320
396
|
: mode === "on" ? {
|
|
321
397
|
thinking: { type: "enabled" },
|
|
322
|
-
reasoning_effort: effortFromBudget(budget),
|
|
398
|
+
...(budget === null ? {} : { reasoning_effort: effortFromBudget(budget) }),
|
|
323
399
|
} : {};
|
|
324
400
|
// Anthropic compat: explicit thinking object. off → disabled; on →
|
|
325
|
-
// enabled with
|
|
401
|
+
// enabled with the explicit budget or resolved reserve; adaptive →
|
|
402
|
+
// omit (the API default).
|
|
326
403
|
case "anthropic": return mode === "off"
|
|
327
404
|
? { thinking: { type: "disabled" } }
|
|
328
|
-
: mode === "on" ? {
|
|
405
|
+
: mode === "on" ? {
|
|
406
|
+
thinking: {
|
|
407
|
+
type: "enabled",
|
|
408
|
+
budget_tokens: budget ?? this.reasoningReserve,
|
|
409
|
+
},
|
|
410
|
+
} : {};
|
|
329
411
|
case "none": return {};
|
|
330
412
|
}
|
|
331
413
|
}
|
|
@@ -390,7 +472,7 @@ export default class AiSdkProvider {
|
|
|
390
472
|
case "none": return this.#frequencyPenalty > 0 ? { frequency_penalty: this.#frequencyPenalty } : {};
|
|
391
473
|
}
|
|
392
474
|
}
|
|
393
|
-
// First-party telemetry headers ({§provider-request-authority}): forwarded only when the spec
|
|
475
|
+
// First-party telemetry headers ({§provider-request-authority} {§provider-call-kind}): forwarded only when the spec
|
|
394
476
|
// opted in (the plurnk endpoint). The gate is here, not at the call site, so
|
|
395
477
|
// attributions/client/strikes can never reach a third-party backend even if
|
|
396
478
|
// the consumer passes them to the wrong provider. Empty values emit no header
|
|
@@ -398,7 +480,7 @@ export default class AiSdkProvider {
|
|
|
398
480
|
// absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
|
|
399
481
|
// ride HTTP headers only — the packet never carries them (the model must
|
|
400
482
|
// never see strike state; engine accounting is not a metric to game).
|
|
401
|
-
#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn) {
|
|
483
|
+
#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind) {
|
|
402
484
|
if (!this.#firstPartyMetadata)
|
|
403
485
|
return {};
|
|
404
486
|
const h = {};
|
|
@@ -431,6 +513,8 @@ export default class AiSdkProvider {
|
|
|
431
513
|
h["Plurnk-Loop"] = String(loop);
|
|
432
514
|
if (turn !== undefined && Number.isInteger(turn) && turn >= 1)
|
|
433
515
|
h["Plurnk-Turn"] = String(turn);
|
|
516
|
+
if (callKind !== undefined)
|
|
517
|
+
h["Plurnk-Call-Kind"] = callKind;
|
|
434
518
|
return h;
|
|
435
519
|
}
|
|
436
520
|
// PLURNK_PROVIDERS_GBNF_DEBUG ({§gbnf-response-observation}): validate the supplied GBNF locally and fail
|
|
@@ -470,10 +554,40 @@ export default class AiSdkProvider {
|
|
|
470
554
|
out[k] = v;
|
|
471
555
|
return out;
|
|
472
556
|
}
|
|
473
|
-
|
|
557
|
+
#requestProviderOptions(workerId) {
|
|
558
|
+
const reasoningOptions = this.#reasoning.mode === "off"
|
|
559
|
+
? undefined
|
|
560
|
+
: this.#reasoningResponseProviderOptions;
|
|
561
|
+
if (this.#cacheAffinity?.target !== "provider-option")
|
|
562
|
+
return reasoningOptions;
|
|
563
|
+
const { provider, name } = this.#cacheAffinity;
|
|
564
|
+
return {
|
|
565
|
+
...reasoningOptions,
|
|
566
|
+
[provider]: {
|
|
567
|
+
...reasoningOptions?.[provider],
|
|
568
|
+
[name]: workerId,
|
|
569
|
+
},
|
|
570
|
+
};
|
|
571
|
+
}
|
|
572
|
+
#accounting(outcome, usage, evidence, status) {
|
|
573
|
+
const knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
|
|
574
|
+
const direct = this.#normalizeCost?.(evidence);
|
|
575
|
+
return validateProviderRequestAccounting({
|
|
576
|
+
provider: this.#source,
|
|
577
|
+
model: this.#model,
|
|
578
|
+
outcome,
|
|
579
|
+
...(status === undefined ? {} : { status }),
|
|
580
|
+
...(knownUsage === undefined ? {} : { usage: knownUsage }),
|
|
581
|
+
cost: resolveProviderCost(direct, this.#estimateCost(knownUsage)),
|
|
582
|
+
});
|
|
583
|
+
}
|
|
584
|
+
async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling, observeRequest, callKind }) {
|
|
474
585
|
// {§provider-interface} The worker identity is required.
|
|
475
586
|
if (workerId === undefined || workerId.length === 0)
|
|
476
587
|
throw new Error("generate: workerId is required — the worker's stable, opaque identity");
|
|
588
|
+
if (callKind !== undefined && callKind !== "emission" && callKind !== "bare") {
|
|
589
|
+
throw new Error(`generate: unsupported callKind ${JSON.stringify(callKind)}`);
|
|
590
|
+
}
|
|
477
591
|
// Reject before any wire call when already aborted
|
|
478
592
|
// ({§provider-failure-normalization}).
|
|
479
593
|
signal?.throwIfAborted();
|
|
@@ -504,73 +618,174 @@ export default class AiSdkProvider {
|
|
|
504
618
|
// reserved from caller sampling; the env flag is the single control).
|
|
505
619
|
...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
|
|
506
620
|
...this.#slotBody(workerId),
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
|
|
621
|
+
...(this.#cacheAffinity?.target === "body"
|
|
622
|
+
? { [this.#cacheAffinity.name]: workerId }
|
|
623
|
+
: {}),
|
|
511
624
|
};
|
|
512
625
|
// Per-request headers = static auth/routing + any first-party telemetry.
|
|
513
|
-
const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
|
|
514
|
-
const headers =
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
626
|
+
const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind);
|
|
627
|
+
const headers = new Headers(this.#headers);
|
|
628
|
+
if (this.#cacheAffinity?.target === "header") {
|
|
629
|
+
headers.set(this.#cacheAffinity.name, workerId);
|
|
630
|
+
}
|
|
631
|
+
for (const [name, value] of Object.entries(metaHeaders))
|
|
632
|
+
headers.set(name, value);
|
|
633
|
+
const requestHeaders = Object.fromEntries(headers.entries());
|
|
634
|
+
const accounting = [];
|
|
635
|
+
const operationTimeout = this.#operationTimeoutMs > 0
|
|
636
|
+
? AbortSignal.timeout(this.#operationTimeoutMs)
|
|
637
|
+
: undefined;
|
|
638
|
+
const operationSignal = signal === undefined
|
|
639
|
+
? operationTimeout
|
|
640
|
+
: operationTimeout === undefined
|
|
641
|
+
? signal
|
|
642
|
+
: AbortSignal.any([signal, operationTimeout]);
|
|
643
|
+
const executeRequest = async () => {
|
|
644
|
+
let settle;
|
|
645
|
+
try {
|
|
646
|
+
settle = await observeRequest?.({
|
|
647
|
+
provider: this.#source,
|
|
522
648
|
model: this.#model,
|
|
523
|
-
headers,
|
|
524
|
-
body,
|
|
525
|
-
messages,
|
|
526
|
-
signal,
|
|
527
|
-
fetch: this.#fetch,
|
|
528
|
-
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
529
|
-
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
530
|
-
retryAttempts: this.#retryAttempts,
|
|
531
|
-
streaming: this.#streaming,
|
|
532
|
-
captureRawBody: this.#rawBody,
|
|
533
|
-
})
|
|
534
|
-
: await executeAiSdkModel({
|
|
535
|
-
languageModel: this.#languageModel,
|
|
536
|
-
headers,
|
|
537
|
-
messages,
|
|
538
|
-
signal,
|
|
539
|
-
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
540
|
-
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
541
|
-
retryAttempts: this.#retryAttempts,
|
|
542
|
-
streaming: this.#streaming,
|
|
543
|
-
captureRawBody: this.#rawBody,
|
|
544
|
-
temperature: this.#tuningFloors
|
|
545
|
-
? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
|
|
546
|
-
: typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
|
|
547
|
-
topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
|
|
548
|
-
topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
|
|
549
|
-
presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
|
|
550
|
-
frequencyPenalty: typeof sampling?.frequency_penalty === "number"
|
|
551
|
-
? sampling.frequency_penalty
|
|
552
|
-
: this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
|
|
553
|
-
stopSequences: typeof sampling?.stop === "string"
|
|
554
|
-
? [sampling.stop]
|
|
555
|
-
: Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
|
|
556
|
-
? sampling.stop
|
|
557
|
-
: undefined,
|
|
558
|
-
seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
|
|
559
|
-
maxOutputTokens: maxTokens,
|
|
560
|
-
reasoning: this.#reasoning.mode === "off"
|
|
561
|
-
? "none"
|
|
562
|
-
: this.#reasoning.mode === "adaptive"
|
|
563
|
-
? "provider-default"
|
|
564
|
-
: effortFromBudget(this.#reasoning.budget),
|
|
565
649
|
});
|
|
650
|
+
}
|
|
651
|
+
catch (cause) {
|
|
652
|
+
throw new ProviderRequestObserverError(cause);
|
|
653
|
+
}
|
|
654
|
+
const settleAccounting = async (outcome, usage, evidence, status) => {
|
|
655
|
+
let requestAccounting;
|
|
656
|
+
let normalizationFailure;
|
|
657
|
+
try {
|
|
658
|
+
requestAccounting = this.#accounting(outcome, usage, evidence, status);
|
|
659
|
+
}
|
|
660
|
+
catch (cause) {
|
|
661
|
+
normalizationFailure = { cause };
|
|
662
|
+
let knownUsage;
|
|
663
|
+
try {
|
|
664
|
+
knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
|
|
665
|
+
}
|
|
666
|
+
catch {
|
|
667
|
+
knownUsage = undefined;
|
|
668
|
+
}
|
|
669
|
+
requestAccounting = validateProviderRequestAccounting({
|
|
670
|
+
provider: this.#source,
|
|
671
|
+
model: this.#model,
|
|
672
|
+
outcome,
|
|
673
|
+
...(status === undefined ? {} : { status }),
|
|
674
|
+
...(knownUsage === undefined ? {} : { usage: knownUsage }),
|
|
675
|
+
cost: {
|
|
676
|
+
kind: "unknown",
|
|
677
|
+
reason: "provider request accounting could not be normalized after physical I/O",
|
|
678
|
+
},
|
|
679
|
+
});
|
|
680
|
+
}
|
|
681
|
+
accounting.push(requestAccounting);
|
|
682
|
+
try {
|
|
683
|
+
await settle?.(requestAccounting);
|
|
684
|
+
}
|
|
685
|
+
catch (cause) {
|
|
686
|
+
throw new ProviderRequestObserverError(cause);
|
|
687
|
+
}
|
|
688
|
+
if (normalizationFailure !== undefined) {
|
|
689
|
+
throw new ProviderRequestAccountingError(normalizationFailure.cause);
|
|
690
|
+
}
|
|
691
|
+
return requestAccounting;
|
|
692
|
+
};
|
|
693
|
+
let response;
|
|
694
|
+
try {
|
|
695
|
+
response = this.#languageModel === undefined
|
|
696
|
+
? await executeOpenAICompatible({
|
|
697
|
+
url: this.#url,
|
|
698
|
+
model: this.#model,
|
|
699
|
+
headers: requestHeaders,
|
|
700
|
+
body,
|
|
701
|
+
messages,
|
|
702
|
+
signal: operationSignal,
|
|
703
|
+
fetch: this.#fetch,
|
|
704
|
+
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
705
|
+
firstContentTimeoutMs: this.#firstContentTimeoutMs,
|
|
706
|
+
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
707
|
+
streaming: this.#streaming,
|
|
708
|
+
captureRawBody: this.#rawBody,
|
|
709
|
+
})
|
|
710
|
+
: await executeAiSdkModel({
|
|
711
|
+
languageModel: this.#languageModel,
|
|
712
|
+
headers: requestHeaders,
|
|
713
|
+
providerOptions: this.#requestProviderOptions(workerId),
|
|
714
|
+
systemProviderOptions: this.#systemCacheProviderOptions,
|
|
715
|
+
messages,
|
|
716
|
+
signal: operationSignal,
|
|
717
|
+
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
718
|
+
firstContentTimeoutMs: this.#firstContentTimeoutMs,
|
|
719
|
+
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
720
|
+
streaming: this.#streaming,
|
|
721
|
+
captureRawBody: this.#rawBody,
|
|
722
|
+
temperature: this.#tuningFloors
|
|
723
|
+
? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
|
|
724
|
+
: typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
|
|
725
|
+
topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
|
|
726
|
+
topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
|
|
727
|
+
presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
|
|
728
|
+
frequencyPenalty: typeof sampling?.frequency_penalty === "number"
|
|
729
|
+
? sampling.frequency_penalty
|
|
730
|
+
: this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
|
|
731
|
+
stopSequences: typeof sampling?.stop === "string"
|
|
732
|
+
? [sampling.stop]
|
|
733
|
+
: Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
|
|
734
|
+
? sampling.stop
|
|
735
|
+
: undefined,
|
|
736
|
+
seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
|
|
737
|
+
maxOutputTokens: maxTokens,
|
|
738
|
+
reasoning: this.#reasoning.mode === "off"
|
|
739
|
+
? "none"
|
|
740
|
+
: this.#reasoning.mode === "adaptive"
|
|
741
|
+
? "provider-default"
|
|
742
|
+
: effortFromReasoning(this.#reasoning),
|
|
743
|
+
});
|
|
744
|
+
}
|
|
745
|
+
catch (error) {
|
|
746
|
+
const failure = transportFailureEvidence(error);
|
|
747
|
+
await settleAccounting("error", failure.usage, failure.chargeEvidence, failure.status);
|
|
748
|
+
throw error;
|
|
749
|
+
}
|
|
750
|
+
await settleAccounting("response", response.usage, response.chargeEvidence);
|
|
751
|
+
return response;
|
|
752
|
+
};
|
|
753
|
+
let raw;
|
|
754
|
+
try {
|
|
755
|
+
const { retry } = prepareRetries({
|
|
756
|
+
maxRetries: this.#retryAttempts,
|
|
757
|
+
abortSignal: operationSignal,
|
|
758
|
+
});
|
|
759
|
+
raw = await retry(executeRequest);
|
|
566
760
|
}
|
|
567
761
|
catch (err) {
|
|
762
|
+
if (err instanceof ProviderRequestObserverError
|
|
763
|
+
|| err instanceof ProviderRequestAccountingError)
|
|
764
|
+
throw err.cause;
|
|
568
765
|
if (signal?.aborted)
|
|
569
766
|
throw err;
|
|
767
|
+
if (operationTimeout?.aborted) {
|
|
768
|
+
const timeout = new ProviderTimeoutError("operation", this.#operationTimeoutMs, err);
|
|
769
|
+
throw new ProviderError(this.#source, "deadline_exceeded", timeout.message, {
|
|
770
|
+
status: 504,
|
|
771
|
+
cause: timeout,
|
|
772
|
+
retryable: false,
|
|
773
|
+
extensions: {
|
|
774
|
+
timeoutPhase: timeout.phase,
|
|
775
|
+
timeoutMs: timeout.timeoutMs,
|
|
776
|
+
},
|
|
777
|
+
accounting,
|
|
778
|
+
});
|
|
779
|
+
}
|
|
570
780
|
const pe = toProviderError(err, this.#source, this.#errorDetailLimit);
|
|
571
781
|
if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
|
|
572
|
-
throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, {
|
|
782
|
+
throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, {
|
|
783
|
+
status: pe.status,
|
|
784
|
+
cause: err,
|
|
785
|
+
accounting,
|
|
786
|
+
});
|
|
573
787
|
}
|
|
788
|
+
pe.prependAccounting(accounting);
|
|
574
789
|
throw pe;
|
|
575
790
|
}
|
|
576
791
|
// llama-server --special renders EOG tokens as text, so a turn ending
|
|
@@ -585,10 +800,10 @@ export default class AiSdkProvider {
|
|
|
585
800
|
const projectedReasoning = preserveGrammarSentence && !raw.reasoningProjected
|
|
586
801
|
? projectTemplateReasoning(raw.content)
|
|
587
802
|
: projectTaggedReasoning(raw.content, raw.reasoning, this.#reasoningResponseStyle);
|
|
588
|
-
// Preserve the exact
|
|
589
|
-
//
|
|
590
|
-
//
|
|
591
|
-
// supply independent
|
|
803
|
+
// Preserve the exact pre-projection response. Constrained template turns
|
|
804
|
+
// request `reasoning_format: "none"`, so even an empty channel and any
|
|
805
|
+
// template-provided opener remain observable. An unexpectedly projected
|
|
806
|
+
// response cannot supply independent evidence.
|
|
592
807
|
let grammarEvidence;
|
|
593
808
|
if (wantGrammar) {
|
|
594
809
|
if (preserveGrammarSentence) {
|
|
@@ -618,11 +833,12 @@ export default class AiSdkProvider {
|
|
|
618
833
|
if (projectedReasoning.projected) {
|
|
619
834
|
raw.content = projectedReasoning.content;
|
|
620
835
|
raw.reasoning = projectedReasoning.reasoning;
|
|
621
|
-
raw.usage = attributeUnitemizedReasoning(raw.usage, raw.reasoning, raw.content);
|
|
622
836
|
}
|
|
623
837
|
let notices;
|
|
624
838
|
const usage = raw.usage;
|
|
625
|
-
if (sendGrammar !== undefined
|
|
839
|
+
if (sendGrammar !== undefined
|
|
840
|
+
&& this.tokenize !== undefined
|
|
841
|
+
&& usage?.outputTokens !== undefined) {
|
|
626
842
|
// Channel-escape detector: completion tokens
|
|
627
843
|
// billed far beyond every visible channel mean the decode ESCAPED into
|
|
628
844
|
// a server-discarded reasoning block mid-emission. This diagnostic
|
|
@@ -633,12 +849,12 @@ export default class AiSdkProvider {
|
|
|
633
849
|
this.tokenize(raw.reasoning),
|
|
634
850
|
]);
|
|
635
851
|
const visible = contentTokens.length + reasoningTokens.length;
|
|
636
|
-
if (usage.
|
|
852
|
+
if (usage.outputTokens > visible + 64) {
|
|
637
853
|
(notices ??= []).push({
|
|
638
854
|
source: this.#source,
|
|
639
855
|
kind: "grammar_unenforced",
|
|
640
856
|
level: "warn",
|
|
641
|
-
message: `decode escaped the grammar: ${usage.
|
|
857
|
+
message: `decode escaped the grammar: ${usage.outputTokens} output tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
|
|
642
858
|
position: [...raw.content].length,
|
|
643
859
|
});
|
|
644
860
|
}
|
|
@@ -658,17 +874,12 @@ export default class AiSdkProvider {
|
|
|
658
874
|
...(raw.reasoningEncrypted.length > 0
|
|
659
875
|
? { reasoningEncrypted: raw.reasoningEncrypted }
|
|
660
876
|
: {}),
|
|
661
|
-
usage,
|
|
662
877
|
model: raw.model,
|
|
663
878
|
...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
|
|
664
879
|
};
|
|
665
|
-
const normalizedCharge = this.#normalizeCharge?.(raw.chargeEvidence);
|
|
666
|
-
const charge = normalizedCharge === undefined
|
|
667
|
-
? undefined
|
|
668
|
-
: validateAuthoritativeCharge(normalizedCharge);
|
|
669
880
|
const evidence = {
|
|
670
881
|
assistantRaw: raw,
|
|
671
|
-
|
|
882
|
+
accounting,
|
|
672
883
|
...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
|
|
673
884
|
...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
|
|
674
885
|
...(meta !== undefined ? { meta } : {}),
|
|
@@ -681,6 +892,7 @@ export default class AiSdkProvider {
|
|
|
681
892
|
};
|
|
682
893
|
throw new ProviderError(this.#source, "resource_interrupted", "The provider interrupted generation because inference resources were unavailable.", {
|
|
683
894
|
attempt,
|
|
895
|
+
accounting,
|
|
684
896
|
extensions: {
|
|
685
897
|
stage: "provider-response",
|
|
686
898
|
finishReason: "resource_interrupted",
|