@plurnk/plurnk-providers 1.5.0 → 1.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +41 -34
- package/README.md +15 -0
- package/SPEC.md +242 -89
- package/dist/AiSdkProvider.d.ts +33 -33
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +442 -133
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +10 -11
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +87 -25
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +9 -24
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +86 -25
- package/dist/Pool.js.map +1 -1
- package/dist/accounting.d.ts +5 -2
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js +100 -16
- package/dist/accounting.js.map +1 -1
- package/dist/accountingPublic.d.ts +5 -0
- package/dist/accountingPublic.d.ts.map +1 -0
- package/dist/accountingPublic.js +3 -0
- package/dist/accountingPublic.js.map +1 -0
- package/dist/aiSdkTransport.d.ts +9 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +160 -62
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/capacity.d.ts +26 -0
- package/dist/capacity.d.ts.map +1 -0
- package/dist/capacity.js +90 -0
- package/dist/capacity.js.map +1 -0
- package/dist/catalogProvider.d.ts +8 -3
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +45 -41
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +26 -12
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +10 -10
- package/dist/cost.d.ts.map +1 -1
- package/dist/cost.js +90 -42
- package/dist/cost.js.map +1 -1
- package/dist/env.d.ts +13 -11
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +83 -46
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +17 -3
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +91 -8
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +7 -6
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5 -3
- package/dist/index.js.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/promptTokens.d.ts.map +1 -1
- package/dist/promptTokens.js +7 -4
- package/dist/promptTokens.js.map +1 -1
- package/dist/sdkModels.d.ts +7 -2
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +43 -13
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +55 -33
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +22 -5
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +169 -83
- package/dist/usage.js.map +1 -1
- package/package.json +18 -7
- package/src/AiSdkProvider.test.ts +964 -206
- package/src/AiSdkProvider.ts +545 -155
- package/src/Mock.test.ts +69 -30
- package/src/Mock.ts +99 -29
- package/src/Pool.test.ts +90 -19
- package/src/Pool.ts +96 -27
- package/src/ProviderRegistry.test.ts +16 -11
- package/src/accounting.test.ts +58 -22
- package/src/accounting.ts +119 -18
- package/src/accountingPublic.ts +9 -0
- package/src/aiSdkTransport.test.ts +42 -49
- package/src/aiSdkTransport.ts +174 -62
- package/src/boundaries.test.ts +2 -0
- package/src/capacity.test.ts +92 -0
- package/src/capacity.ts +140 -0
- package/src/catalogProvider.test.ts +339 -30
- package/src/catalogProvider.ts +65 -47
- package/src/compatibleProvider.test.ts +7 -5
- package/src/compatibleProvider.ts +29 -13
- package/src/cost.test.ts +86 -36
- package/src/cost.ts +111 -50
- package/src/defaults.test.ts +13 -3
- package/src/env.test.ts +103 -25
- package/src/env.ts +153 -65
- package/src/errors.test.ts +80 -2
- package/src/errors.ts +107 -8
- package/src/index.ts +26 -7
- package/src/ollama.test.ts +5 -3
- package/src/ollama.ts +3 -3
- package/src/promptTokens.ts +8 -5
- package/src/sdkModels.test.ts +77 -8
- package/src/sdkModels.ts +51 -15
- package/src/types.ts +112 -51
- package/src/usage.test.ts +112 -116
- package/src/usage.ts +214 -93
package/dist/AiSdkProvider.js
CHANGED
|
@@ -5,13 +5,29 @@
|
|
|
5
5
|
// Composition, not inheritance: an official AI SDK language model supplies the
|
|
6
6
|
// ordinary vendor protocol. The compatible URL path remains only for PLURNK
|
|
7
7
|
// extensions and local endpoint probes the SDK cannot represent.
|
|
8
|
-
import {
|
|
9
|
-
import {
|
|
10
|
-
import {
|
|
8
|
+
import { MAX_PROVIDER_TIMEOUT_MS } from "./env.js";
|
|
9
|
+
import { executeAiSdkModel, executeOpenAICompatible, transportFailureEvidence, } from "./aiSdkTransport.js";
|
|
10
|
+
import { prepareRetries } from "ai/internal";
|
|
11
|
+
import { toProviderError, ProviderError, ProviderTimeoutError } from "./errors.js";
|
|
11
12
|
import { validateGbnf } from "@plurnk/gbnf";
|
|
12
13
|
import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.js";
|
|
13
14
|
import { emitWarningOnce } from "./warnings.js";
|
|
14
|
-
import {
|
|
15
|
+
import { resolveProviderCost } from "./cost.js";
|
|
16
|
+
import { validateProviderRequestAccounting } from "./accounting.js";
|
|
17
|
+
import { validateProviderUsage } from "./usage.js";
|
|
18
|
+
import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.js";
|
|
19
|
+
class ProviderRequestObserverError extends Error {
|
|
20
|
+
constructor(cause) {
|
|
21
|
+
super("provider request accounting could not be durably settled", { cause });
|
|
22
|
+
this.name = "ProviderRequestObserverError";
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
class ProviderRequestAccountingError extends Error {
|
|
26
|
+
constructor(cause) {
|
|
27
|
+
super("provider request accounting could not be normalized", { cause });
|
|
28
|
+
this.name = "ProviderRequestAccountingError";
|
|
29
|
+
}
|
|
30
|
+
}
|
|
15
31
|
// Drop trailing occurrences of a server-rendered EOG marker. llama-server
|
|
16
32
|
// under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
|
|
17
33
|
// trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
|
|
@@ -52,11 +68,20 @@ const projectLeadingReasoning = (content, structuredReasoning, opening, closing)
|
|
|
52
68
|
const projectTaggedReasoning = (content, structuredReasoning, style) => style === "think-tags"
|
|
53
69
|
? projectLeadingReasoning(content, structuredReasoning, "<think>", "</think>")
|
|
54
70
|
: { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
|
|
55
|
-
// llama-server's template reasoning parser can project
|
|
56
|
-
// of the OpenAI-compatible response. Grammar evidence
|
|
57
|
-
// that lossy projection, so constrained template turns
|
|
58
|
-
// split the observed enclosure here.
|
|
59
|
-
const projectTemplateReasoning = (content) =>
|
|
71
|
+
// llama-server's template reasoning parser can project either supported leading
|
|
72
|
+
// reasoning envelope out of the OpenAI-compatible response. Grammar evidence
|
|
73
|
+
// needs the sentence before that lossy projection, so constrained template turns
|
|
74
|
+
// request it verbatim and split the observed enclosure here.
|
|
75
|
+
const projectTemplateReasoning = (content) => {
|
|
76
|
+
for (const [opening, closing] of [
|
|
77
|
+
["<|channel>thought\n", "<channel|>"],
|
|
78
|
+
["<think>\n", "</think>"],
|
|
79
|
+
]) {
|
|
80
|
+
if (content.startsWith(opening))
|
|
81
|
+
return projectLeadingReasoning(content, "", opening, closing);
|
|
82
|
+
}
|
|
83
|
+
return { content, reasoning: "", projected: false, contentStart: 0 };
|
|
84
|
+
};
|
|
60
85
|
// Shared budget→effort breakpoints (xai and google had identical copies).
|
|
61
86
|
export const effortFromBudget = (budget) => {
|
|
62
87
|
if (budget <= 1000)
|
|
@@ -65,6 +90,11 @@ export const effortFromBudget = (budget) => {
|
|
|
65
90
|
return "medium";
|
|
66
91
|
return "high";
|
|
67
92
|
};
|
|
93
|
+
// AI SDK's portable reasoning control has no boolean-enabled value. `medium`
|
|
94
|
+
// is the neutral activation projection for an explicit, unqualified `on`; it
|
|
95
|
+
// changes no PLURNK output budget. An operator reasoning subset, when present, remains
|
|
96
|
+
// the only input to the existing magnitude-to-tier projection.
|
|
97
|
+
const effortFromReasoning = (reasoning) => reasoning.budget === null ? "medium" : effortFromBudget(reasoning.budget);
|
|
68
98
|
// Body keys the provider owns — a caller's `sampling` passthrough may not set
|
|
69
99
|
// these. Two families:
|
|
70
100
|
// transport/managed — grammar transport, the stream/JSON choice, slot pinning,
|
|
@@ -73,7 +103,7 @@ export const effortFromBudget = (budget) => {
|
|
|
73
103
|
// response; n>1 = paid, dropped output), the tool-calling family (tools-in-
|
|
74
104
|
// body doctrine, §2: native tool_calls return null content = a broken turn),
|
|
75
105
|
// modalities/audio (text-only contract), prediction (decode semantics, not
|
|
76
|
-
// sampling), and the token caps (the envelope is the managed
|
|
106
|
+
// sampling), and the token caps (the envelope is the managed maxOutputTokens —
|
|
77
107
|
// sampling must not bypass the consumer's cap).
|
|
78
108
|
// Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
|
|
79
109
|
// platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
|
|
@@ -91,6 +121,8 @@ export default class AiSdkProvider {
|
|
|
91
121
|
#url;
|
|
92
122
|
#languageModel;
|
|
93
123
|
#fetchTimeoutMs;
|
|
124
|
+
#operationTimeoutMs;
|
|
125
|
+
#firstContentTimeoutMs;
|
|
94
126
|
#streamIdleTimeoutMs;
|
|
95
127
|
#headers;
|
|
96
128
|
#fetch;
|
|
@@ -98,6 +130,11 @@ export default class AiSdkProvider {
|
|
|
98
130
|
#apiKeyRejectedMessage;
|
|
99
131
|
#eosText;
|
|
100
132
|
#contextWindow;
|
|
133
|
+
#maxInputTokens;
|
|
134
|
+
#maxOutputTokens;
|
|
135
|
+
#outputBudget;
|
|
136
|
+
#reasoningBudget;
|
|
137
|
+
#additiveReasoningProvider;
|
|
101
138
|
#reasoning;
|
|
102
139
|
#temperature;
|
|
103
140
|
#repeatPenalty;
|
|
@@ -110,12 +147,13 @@ export default class AiSdkProvider {
|
|
|
110
147
|
#reasoningResponseStyle;
|
|
111
148
|
#countPromptTokens;
|
|
112
149
|
#promptTokensUrl;
|
|
113
|
-
#
|
|
114
|
-
#
|
|
115
|
-
#normalizeCharge;
|
|
150
|
+
#estimateCost;
|
|
151
|
+
#normalizeCost;
|
|
116
152
|
#source;
|
|
117
153
|
#grammarStyle;
|
|
118
|
-
#
|
|
154
|
+
#cacheAffinity;
|
|
155
|
+
#systemCacheProviderOptions;
|
|
156
|
+
#reasoningResponseProviderOptions;
|
|
119
157
|
#serviceTier;
|
|
120
158
|
#gbnfDebug;
|
|
121
159
|
#streaming;
|
|
@@ -125,12 +163,10 @@ export default class AiSdkProvider {
|
|
|
125
163
|
#retryAttempts;
|
|
126
164
|
#errorDetailLimit;
|
|
127
165
|
#topLogprobs;
|
|
128
|
-
#reasoningReserve;
|
|
129
|
-
#completionReserve;
|
|
130
166
|
#tuningFloors;
|
|
131
167
|
#rawBody;
|
|
132
168
|
#servedModel;
|
|
133
|
-
#
|
|
169
|
+
#requiresOutputBudget;
|
|
134
170
|
attributions;
|
|
135
171
|
// Optional capability ({§provider-local-capabilities}): exact tokenization served by the backend's
|
|
136
172
|
// own vocab. Assigned in the constructor ONLY when the config carries a
|
|
@@ -145,11 +181,28 @@ export default class AiSdkProvider {
|
|
|
145
181
|
if ((this.#url === undefined) === (this.#languageModel === undefined)) {
|
|
146
182
|
throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
|
|
147
183
|
}
|
|
184
|
+
for (const [name, value] of [
|
|
185
|
+
["fetchTimeoutMs", config.fetchTimeoutMs],
|
|
186
|
+
["operationTimeoutMs", config.operationTimeoutMs],
|
|
187
|
+
["firstContentTimeoutMs", config.firstContentTimeoutMs],
|
|
188
|
+
["streamIdleTimeoutMs", config.streamIdleTimeoutMs],
|
|
189
|
+
]) {
|
|
190
|
+
if (value !== undefined && (!Number.isInteger(value) || value < 0 || value > MAX_PROVIDER_TIMEOUT_MS)) {
|
|
191
|
+
throw new Error(`${config.source ?? "provider"}: ${name} must be an integer from 0 through ${MAX_PROVIDER_TIMEOUT_MS} milliseconds`);
|
|
192
|
+
}
|
|
193
|
+
}
|
|
148
194
|
this.#fetchTimeoutMs = config.fetchTimeoutMs;
|
|
195
|
+
this.#operationTimeoutMs = config.operationTimeoutMs;
|
|
196
|
+
this.#firstContentTimeoutMs = config.firstContentTimeoutMs;
|
|
149
197
|
this.#streamIdleTimeoutMs = config.streamIdleTimeoutMs;
|
|
150
198
|
this.#headers = config.headers ?? {};
|
|
151
199
|
this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
|
|
152
200
|
this.#contextWindow = config.contextWindow ?? null;
|
|
201
|
+
this.#maxInputTokens = config.maxInputTokens ?? null;
|
|
202
|
+
this.#maxOutputTokens = config.maxOutputTokens ?? null;
|
|
203
|
+
this.#outputBudget = config.outputBudget ?? null;
|
|
204
|
+
this.#reasoningBudget = config.reasoningBudget ?? null;
|
|
205
|
+
this.#additiveReasoningProvider = config.additiveReasoningProvider;
|
|
153
206
|
this.#reasoning = config.reasoning;
|
|
154
207
|
// Loud guard: an out-of-date consumer (stale plugin dist) omitting the
|
|
155
208
|
// required tuning fields must fail at construction, not silently send
|
|
@@ -173,12 +226,33 @@ export default class AiSdkProvider {
|
|
|
173
226
|
}
|
|
174
227
|
this.#countPromptTokens = config.countPromptTokens ?? ((messages) => estimatePromptTokens(messages));
|
|
175
228
|
this.#promptTokensUrl = config.promptTokensUrl;
|
|
176
|
-
this.#
|
|
177
|
-
|
|
178
|
-
|
|
229
|
+
this.#estimateCost = config.estimateCost
|
|
230
|
+
?? (() => ({
|
|
231
|
+
kind: "unknown",
|
|
232
|
+
reason: "the request reported no direct cost and no model rate is configured",
|
|
233
|
+
}));
|
|
234
|
+
this.#normalizeCost = config.normalizeCost;
|
|
179
235
|
this.#source = config.source ?? "provider";
|
|
180
236
|
this.#grammarStyle = config.grammarStyle ?? "none";
|
|
181
|
-
this.#
|
|
237
|
+
this.#cacheAffinity = config.cacheAffinity;
|
|
238
|
+
this.#systemCacheProviderOptions = config.systemCacheProviderOptions;
|
|
239
|
+
this.#reasoningResponseProviderOptions = config.reasoningResponseProviderOptions;
|
|
240
|
+
if (this.#languageModel === undefined && this.#cacheAffinity?.target === "provider-option") {
|
|
241
|
+
throw new Error(`${this.#source}: native provider-option cache affinity requires an AI SDK model`);
|
|
242
|
+
}
|
|
243
|
+
if (this.#languageModel !== undefined && this.#cacheAffinity?.target === "body") {
|
|
244
|
+
throw new Error(`${this.#source}: body cache affinity requires an OpenAI-compatible URL`);
|
|
245
|
+
}
|
|
246
|
+
if (this.#languageModel === undefined && this.#systemCacheProviderOptions !== undefined) {
|
|
247
|
+
throw new Error(`${this.#source}: system cache provider options require an AI SDK model`);
|
|
248
|
+
}
|
|
249
|
+
if (this.#languageModel === undefined && this.#reasoningResponseProviderOptions !== undefined) {
|
|
250
|
+
throw new Error(`${this.#source}: reasoning response provider options require an AI SDK model`);
|
|
251
|
+
}
|
|
252
|
+
if (this.#cacheAffinity?.target === "provider-option"
|
|
253
|
+
&& Object.hasOwn(this.#reasoningResponseProviderOptions?.[this.#cacheAffinity.provider] ?? {}, this.#cacheAffinity.name)) {
|
|
254
|
+
throw new Error(`${this.#source}: reasoning response options conflict with cache-affinity option ${this.#cacheAffinity.provider}.${this.#cacheAffinity.name}`);
|
|
255
|
+
}
|
|
182
256
|
this.#serviceTier = config.serviceTier;
|
|
183
257
|
this.#gbnfDebug = config.gbnfDebug ?? false;
|
|
184
258
|
this.#streaming = config.streaming ?? true;
|
|
@@ -189,18 +263,42 @@ export default class AiSdkProvider {
|
|
|
189
263
|
this.#supportsSlotPinning = config.supportsSlotPinning ?? false;
|
|
190
264
|
this.#slotCount = config.slotCount ?? null;
|
|
191
265
|
this.#topLogprobs = config.topLogprobs ?? null;
|
|
192
|
-
this.#reasoningReserve = config.reasoningReserve;
|
|
193
|
-
this.#completionReserve = config.completionReserve;
|
|
194
266
|
this.#tuningFloors = config.tuningFloors ?? true;
|
|
195
267
|
this.#rawBody = config.rawBody ?? false;
|
|
196
268
|
this.#servedModel = config.servedModel;
|
|
197
|
-
this.#
|
|
198
|
-
const
|
|
199
|
-
|
|
269
|
+
this.#requiresOutputBudget = config.requiresOutputBudget;
|
|
270
|
+
for (const [name, value] of [
|
|
271
|
+
["contextWindow", this.#contextWindow],
|
|
272
|
+
["maxInputTokens", this.#maxInputTokens],
|
|
273
|
+
["maxOutputTokens", this.#maxOutputTokens],
|
|
274
|
+
["outputBudget", this.#outputBudget],
|
|
275
|
+
["reasoningBudget", this.#reasoningBudget],
|
|
276
|
+
]) {
|
|
277
|
+
if (value !== null && (!Number.isSafeInteger(value) || value <= 0)) {
|
|
278
|
+
throw new Error(`${this.#source}: ${name} must be a positive safe integer or null`);
|
|
279
|
+
}
|
|
280
|
+
}
|
|
281
|
+
if (this.#reasoningBudget !== null
|
|
282
|
+
&& this.#outputBudget !== null
|
|
283
|
+
&& this.#reasoningBudget >= this.#outputBudget) {
|
|
284
|
+
throw new Error(`${this.#source}: reasoningBudget must be smaller than the total outputBudget`);
|
|
285
|
+
}
|
|
286
|
+
if (this.#reasoning.budget !== this.#reasoningBudget) {
|
|
287
|
+
throw new Error(`${this.#source}: reasoning intent and generation envelope disagree on reasoningBudget`);
|
|
288
|
+
}
|
|
289
|
+
if (this.#reasoningStyle === "anthropic"
|
|
290
|
+
&& this.#reasoning.mode === "on"
|
|
291
|
+
&& this.#reasoning.budget === null
|
|
292
|
+
&& this.#reasoningBudget === null) {
|
|
293
|
+
throw new Error(`${this.#source}: explicit Anthropic reasoning requires PLURNK_PROVIDERS_REASONING_BUDGET`);
|
|
294
|
+
}
|
|
295
|
+
if (this.#additiveReasoningProvider !== undefined
|
|
200
296
|
&& this.#reasoning.mode === "on"
|
|
201
|
-
&&
|
|
202
|
-
|
|
203
|
-
|
|
297
|
+
&& this.#reasoningBudget === null) {
|
|
298
|
+
throw new Error(`${this.#source}: explicit ${this.#additiveReasoningProvider} reasoning requires PLURNK_PROVIDERS_REASONING_BUDGET so the total output budget remains bounded`);
|
|
299
|
+
}
|
|
300
|
+
if (this.#requiresOutputBudget === true && this.#outputBudget === null) {
|
|
301
|
+
throw new Error(`${this.#source}: this backend requires a resolved PLURNK_PROVIDERS_OUTPUT_BUDGET`);
|
|
204
302
|
}
|
|
205
303
|
const { tokenizeUrl } = config;
|
|
206
304
|
if (tokenizeUrl !== undefined) {
|
|
@@ -209,7 +307,9 @@ export default class AiSdkProvider {
|
|
|
209
307
|
method: "POST",
|
|
210
308
|
headers: { "Content-Type": "application/json", ...this.#headers },
|
|
211
309
|
body: JSON.stringify({ content: text }),
|
|
212
|
-
|
|
310
|
+
...(this.#fetchTimeoutMs > 0
|
|
311
|
+
? { signal: AbortSignal.timeout(this.#fetchTimeoutMs) }
|
|
312
|
+
: {}),
|
|
213
313
|
});
|
|
214
314
|
if (!res.ok)
|
|
215
315
|
throw new Error(`${this.#source}: tokenize endpoint returned ${res.status}`);
|
|
@@ -222,22 +322,22 @@ export default class AiSdkProvider {
|
|
|
222
322
|
}
|
|
223
323
|
}
|
|
224
324
|
get contextWindow() { return this.#contextWindow; }
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
325
|
+
get maxInputTokens() { return this.#maxInputTokens; }
|
|
326
|
+
get maxOutputTokens() { return this.#maxOutputTokens; }
|
|
327
|
+
get outputBudget() { return this.#outputBudget; }
|
|
328
|
+
get reasoningBudget() { return this.#reasoningBudget; }
|
|
329
|
+
get inputCapacity() {
|
|
330
|
+
return effectiveInputCapacity({
|
|
331
|
+
contextWindow: this.#contextWindow,
|
|
332
|
+
maxInputTokens: this.#maxInputTokens,
|
|
333
|
+
outputBudget: this.#outputBudget,
|
|
334
|
+
});
|
|
233
335
|
}
|
|
234
|
-
get reasoningReserve() { return this.#resolveReserve(this.#reasoningReserve); }
|
|
235
|
-
get completionReserve() { return this.#resolveReserve(this.#completionReserve); }
|
|
236
336
|
get model() { return this.#model; }
|
|
237
337
|
// Backend's self-reported served id; undefined when unprobed/unknown.
|
|
238
338
|
get servedModel() { return this.#servedModel; }
|
|
239
339
|
// Resolved "decodes unbounded without a cap" fact; undefined = no claim.
|
|
240
|
-
get
|
|
340
|
+
get requiresOutputBudget() { return this.#requiresOutputBudget; }
|
|
241
341
|
// Resolved capability: will a transported grammar actually constrain
|
|
242
342
|
// this backend's decode? Introspectable so a consumer can verify the rails
|
|
243
343
|
// are LIVE without spending a generation on a forcing-grammar probe.
|
|
@@ -248,7 +348,14 @@ export default class AiSdkProvider {
|
|
|
248
348
|
}
|
|
249
349
|
signal?.throwIfAborted();
|
|
250
350
|
try {
|
|
251
|
-
const timeout =
|
|
351
|
+
const timeout = this.#fetchTimeoutMs > 0
|
|
352
|
+
? AbortSignal.timeout(this.#fetchTimeoutMs)
|
|
353
|
+
: undefined;
|
|
354
|
+
const requestSignal = signal === undefined
|
|
355
|
+
? timeout
|
|
356
|
+
: timeout === undefined
|
|
357
|
+
? signal
|
|
358
|
+
: AbortSignal.any([signal, timeout]);
|
|
252
359
|
const response = await this.#fetch(this.#promptTokensUrl, {
|
|
253
360
|
method: "POST",
|
|
254
361
|
headers: { "Content-Type": "application/json", ...this.#headers },
|
|
@@ -257,7 +364,7 @@ export default class AiSdkProvider {
|
|
|
257
364
|
messages,
|
|
258
365
|
...this.#reasoningBody(),
|
|
259
366
|
}),
|
|
260
|
-
|
|
367
|
+
...(requestSignal === undefined ? {} : { signal: requestSignal }),
|
|
261
368
|
});
|
|
262
369
|
if (!response.ok) {
|
|
263
370
|
return estimatePromptTokens(messages, `llama-server input-token endpoint returned HTTP ${response.status}`);
|
|
@@ -277,22 +384,38 @@ export default class AiSdkProvider {
|
|
|
277
384
|
return estimatePromptTokens(messages, `llama-server input-token measurement failed: ${cause instanceof Error ? cause.message : String(cause)}`);
|
|
278
385
|
}
|
|
279
386
|
}
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
387
|
+
async assessRequestCapacity(messages, maxOutputTokens, signal) {
|
|
388
|
+
const outputBudget = effectiveOutputBudget({
|
|
389
|
+
requested: maxOutputTokens,
|
|
390
|
+
configured: this.#outputBudget,
|
|
391
|
+
maxOutputTokens: this.#maxOutputTokens,
|
|
392
|
+
contextWindow: this.#contextWindow,
|
|
393
|
+
});
|
|
394
|
+
const reasoningBudget = effectiveReasoningBudget({
|
|
395
|
+
configured: this.#reasoningBudget,
|
|
396
|
+
outputBudget,
|
|
397
|
+
});
|
|
398
|
+
return assessRequestCapacity({
|
|
399
|
+
contextWindow: this.#contextWindow,
|
|
400
|
+
maxInputTokens: this.#maxInputTokens,
|
|
401
|
+
maxOutputTokens: this.#maxOutputTokens,
|
|
402
|
+
outputBudget,
|
|
403
|
+
reasoningBudget,
|
|
404
|
+
measurement: await this.countPromptTokens(messages, signal),
|
|
405
|
+
});
|
|
284
406
|
}
|
|
285
407
|
// Reasoning activation and allowance are independent of grammar transport;
|
|
286
408
|
// only the response representation becomes lossless when evidence is needed.
|
|
287
409
|
// The llama-server template mapping is owned by {§llama-reasoning-request}.
|
|
288
|
-
#reasoningBody(preserveGrammarSentence = false) {
|
|
289
|
-
const { mode
|
|
410
|
+
#reasoningBody(preserveGrammarSentence = false, reasoningBudget = this.#reasoningBudget) {
|
|
411
|
+
const { mode } = this.#reasoning;
|
|
412
|
+
const budget = reasoningBudget;
|
|
290
413
|
const on = mode !== "off";
|
|
291
414
|
switch (this.#reasoningStyle) {
|
|
292
415
|
case "template": {
|
|
293
416
|
const allowance = mode === "off"
|
|
294
417
|
? 0
|
|
295
|
-
: mode === "on" ? budget : this
|
|
418
|
+
: mode === "on" && budget !== null ? budget : this.#reasoningBudget;
|
|
296
419
|
return {
|
|
297
420
|
chat_template_kwargs: { enable_thinking: on },
|
|
298
421
|
reasoning_format: preserveGrammarSentence ? "none" : "auto",
|
|
@@ -301,9 +424,9 @@ export default class AiSdkProvider {
|
|
|
301
424
|
}
|
|
302
425
|
case "think": return on ? { think: true } : {};
|
|
303
426
|
case "include_reasoning": return on ? { include_reasoning: true } : {};
|
|
304
|
-
//
|
|
305
|
-
//
|
|
306
|
-
case "effort": return mode === "on" ? { reasoning_effort:
|
|
427
|
+
// Explicit on uses the portable enabled posture or a tier derived
|
|
428
|
+
// from an explicit budget; off/adaptive omit the field.
|
|
429
|
+
case "effort": return mode === "on" ? { reasoning_effort: effortFromReasoning({ mode, budget }) } : {};
|
|
307
430
|
// Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
|
|
308
431
|
// reason-by-default model (DeepSeek V4: default 'high') reasoning.
|
|
309
432
|
// ADAPTIVE omits the field: the backend's own default posture IS the
|
|
@@ -313,19 +436,25 @@ export default class AiSdkProvider {
|
|
|
313
436
|
// efforts 400.
|
|
314
437
|
case "effort_explicit": return mode === "off"
|
|
315
438
|
? { reasoning_effort: "none" }
|
|
316
|
-
: mode === "on" ? { reasoning_effort:
|
|
439
|
+
: mode === "on" ? { reasoning_effort: effortFromReasoning({ mode, budget }) } : {};
|
|
317
440
|
// {§deepseek-reasoning-request}
|
|
318
441
|
case "thinking_effort": return mode === "off"
|
|
319
442
|
? { thinking: { type: "disabled" } }
|
|
320
443
|
: mode === "on" ? {
|
|
321
444
|
thinking: { type: "enabled" },
|
|
322
|
-
reasoning_effort: effortFromBudget(budget),
|
|
445
|
+
...(budget === null ? {} : { reasoning_effort: effortFromBudget(budget) }),
|
|
323
446
|
} : {};
|
|
324
447
|
// Anthropic compat: explicit thinking object. off → disabled; on →
|
|
325
|
-
// enabled with
|
|
448
|
+
// enabled with the explicit reasoning subset; adaptive →
|
|
449
|
+
// omit (the API default).
|
|
326
450
|
case "anthropic": return mode === "off"
|
|
327
451
|
? { thinking: { type: "disabled" } }
|
|
328
|
-
: mode === "on" ? {
|
|
452
|
+
: mode === "on" ? {
|
|
453
|
+
thinking: {
|
|
454
|
+
type: "enabled",
|
|
455
|
+
budget_tokens: budget,
|
|
456
|
+
},
|
|
457
|
+
} : {};
|
|
329
458
|
case "none": return {};
|
|
330
459
|
}
|
|
331
460
|
}
|
|
@@ -390,7 +519,7 @@ export default class AiSdkProvider {
|
|
|
390
519
|
case "none": return this.#frequencyPenalty > 0 ? { frequency_penalty: this.#frequencyPenalty } : {};
|
|
391
520
|
}
|
|
392
521
|
}
|
|
393
|
-
// First-party telemetry headers ({§provider-request-authority}): forwarded only when the spec
|
|
522
|
+
// First-party telemetry headers ({§provider-request-authority} {§provider-call-kind}): forwarded only when the spec
|
|
394
523
|
// opted in (the plurnk endpoint). The gate is here, not at the call site, so
|
|
395
524
|
// attributions/client/strikes can never reach a third-party backend even if
|
|
396
525
|
// the consumer passes them to the wrong provider. Empty values emit no header
|
|
@@ -398,7 +527,7 @@ export default class AiSdkProvider {
|
|
|
398
527
|
// absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
|
|
399
528
|
// ride HTTP headers only — the packet never carries them (the model must
|
|
400
529
|
// never see strike state; engine accounting is not a metric to game).
|
|
401
|
-
#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn) {
|
|
530
|
+
#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind) {
|
|
402
531
|
if (!this.#firstPartyMetadata)
|
|
403
532
|
return {};
|
|
404
533
|
const h = {};
|
|
@@ -431,6 +560,8 @@ export default class AiSdkProvider {
|
|
|
431
560
|
h["Plurnk-Loop"] = String(loop);
|
|
432
561
|
if (turn !== undefined && Number.isInteger(turn) && turn >= 1)
|
|
433
562
|
h["Plurnk-Turn"] = String(turn);
|
|
563
|
+
if (callKind !== undefined)
|
|
564
|
+
h["Plurnk-Call-Kind"] = callKind;
|
|
434
565
|
return h;
|
|
435
566
|
}
|
|
436
567
|
// PLURNK_PROVIDERS_GBNF_DEBUG ({§gbnf-response-observation}): validate the supplied GBNF locally and fail
|
|
@@ -470,10 +601,57 @@ export default class AiSdkProvider {
|
|
|
470
601
|
out[k] = v;
|
|
471
602
|
return out;
|
|
472
603
|
}
|
|
473
|
-
|
|
604
|
+
#requestProviderOptions(workerId, reasoningBudget) {
|
|
605
|
+
const responseOptions = this.#reasoning.mode === "off"
|
|
606
|
+
? undefined
|
|
607
|
+
: this.#reasoningResponseProviderOptions;
|
|
608
|
+
const nativeReasoning = this.#reasoning.mode === "on" && reasoningBudget !== null
|
|
609
|
+
? this.#additiveReasoningProvider === "anthropic"
|
|
610
|
+
? { anthropic: { thinking: { type: "enabled", budgetTokens: reasoningBudget } } }
|
|
611
|
+
: this.#additiveReasoningProvider === "bedrock"
|
|
612
|
+
? { bedrock: { reasoningConfig: { type: "enabled", budgetTokens: reasoningBudget } } }
|
|
613
|
+
: undefined
|
|
614
|
+
: undefined;
|
|
615
|
+
const options = {};
|
|
616
|
+
for (const part of [responseOptions, nativeReasoning]) {
|
|
617
|
+
for (const [provider, values] of Object.entries(part ?? {})) {
|
|
618
|
+
options[provider] = { ...options[provider], ...values };
|
|
619
|
+
}
|
|
620
|
+
}
|
|
621
|
+
if (this.#cacheAffinity?.target === "provider-option") {
|
|
622
|
+
const { provider, name } = this.#cacheAffinity;
|
|
623
|
+
options[provider] = { ...options[provider], [name]: workerId };
|
|
624
|
+
}
|
|
625
|
+
return Object.keys(options).length === 0 ? undefined : options;
|
|
626
|
+
}
|
|
627
|
+
#nativeMaxOutputTokens(outputBudget, reasoningBudget) {
|
|
628
|
+
if (outputBudget === null)
|
|
629
|
+
return undefined;
|
|
630
|
+
return this.#additiveReasoningProvider !== undefined
|
|
631
|
+
&& this.#reasoning.mode === "on"
|
|
632
|
+
&& reasoningBudget !== null
|
|
633
|
+
? outputBudget - reasoningBudget
|
|
634
|
+
: outputBudget;
|
|
635
|
+
}
|
|
636
|
+
#accounting(outcome, usage, evidence, status) {
|
|
637
|
+
const knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
|
|
638
|
+
const direct = this.#normalizeCost?.(evidence);
|
|
639
|
+
return validateProviderRequestAccounting({
|
|
640
|
+
provider: this.#source,
|
|
641
|
+
model: this.#model,
|
|
642
|
+
outcome,
|
|
643
|
+
...(status === undefined ? {} : { status }),
|
|
644
|
+
...(knownUsage === undefined ? {} : { usage: knownUsage }),
|
|
645
|
+
cost: resolveProviderCost(direct, this.#estimateCost(knownUsage)),
|
|
646
|
+
});
|
|
647
|
+
}
|
|
648
|
+
async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxOutputTokens, attributions, client, strikes, workspaceId, loop, turn, sampling, observeRequest, callKind }) {
|
|
474
649
|
// {§provider-interface} The worker identity is required.
|
|
475
650
|
if (workerId === undefined || workerId.length === 0)
|
|
476
651
|
throw new Error("generate: workerId is required — the worker's stable, opaque identity");
|
|
652
|
+
if (callKind !== undefined && callKind !== "emission" && callKind !== "bare") {
|
|
653
|
+
throw new Error(`generate: unsupported callKind ${JSON.stringify(callKind)}`);
|
|
654
|
+
}
|
|
477
655
|
// Reject before any wire call when already aborted
|
|
478
656
|
// ({§provider-failure-normalization}).
|
|
479
657
|
signal?.throwIfAborted();
|
|
@@ -485,6 +663,14 @@ export default class AiSdkProvider {
|
|
|
485
663
|
const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
|
|
486
664
|
const preserveGrammarSentence = wantGrammar
|
|
487
665
|
&& this.#reasoningStyle === "template";
|
|
666
|
+
const capacity = await this.assessRequestCapacity(messages, maxOutputTokens, signal);
|
|
667
|
+
if (capacity.decision === "reject") {
|
|
668
|
+
if (capacity.prompt.kind !== "exact") {
|
|
669
|
+
throw new TypeError(`${this.#source}: only an exact prompt measurement may reject capacity`);
|
|
670
|
+
}
|
|
671
|
+
throw new ProviderError(this.#source, "capacity_exceeded", `The exact provider request uses ${capacity.prompt.tokens} input tokens, exceeding its ${capacity.inputCapacity} token input capacity.`, { capacity, extensions: { capacityStage: "preflight", capacity } });
|
|
672
|
+
}
|
|
673
|
+
const effectiveMaxOutputTokens = capacity.outputBudget ?? undefined;
|
|
488
674
|
// Assembly order = precedence: the family's sampling DEFAULTS
|
|
489
675
|
// (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
|
|
490
676
|
// paths and the name promises every request) < the caller's `sampling`
|
|
@@ -497,80 +683,188 @@ export default class AiSdkProvider {
|
|
|
497
683
|
...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
|
|
498
684
|
model: this.#model,
|
|
499
685
|
messages,
|
|
500
|
-
...this.#reasoningBody(preserveGrammarSentence),
|
|
686
|
+
...this.#reasoningBody(preserveGrammarSentence, capacity.reasoningBudget),
|
|
501
687
|
...this.#grammarBody(sendGrammar),
|
|
502
|
-
...(
|
|
688
|
+
...(effectiveMaxOutputTokens !== undefined ? { max_tokens: effectiveMaxOutputTokens } : {}),
|
|
503
689
|
// Request per-token logprobs only when enabled (managed field —
|
|
504
690
|
// reserved from caller sampling; the env flag is the single control).
|
|
505
691
|
...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
|
|
506
692
|
...this.#slotBody(workerId),
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
|
|
693
|
+
...(this.#cacheAffinity?.target === "body"
|
|
694
|
+
? { [this.#cacheAffinity.name]: workerId }
|
|
695
|
+
: {}),
|
|
511
696
|
};
|
|
512
697
|
// Per-request headers = static auth/routing + any first-party telemetry.
|
|
513
|
-
const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
|
|
514
|
-
const headers =
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
698
|
+
const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind);
|
|
699
|
+
const headers = new Headers(this.#headers);
|
|
700
|
+
if (this.#cacheAffinity?.target === "header") {
|
|
701
|
+
headers.set(this.#cacheAffinity.name, workerId);
|
|
702
|
+
}
|
|
703
|
+
for (const [name, value] of Object.entries(metaHeaders))
|
|
704
|
+
headers.set(name, value);
|
|
705
|
+
const requestHeaders = Object.fromEntries(headers.entries());
|
|
706
|
+
const accounting = [];
|
|
707
|
+
const operationTimeout = this.#operationTimeoutMs > 0
|
|
708
|
+
? AbortSignal.timeout(this.#operationTimeoutMs)
|
|
709
|
+
: undefined;
|
|
710
|
+
const operationSignal = signal === undefined
|
|
711
|
+
? operationTimeout
|
|
712
|
+
: operationTimeout === undefined
|
|
713
|
+
? signal
|
|
714
|
+
: AbortSignal.any([signal, operationTimeout]);
|
|
715
|
+
const executeRequest = async () => {
|
|
716
|
+
let settle;
|
|
717
|
+
try {
|
|
718
|
+
settle = await observeRequest?.({
|
|
719
|
+
provider: this.#source,
|
|
522
720
|
model: this.#model,
|
|
523
|
-
headers,
|
|
524
|
-
body,
|
|
525
|
-
messages,
|
|
526
|
-
signal,
|
|
527
|
-
fetch: this.#fetch,
|
|
528
|
-
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
529
|
-
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
530
|
-
retryAttempts: this.#retryAttempts,
|
|
531
|
-
streaming: this.#streaming,
|
|
532
|
-
captureRawBody: this.#rawBody,
|
|
533
|
-
})
|
|
534
|
-
: await executeAiSdkModel({
|
|
535
|
-
languageModel: this.#languageModel,
|
|
536
|
-
headers,
|
|
537
|
-
messages,
|
|
538
|
-
signal,
|
|
539
|
-
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
540
|
-
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
541
|
-
retryAttempts: this.#retryAttempts,
|
|
542
|
-
streaming: this.#streaming,
|
|
543
|
-
captureRawBody: this.#rawBody,
|
|
544
|
-
temperature: this.#tuningFloors
|
|
545
|
-
? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
|
|
546
|
-
: typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
|
|
547
|
-
topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
|
|
548
|
-
topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
|
|
549
|
-
presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
|
|
550
|
-
frequencyPenalty: typeof sampling?.frequency_penalty === "number"
|
|
551
|
-
? sampling.frequency_penalty
|
|
552
|
-
: this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
|
|
553
|
-
stopSequences: typeof sampling?.stop === "string"
|
|
554
|
-
? [sampling.stop]
|
|
555
|
-
: Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
|
|
556
|
-
? sampling.stop
|
|
557
|
-
: undefined,
|
|
558
|
-
seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
|
|
559
|
-
maxOutputTokens: maxTokens,
|
|
560
|
-
reasoning: this.#reasoning.mode === "off"
|
|
561
|
-
? "none"
|
|
562
|
-
: this.#reasoning.mode === "adaptive"
|
|
563
|
-
? "provider-default"
|
|
564
|
-
: effortFromBudget(this.#reasoning.budget),
|
|
565
721
|
});
|
|
722
|
+
}
|
|
723
|
+
catch (cause) {
|
|
724
|
+
throw new ProviderRequestObserverError(cause);
|
|
725
|
+
}
|
|
726
|
+
const settleAccounting = async (outcome, usage, evidence, status) => {
|
|
727
|
+
let requestAccounting;
|
|
728
|
+
let normalizationFailure;
|
|
729
|
+
try {
|
|
730
|
+
requestAccounting = this.#accounting(outcome, usage, evidence, status);
|
|
731
|
+
}
|
|
732
|
+
catch (cause) {
|
|
733
|
+
normalizationFailure = { cause };
|
|
734
|
+
let knownUsage;
|
|
735
|
+
try {
|
|
736
|
+
knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
|
|
737
|
+
}
|
|
738
|
+
catch {
|
|
739
|
+
knownUsage = undefined;
|
|
740
|
+
}
|
|
741
|
+
requestAccounting = validateProviderRequestAccounting({
|
|
742
|
+
provider: this.#source,
|
|
743
|
+
model: this.#model,
|
|
744
|
+
outcome,
|
|
745
|
+
...(status === undefined ? {} : { status }),
|
|
746
|
+
...(knownUsage === undefined ? {} : { usage: knownUsage }),
|
|
747
|
+
cost: {
|
|
748
|
+
kind: "unknown",
|
|
749
|
+
reason: "provider request accounting could not be normalized after physical I/O",
|
|
750
|
+
},
|
|
751
|
+
});
|
|
752
|
+
}
|
|
753
|
+
accounting.push(requestAccounting);
|
|
754
|
+
try {
|
|
755
|
+
await settle?.(requestAccounting);
|
|
756
|
+
}
|
|
757
|
+
catch (cause) {
|
|
758
|
+
throw new ProviderRequestObserverError(cause);
|
|
759
|
+
}
|
|
760
|
+
if (normalizationFailure !== undefined) {
|
|
761
|
+
throw new ProviderRequestAccountingError(normalizationFailure.cause);
|
|
762
|
+
}
|
|
763
|
+
return requestAccounting;
|
|
764
|
+
};
|
|
765
|
+
let response;
|
|
766
|
+
try {
|
|
767
|
+
response = this.#languageModel === undefined
|
|
768
|
+
? await executeOpenAICompatible({
|
|
769
|
+
url: this.#url,
|
|
770
|
+
model: this.#model,
|
|
771
|
+
headers: requestHeaders,
|
|
772
|
+
body,
|
|
773
|
+
messages,
|
|
774
|
+
signal: operationSignal,
|
|
775
|
+
fetch: this.#fetch,
|
|
776
|
+
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
777
|
+
firstContentTimeoutMs: this.#firstContentTimeoutMs,
|
|
778
|
+
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
779
|
+
streaming: this.#streaming,
|
|
780
|
+
captureRawBody: this.#rawBody,
|
|
781
|
+
})
|
|
782
|
+
: await executeAiSdkModel({
|
|
783
|
+
languageModel: this.#languageModel,
|
|
784
|
+
headers: requestHeaders,
|
|
785
|
+
providerOptions: this.#requestProviderOptions(workerId, capacity.reasoningBudget),
|
|
786
|
+
systemProviderOptions: this.#systemCacheProviderOptions,
|
|
787
|
+
messages,
|
|
788
|
+
signal: operationSignal,
|
|
789
|
+
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
790
|
+
firstContentTimeoutMs: this.#firstContentTimeoutMs,
|
|
791
|
+
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
792
|
+
streaming: this.#streaming,
|
|
793
|
+
captureRawBody: this.#rawBody,
|
|
794
|
+
temperature: this.#tuningFloors
|
|
795
|
+
? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
|
|
796
|
+
: typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
|
|
797
|
+
topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
|
|
798
|
+
topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
|
|
799
|
+
presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
|
|
800
|
+
frequencyPenalty: typeof sampling?.frequency_penalty === "number"
|
|
801
|
+
? sampling.frequency_penalty
|
|
802
|
+
: this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
|
|
803
|
+
stopSequences: typeof sampling?.stop === "string"
|
|
804
|
+
? [sampling.stop]
|
|
805
|
+
: Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
|
|
806
|
+
? sampling.stop
|
|
807
|
+
: undefined,
|
|
808
|
+
seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
|
|
809
|
+
maxOutputTokens: this.#nativeMaxOutputTokens(capacity.outputBudget, capacity.reasoningBudget),
|
|
810
|
+
reasoning: this.#reasoning.mode === "off"
|
|
811
|
+
? "none"
|
|
812
|
+
: this.#reasoning.mode === "adaptive"
|
|
813
|
+
? "provider-default"
|
|
814
|
+
: this.#additiveReasoningProvider !== undefined && capacity.reasoningBudget !== null
|
|
815
|
+
? "provider-default"
|
|
816
|
+
: effortFromReasoning({
|
|
817
|
+
mode: this.#reasoning.mode,
|
|
818
|
+
budget: capacity.reasoningBudget,
|
|
819
|
+
}),
|
|
820
|
+
});
|
|
821
|
+
}
|
|
822
|
+
catch (error) {
|
|
823
|
+
const failure = transportFailureEvidence(error);
|
|
824
|
+
await settleAccounting("error", failure.usage, failure.chargeEvidence, failure.status);
|
|
825
|
+
throw error;
|
|
826
|
+
}
|
|
827
|
+
await settleAccounting("response", response.usage, response.chargeEvidence);
|
|
828
|
+
return response;
|
|
829
|
+
};
|
|
830
|
+
let raw;
|
|
831
|
+
try {
|
|
832
|
+
const { retry } = prepareRetries({
|
|
833
|
+
maxRetries: this.#retryAttempts,
|
|
834
|
+
abortSignal: operationSignal,
|
|
835
|
+
});
|
|
836
|
+
raw = await retry(executeRequest);
|
|
566
837
|
}
|
|
567
838
|
catch (err) {
|
|
839
|
+
if (err instanceof ProviderRequestObserverError
|
|
840
|
+
|| err instanceof ProviderRequestAccountingError)
|
|
841
|
+
throw err.cause;
|
|
568
842
|
if (signal?.aborted)
|
|
569
843
|
throw err;
|
|
570
|
-
|
|
844
|
+
if (operationTimeout?.aborted) {
|
|
845
|
+
const timeout = new ProviderTimeoutError("operation", this.#operationTimeoutMs, err);
|
|
846
|
+
throw new ProviderError(this.#source, "deadline_exceeded", timeout.message, {
|
|
847
|
+
status: 504,
|
|
848
|
+
cause: timeout,
|
|
849
|
+
retryable: false,
|
|
850
|
+
extensions: {
|
|
851
|
+
timeoutPhase: timeout.phase,
|
|
852
|
+
timeoutMs: timeout.timeoutMs,
|
|
853
|
+
},
|
|
854
|
+
accounting,
|
|
855
|
+
capacity,
|
|
856
|
+
});
|
|
857
|
+
}
|
|
858
|
+
const pe = toProviderError(err, this.#source, this.#errorDetailLimit, capacity);
|
|
571
859
|
if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
|
|
572
|
-
throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, {
|
|
860
|
+
throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, {
|
|
861
|
+
status: pe.status,
|
|
862
|
+
cause: err,
|
|
863
|
+
accounting,
|
|
864
|
+
capacity,
|
|
865
|
+
});
|
|
573
866
|
}
|
|
867
|
+
pe.prependAccounting(accounting);
|
|
574
868
|
throw pe;
|
|
575
869
|
}
|
|
576
870
|
// llama-server --special renders EOG tokens as text, so a turn ending
|
|
@@ -585,10 +879,10 @@ export default class AiSdkProvider {
|
|
|
585
879
|
const projectedReasoning = preserveGrammarSentence && !raw.reasoningProjected
|
|
586
880
|
? projectTemplateReasoning(raw.content)
|
|
587
881
|
: projectTaggedReasoning(raw.content, raw.reasoning, this.#reasoningResponseStyle);
|
|
588
|
-
// Preserve the exact
|
|
589
|
-
//
|
|
590
|
-
//
|
|
591
|
-
// supply independent
|
|
882
|
+
// Preserve the exact pre-projection response. Constrained template turns
|
|
883
|
+
// request `reasoning_format: "none"`, so even an empty channel and any
|
|
884
|
+
// template-provided opener remain observable. An unexpectedly projected
|
|
885
|
+
// response cannot supply independent evidence.
|
|
592
886
|
let grammarEvidence;
|
|
593
887
|
if (wantGrammar) {
|
|
594
888
|
if (preserveGrammarSentence) {
|
|
@@ -618,11 +912,12 @@ export default class AiSdkProvider {
|
|
|
618
912
|
if (projectedReasoning.projected) {
|
|
619
913
|
raw.content = projectedReasoning.content;
|
|
620
914
|
raw.reasoning = projectedReasoning.reasoning;
|
|
621
|
-
raw.usage = attributeUnitemizedReasoning(raw.usage, raw.reasoning, raw.content);
|
|
622
915
|
}
|
|
623
916
|
let notices;
|
|
624
917
|
const usage = raw.usage;
|
|
625
|
-
if (sendGrammar !== undefined
|
|
918
|
+
if (sendGrammar !== undefined
|
|
919
|
+
&& this.tokenize !== undefined
|
|
920
|
+
&& usage?.outputTokens !== undefined) {
|
|
626
921
|
// Channel-escape detector: completion tokens
|
|
627
922
|
// billed far beyond every visible channel mean the decode ESCAPED into
|
|
628
923
|
// a server-discarded reasoning block mid-emission. This diagnostic
|
|
@@ -633,12 +928,12 @@ export default class AiSdkProvider {
|
|
|
633
928
|
this.tokenize(raw.reasoning),
|
|
634
929
|
]);
|
|
635
930
|
const visible = contentTokens.length + reasoningTokens.length;
|
|
636
|
-
if (usage.
|
|
931
|
+
if (usage.outputTokens > visible + 64) {
|
|
637
932
|
(notices ??= []).push({
|
|
638
933
|
source: this.#source,
|
|
639
934
|
kind: "grammar_unenforced",
|
|
640
935
|
level: "warn",
|
|
641
|
-
message: `decode escaped the grammar: ${usage.
|
|
936
|
+
message: `decode escaped the grammar: ${usage.outputTokens} output tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
|
|
642
937
|
position: [...raw.content].length,
|
|
643
938
|
});
|
|
644
939
|
}
|
|
@@ -658,22 +953,35 @@ export default class AiSdkProvider {
|
|
|
658
953
|
...(raw.reasoningEncrypted.length > 0
|
|
659
954
|
? { reasoningEncrypted: raw.reasoningEncrypted }
|
|
660
955
|
: {}),
|
|
661
|
-
usage,
|
|
662
956
|
model: raw.model,
|
|
663
957
|
...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
|
|
664
958
|
};
|
|
665
|
-
const normalizedCharge = this.#normalizeCharge?.(raw.chargeEvidence);
|
|
666
|
-
const charge = normalizedCharge === undefined
|
|
667
|
-
? undefined
|
|
668
|
-
: validateAuthoritativeCharge(normalizedCharge);
|
|
669
959
|
const evidence = {
|
|
670
960
|
assistantRaw: raw,
|
|
671
|
-
|
|
961
|
+
accounting,
|
|
962
|
+
capacity,
|
|
672
963
|
...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
|
|
673
964
|
...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
|
|
674
965
|
...(meta !== undefined ? { meta } : {}),
|
|
675
966
|
...(notices !== undefined ? { notices } : {}),
|
|
676
967
|
};
|
|
968
|
+
if (capacity.outputBudget !== null
|
|
969
|
+
&& usage?.outputTokens !== undefined
|
|
970
|
+
&& usage.outputTokens > capacity.outputBudget) {
|
|
971
|
+
const attempt = {
|
|
972
|
+
assistant: { ...assistant, finishReason: raw.finishReason },
|
|
973
|
+
...evidence,
|
|
974
|
+
};
|
|
975
|
+
throw new ProviderError(this.#source, "invalid_response", `The provider reported ${usage.outputTokens} output tokens after receiving a total output budget of ${capacity.outputBudget}.`, {
|
|
976
|
+
attempt,
|
|
977
|
+
accounting,
|
|
978
|
+
extensions: {
|
|
979
|
+
stage: "provider-response",
|
|
980
|
+
outputBudget: capacity.outputBudget,
|
|
981
|
+
reportedOutputTokens: usage.outputTokens,
|
|
982
|
+
},
|
|
983
|
+
});
|
|
984
|
+
}
|
|
677
985
|
if (raw.finishReason === "resource_interrupted") {
|
|
678
986
|
const attempt = {
|
|
679
987
|
assistant: { ...assistant, finishReason: raw.finishReason },
|
|
@@ -681,6 +989,7 @@ export default class AiSdkProvider {
|
|
|
681
989
|
};
|
|
682
990
|
throw new ProviderError(this.#source, "resource_interrupted", "The provider interrupted generation because inference resources were unavailable.", {
|
|
683
991
|
attempt,
|
|
992
|
+
accounting,
|
|
684
993
|
extensions: {
|
|
685
994
|
stage: "provider-response",
|
|
686
995
|
finishReason: "resource_interrupted",
|