@plurnk/plurnk-providers 1.5.0 → 1.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +41 -34
- package/README.md +15 -0
- package/SPEC.md +242 -89
- package/dist/AiSdkProvider.d.ts +33 -33
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +442 -133
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +10 -11
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +87 -25
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +9 -24
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +86 -25
- package/dist/Pool.js.map +1 -1
- package/dist/accounting.d.ts +5 -2
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js +100 -16
- package/dist/accounting.js.map +1 -1
- package/dist/accountingPublic.d.ts +5 -0
- package/dist/accountingPublic.d.ts.map +1 -0
- package/dist/accountingPublic.js +3 -0
- package/dist/accountingPublic.js.map +1 -0
- package/dist/aiSdkTransport.d.ts +9 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +160 -62
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/capacity.d.ts +26 -0
- package/dist/capacity.d.ts.map +1 -0
- package/dist/capacity.js +90 -0
- package/dist/capacity.js.map +1 -0
- package/dist/catalogProvider.d.ts +8 -3
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +45 -41
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +26 -12
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +10 -10
- package/dist/cost.d.ts.map +1 -1
- package/dist/cost.js +90 -42
- package/dist/cost.js.map +1 -1
- package/dist/env.d.ts +13 -11
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +83 -46
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +17 -3
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +91 -8
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +7 -6
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5 -3
- package/dist/index.js.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/promptTokens.d.ts.map +1 -1
- package/dist/promptTokens.js +7 -4
- package/dist/promptTokens.js.map +1 -1
- package/dist/sdkModels.d.ts +7 -2
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +43 -13
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +55 -33
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +22 -5
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +169 -83
- package/dist/usage.js.map +1 -1
- package/package.json +18 -7
- package/src/AiSdkProvider.test.ts +964 -206
- package/src/AiSdkProvider.ts +545 -155
- package/src/Mock.test.ts +69 -30
- package/src/Mock.ts +99 -29
- package/src/Pool.test.ts +90 -19
- package/src/Pool.ts +96 -27
- package/src/ProviderRegistry.test.ts +16 -11
- package/src/accounting.test.ts +58 -22
- package/src/accounting.ts +119 -18
- package/src/accountingPublic.ts +9 -0
- package/src/aiSdkTransport.test.ts +42 -49
- package/src/aiSdkTransport.ts +174 -62
- package/src/boundaries.test.ts +2 -0
- package/src/capacity.test.ts +92 -0
- package/src/capacity.ts +140 -0
- package/src/catalogProvider.test.ts +339 -30
- package/src/catalogProvider.ts +65 -47
- package/src/compatibleProvider.test.ts +7 -5
- package/src/compatibleProvider.ts +29 -13
- package/src/cost.test.ts +86 -36
- package/src/cost.ts +111 -50
- package/src/defaults.test.ts +13 -3
- package/src/env.test.ts +103 -25
- package/src/env.ts +153 -65
- package/src/errors.test.ts +80 -2
- package/src/errors.ts +107 -8
- package/src/index.ts +26 -7
- package/src/ollama.test.ts +5 -3
- package/src/ollama.ts +3 -3
- package/src/promptTokens.ts +8 -5
- package/src/sdkModels.test.ts +77 -8
- package/src/sdkModels.ts +51 -15
- package/src/types.ts +112 -51
- package/src/usage.test.ts +112 -116
- package/src/usage.ts +214 -93
package/src/AiSdkProvider.ts
CHANGED
|
@@ -6,19 +6,42 @@
|
|
|
6
6
|
// ordinary vendor protocol. The compatible URL path remains only for PLURNK
|
|
7
7
|
// extensions and local endpoint probes the SDK cannot represent.
|
|
8
8
|
|
|
9
|
-
import type {
|
|
9
|
+
import type {
|
|
10
|
+
ChatMessage,
|
|
11
|
+
GrammarEvidence,
|
|
12
|
+
PromptTokenMeasurement,
|
|
13
|
+
Provider,
|
|
14
|
+
ProviderAttempt,
|
|
15
|
+
ProviderCostNormalizer,
|
|
16
|
+
ProviderCallKind,
|
|
17
|
+
ProviderGenerateArgs,
|
|
18
|
+
ProviderRequestAccounting,
|
|
19
|
+
ProviderRequestCapacity,
|
|
20
|
+
ProviderRequestSettlement,
|
|
21
|
+
ProviderResponse,
|
|
22
|
+
ProviderUsage,
|
|
23
|
+
} from "./types.ts";
|
|
10
24
|
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
11
|
-
import type {
|
|
12
|
-
import {
|
|
25
|
+
import type { JSONValue } from "ai";
|
|
26
|
+
import { MAX_PROVIDER_TIMEOUT_MS } from "./env.ts";
|
|
27
|
+
import type { Reasoning, ReasoningResponseStyle } from "./env.ts";
|
|
28
|
+
import {
|
|
29
|
+
executeAiSdkModel,
|
|
30
|
+
executeOpenAICompatible,
|
|
31
|
+
transportFailureEvidence,
|
|
32
|
+
} from "./aiSdkTransport.ts";
|
|
13
33
|
import type { LanguageModel } from "ai";
|
|
14
|
-
import {
|
|
15
|
-
import {
|
|
34
|
+
import { prepareRetries } from "ai/internal";
|
|
35
|
+
import { toProviderError, ProviderError, ProviderTimeoutError } from "./errors.ts";
|
|
16
36
|
import type { ProviderNotice } from "./notices.ts";
|
|
17
37
|
import { validateGbnf } from "@plurnk/gbnf";
|
|
18
38
|
import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.ts";
|
|
19
39
|
import { emitWarningOnce } from "./warnings.ts";
|
|
20
40
|
import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
|
|
21
|
-
import {
|
|
41
|
+
import { resolveProviderCost } from "./cost.ts";
|
|
42
|
+
import { validateProviderRequestAccounting } from "./accounting.ts";
|
|
43
|
+
import { validateProviderUsage } from "./usage.ts";
|
|
44
|
+
import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.ts";
|
|
22
45
|
|
|
23
46
|
export type ProviderFetch = typeof globalThis.fetch;
|
|
24
47
|
|
|
@@ -30,30 +53,49 @@ export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" |
|
|
|
30
53
|
// service-managed constrained sampling; endpoint-owned settings are not inferred.
|
|
31
54
|
export type GrammarStyle = "none" | "llamacpp";
|
|
32
55
|
|
|
56
|
+
export type CacheAffinity =
|
|
57
|
+
| { readonly target: "header" | "body"; readonly name: string }
|
|
58
|
+
| { readonly target: "provider-option"; readonly provider: string; readonly name: string };
|
|
59
|
+
|
|
60
|
+
export type AiSdkProviderOptions = Record<string, Record<string, JSONValue | undefined>>;
|
|
61
|
+
|
|
33
62
|
export type AiSdkProviderConfig = {
|
|
34
63
|
model: string;
|
|
35
64
|
url?: string; // OpenAI-compatible chat-completions URL
|
|
36
65
|
languageModel?: LanguageModel; // native AI SDK provider model
|
|
37
66
|
attributions?: (context: PluginAttributionContext) => PluginAttribution;
|
|
38
|
-
fetchTimeoutMs: number;
|
|
39
|
-
|
|
67
|
+
fetchTimeoutMs: number; // one physical generation attempt; zero disables
|
|
68
|
+
operationTimeoutMs: number; // complete logical call across retries/backoff; zero disables
|
|
69
|
+
firstContentTimeoutMs: number; // first semantic streamed content; zero disables
|
|
70
|
+
streamIdleTimeoutMs?: number; // semantic streamed-content idle deadline; zero/unset disables
|
|
40
71
|
headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
|
|
41
72
|
fetch?: ProviderFetch; // per-instance request executor; default globalThis.fetch
|
|
42
73
|
contextWindow?: number | null; // default null; caller resolves-or-fails, narrows to required with the interface
|
|
74
|
+
maxInputTokens?: number | null;
|
|
75
|
+
maxOutputTokens?: number | null;
|
|
76
|
+
outputBudget?: number | null;
|
|
77
|
+
reasoningBudget?: number | null;
|
|
78
|
+
// Native Anthropic and Bedrock SDKs interpret generic maxOutputTokens as
|
|
79
|
+
// visible output and add an explicit provider reasoning budget. This marker lets the
|
|
80
|
+
// adapter subtract that subset so the resulting wire cap remains PLURNK's
|
|
81
|
+
// one total output budget.
|
|
82
|
+
additiveReasoningProvider?: "anthropic" | "bedrock";
|
|
43
83
|
reasoningStyle?: ReasoningStyle; // default "none"
|
|
44
84
|
reasoningResponseStyle?: ReasoningResponseStyle; // {§provider-tagged-reasoning}; default "verbatim"
|
|
45
85
|
countPromptTokens?: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
normalizeCharge?: AuthoritativeChargeNormalizer;
|
|
86
|
+
estimateCost?: (usage: ProviderUsage | undefined) => ProviderCost;
|
|
87
|
+
normalizeCost?: ProviderCostNormalizer;
|
|
49
88
|
source?: string; // notice/problem source, e.g. "provider:openai"; default "provider"
|
|
50
89
|
grammarStyle?: GrammarStyle; // how a GBNF grammar is carried; default "none" (not sent)
|
|
51
|
-
//
|
|
52
|
-
//
|
|
53
|
-
|
|
54
|
-
//
|
|
55
|
-
//
|
|
56
|
-
|
|
90
|
+
// {§provider-cache-affinity} Provider routes own the exact documented
|
|
91
|
+
// projection; the common transport only applies it as managed request state.
|
|
92
|
+
cacheAffinity?: CacheAffinity;
|
|
93
|
+
// {§provider-cache-write-policy} Already policy-gated by provider construction.
|
|
94
|
+
// The transport attaches it to only the final leading system instruction.
|
|
95
|
+
systemCacheProviderOptions?: AiSdkProviderOptions;
|
|
96
|
+
// {§provider-readable-reasoning} Route-owned native option needed to expose
|
|
97
|
+
// readable reasoning. Applied only when the effective posture is not off.
|
|
98
|
+
reasoningResponseProviderOptions?: AiSdkProviderOptions;
|
|
57
99
|
// Optional provider-configured service tier. Unlike caller sampling, this is
|
|
58
100
|
// a fixed deployment choice and therefore wins on every request.
|
|
59
101
|
serviceTier?: string;
|
|
@@ -78,14 +120,14 @@ export type AiSdkProviderConfig = {
|
|
|
78
120
|
// when no probe ran or it read no row.
|
|
79
121
|
servedModel?: string;
|
|
80
122
|
// Backend decodes unbounded without a caller cap (llama-server n_predict
|
|
81
|
-
// to the wall) — surfaced as Provider.
|
|
123
|
+
// to the wall) — surfaced as Provider.requiresOutputBudget so consumers can
|
|
82
124
|
// boot-refuse an envelope-less local alias. Default unset (no claim).
|
|
83
|
-
|
|
125
|
+
requiresOutputBudget?: boolean;
|
|
84
126
|
// The side-channel reasoning intent — REQUIRED, no in-code default
|
|
85
127
|
// (PLURNK_PROVIDERS_REASONING + _BUDGET, read via reasoningFromEnv):
|
|
86
|
-
// { mode: off|adaptive|on, budget:
|
|
87
|
-
// backend's mechanism via reasoningStyle; budget is only ever
|
|
88
|
-
// never a hidden activation flag.
|
|
128
|
+
// { mode: off|adaptive|on, budget: optional when on }. The provider maps it
|
|
129
|
+
// to the backend's mechanism via reasoningStyle; budget is only ever an
|
|
130
|
+
// explicit magnitude, never a hidden activation flag.
|
|
89
131
|
reasoning: Reasoning;
|
|
90
132
|
// Decode tuning: no in-code defaults; the canonical measured values (0.2 /
|
|
91
133
|
// 1.15) ship as the floor in .env.defaults (alias-scopable). `temperature` is the
|
|
@@ -126,20 +168,32 @@ export type AiSdkProviderConfig = {
|
|
|
126
168
|
// gated per-alias.
|
|
127
169
|
topLogprobs?: number | null;
|
|
128
170
|
rawBody?: boolean;
|
|
129
|
-
// {§provider-generation-envelope} The generation
|
|
130
|
-
//
|
|
171
|
+
// {§provider-generation-envelope} The generation budgets, env-read via the
|
|
172
|
+
// common envelope parser — a percentage of the detected window or an absolute token
|
|
131
173
|
// count. Optional so an out-of-date sibling keeps constructing (no claim);
|
|
132
174
|
// the standard factory always supplies them. Resolved against contextWindow
|
|
133
175
|
// at read time (getters), so a probe that lands after config assembly still
|
|
134
176
|
// derives correctly.
|
|
135
|
-
reasoningReserve?: ReserveSpec;
|
|
136
|
-
completionReserve?: ReserveSpec;
|
|
137
177
|
// The plurnk.ai router owns tuning — false suppresses the
|
|
138
178
|
// client-side temperature/penalty FLOORS on this provider (caller `sampling`
|
|
139
179
|
// still passes through verbatim). Default true (floors ride).
|
|
140
180
|
tuningFloors?: boolean;
|
|
141
181
|
};
|
|
142
182
|
|
|
183
|
+
class ProviderRequestObserverError extends Error {
|
|
184
|
+
constructor(cause: unknown) {
|
|
185
|
+
super("provider request accounting could not be durably settled", { cause });
|
|
186
|
+
this.name = "ProviderRequestObserverError";
|
|
187
|
+
}
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
class ProviderRequestAccountingError extends Error {
|
|
191
|
+
constructor(cause: unknown) {
|
|
192
|
+
super("provider request accounting could not be normalized", { cause });
|
|
193
|
+
this.name = "ProviderRequestAccountingError";
|
|
194
|
+
}
|
|
195
|
+
}
|
|
196
|
+
|
|
143
197
|
// Drop trailing occurrences of a server-rendered EOG marker. llama-server
|
|
144
198
|
// under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
|
|
145
199
|
// trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
|
|
@@ -197,12 +251,19 @@ const projectTaggedReasoning = (
|
|
|
197
251
|
? projectLeadingReasoning(content, structuredReasoning, "<think>", "</think>")
|
|
198
252
|
: { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
|
|
199
253
|
|
|
200
|
-
// llama-server's template reasoning parser can project
|
|
201
|
-
// of the OpenAI-compatible response. Grammar evidence
|
|
202
|
-
// that lossy projection, so constrained template turns
|
|
203
|
-
// split the observed enclosure here.
|
|
204
|
-
const projectTemplateReasoning = (content: string): TaggedReasoningProjection =>
|
|
205
|
-
|
|
254
|
+
// llama-server's template reasoning parser can project either supported leading
|
|
255
|
+
// reasoning envelope out of the OpenAI-compatible response. Grammar evidence
|
|
256
|
+
// needs the sentence before that lossy projection, so constrained template turns
|
|
257
|
+
// request it verbatim and split the observed enclosure here.
|
|
258
|
+
const projectTemplateReasoning = (content: string): TaggedReasoningProjection => {
|
|
259
|
+
for (const [opening, closing] of [
|
|
260
|
+
["<|channel>thought\n", "<channel|>"],
|
|
261
|
+
["<think>\n", "</think>"],
|
|
262
|
+
] as const) {
|
|
263
|
+
if (content.startsWith(opening)) return projectLeadingReasoning(content, "", opening, closing);
|
|
264
|
+
}
|
|
265
|
+
return { content, reasoning: "", projected: false, contentStart: 0 };
|
|
266
|
+
};
|
|
206
267
|
|
|
207
268
|
// Shared budget→effort breakpoints (xai and google had identical copies).
|
|
208
269
|
export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
|
|
@@ -211,6 +272,13 @@ export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
|
|
|
211
272
|
return "high";
|
|
212
273
|
};
|
|
213
274
|
|
|
275
|
+
// AI SDK's portable reasoning control has no boolean-enabled value. `medium`
|
|
276
|
+
// is the neutral activation projection for an explicit, unqualified `on`; it
|
|
277
|
+
// changes no PLURNK output budget. An operator reasoning subset, when present, remains
|
|
278
|
+
// the only input to the existing magnitude-to-tier projection.
|
|
279
|
+
const effortFromReasoning = (reasoning: Reasoning): "low" | "medium" | "high" =>
|
|
280
|
+
reasoning.budget === null ? "medium" : effortFromBudget(reasoning.budget);
|
|
281
|
+
|
|
214
282
|
// Body keys the provider owns — a caller's `sampling` passthrough may not set
|
|
215
283
|
// these. Two families:
|
|
216
284
|
// transport/managed — grammar transport, the stream/JSON choice, slot pinning,
|
|
@@ -219,7 +287,7 @@ export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
|
|
|
219
287
|
// response; n>1 = paid, dropped output), the tool-calling family (tools-in-
|
|
220
288
|
// body doctrine, §2: native tool_calls return null content = a broken turn),
|
|
221
289
|
// modalities/audio (text-only contract), prediction (decode semantics, not
|
|
222
|
-
// sampling), and the token caps (the envelope is the managed
|
|
290
|
+
// sampling), and the token caps (the envelope is the managed maxOutputTokens —
|
|
223
291
|
// sampling must not bypass the consumer's cap).
|
|
224
292
|
// Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
|
|
225
293
|
// platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
|
|
@@ -238,6 +306,8 @@ export default class AiSdkProvider implements Provider {
|
|
|
238
306
|
#url: string | undefined;
|
|
239
307
|
#languageModel: LanguageModel | undefined;
|
|
240
308
|
#fetchTimeoutMs: number;
|
|
309
|
+
#operationTimeoutMs: number;
|
|
310
|
+
#firstContentTimeoutMs: number;
|
|
241
311
|
#streamIdleTimeoutMs: number | undefined;
|
|
242
312
|
#headers: Record<string, string>;
|
|
243
313
|
#fetch: ProviderFetch;
|
|
@@ -245,6 +315,11 @@ export default class AiSdkProvider implements Provider {
|
|
|
245
315
|
#apiKeyRejectedMessage: string | undefined;
|
|
246
316
|
#eosText: string | undefined;
|
|
247
317
|
#contextWindow: number | null;
|
|
318
|
+
#maxInputTokens: number | null;
|
|
319
|
+
#maxOutputTokens: number | null;
|
|
320
|
+
#outputBudget: number | null;
|
|
321
|
+
#reasoningBudget: number | null;
|
|
322
|
+
#additiveReasoningProvider: "anthropic" | "bedrock" | undefined;
|
|
248
323
|
#reasoning: Reasoning;
|
|
249
324
|
#temperature: number;
|
|
250
325
|
#repeatPenalty: number;
|
|
@@ -257,12 +332,13 @@ export default class AiSdkProvider implements Provider {
|
|
|
257
332
|
#reasoningResponseStyle: ReasoningResponseStyle;
|
|
258
333
|
#countPromptTokens: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
|
|
259
334
|
#promptTokensUrl: string | undefined;
|
|
260
|
-
#
|
|
261
|
-
#
|
|
262
|
-
#normalizeCharge?: AuthoritativeChargeNormalizer;
|
|
335
|
+
#estimateCost: (usage: ProviderUsage | undefined) => ProviderCost;
|
|
336
|
+
#normalizeCost?: ProviderCostNormalizer;
|
|
263
337
|
#source: string;
|
|
264
338
|
#grammarStyle: GrammarStyle;
|
|
265
|
-
#
|
|
339
|
+
#cacheAffinity: CacheAffinity | undefined;
|
|
340
|
+
#systemCacheProviderOptions: AiSdkProviderOptions | undefined;
|
|
341
|
+
#reasoningResponseProviderOptions: AiSdkProviderOptions | undefined;
|
|
266
342
|
#serviceTier: string | undefined;
|
|
267
343
|
#gbnfDebug: boolean;
|
|
268
344
|
#streaming: boolean;
|
|
@@ -272,12 +348,10 @@ export default class AiSdkProvider implements Provider {
|
|
|
272
348
|
#retryAttempts: number;
|
|
273
349
|
#errorDetailLimit: number | undefined;
|
|
274
350
|
#topLogprobs: number | null;
|
|
275
|
-
#reasoningReserve: ReserveSpec | undefined;
|
|
276
|
-
#completionReserve: ReserveSpec | undefined;
|
|
277
351
|
#tuningFloors: boolean;
|
|
278
352
|
#rawBody: boolean;
|
|
279
353
|
#servedModel: string | undefined;
|
|
280
|
-
#
|
|
354
|
+
#requiresOutputBudget: boolean | undefined;
|
|
281
355
|
readonly attributions?: (context: PluginAttributionContext) => PluginAttribution;
|
|
282
356
|
|
|
283
357
|
// Optional capability ({§provider-local-capabilities}): exact tokenization served by the backend's
|
|
@@ -293,11 +367,28 @@ export default class AiSdkProvider implements Provider {
|
|
|
293
367
|
if ((this.#url === undefined) === (this.#languageModel === undefined)) {
|
|
294
368
|
throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
|
|
295
369
|
}
|
|
370
|
+
for (const [name, value] of [
|
|
371
|
+
["fetchTimeoutMs", config.fetchTimeoutMs],
|
|
372
|
+
["operationTimeoutMs", config.operationTimeoutMs],
|
|
373
|
+
["firstContentTimeoutMs", config.firstContentTimeoutMs],
|
|
374
|
+
["streamIdleTimeoutMs", config.streamIdleTimeoutMs],
|
|
375
|
+
] as const) {
|
|
376
|
+
if (value !== undefined && (!Number.isInteger(value) || value < 0 || value > MAX_PROVIDER_TIMEOUT_MS)) {
|
|
377
|
+
throw new Error(`${config.source ?? "provider"}: ${name} must be an integer from 0 through ${MAX_PROVIDER_TIMEOUT_MS} milliseconds`);
|
|
378
|
+
}
|
|
379
|
+
}
|
|
296
380
|
this.#fetchTimeoutMs = config.fetchTimeoutMs;
|
|
381
|
+
this.#operationTimeoutMs = config.operationTimeoutMs;
|
|
382
|
+
this.#firstContentTimeoutMs = config.firstContentTimeoutMs;
|
|
297
383
|
this.#streamIdleTimeoutMs = config.streamIdleTimeoutMs;
|
|
298
384
|
this.#headers = config.headers ?? {};
|
|
299
385
|
this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
|
|
300
386
|
this.#contextWindow = config.contextWindow ?? null;
|
|
387
|
+
this.#maxInputTokens = config.maxInputTokens ?? null;
|
|
388
|
+
this.#maxOutputTokens = config.maxOutputTokens ?? null;
|
|
389
|
+
this.#outputBudget = config.outputBudget ?? null;
|
|
390
|
+
this.#reasoningBudget = config.reasoningBudget ?? null;
|
|
391
|
+
this.#additiveReasoningProvider = config.additiveReasoningProvider;
|
|
301
392
|
this.#reasoning = config.reasoning;
|
|
302
393
|
// Loud guard: an out-of-date consumer (stale plugin dist) omitting the
|
|
303
394
|
// required tuning fields must fail at construction, not silently send
|
|
@@ -321,12 +412,36 @@ export default class AiSdkProvider implements Provider {
|
|
|
321
412
|
}
|
|
322
413
|
this.#countPromptTokens = config.countPromptTokens ?? ((messages) => estimatePromptTokens(messages));
|
|
323
414
|
this.#promptTokensUrl = config.promptTokensUrl;
|
|
324
|
-
this.#
|
|
325
|
-
|
|
326
|
-
|
|
415
|
+
this.#estimateCost = config.estimateCost
|
|
416
|
+
?? (() => ({
|
|
417
|
+
kind: "unknown",
|
|
418
|
+
reason: "the request reported no direct cost and no model rate is configured",
|
|
419
|
+
}));
|
|
420
|
+
this.#normalizeCost = config.normalizeCost;
|
|
327
421
|
this.#source = config.source ?? "provider";
|
|
328
422
|
this.#grammarStyle = config.grammarStyle ?? "none";
|
|
329
|
-
this.#
|
|
423
|
+
this.#cacheAffinity = config.cacheAffinity;
|
|
424
|
+
this.#systemCacheProviderOptions = config.systemCacheProviderOptions;
|
|
425
|
+
this.#reasoningResponseProviderOptions = config.reasoningResponseProviderOptions;
|
|
426
|
+
if (this.#languageModel === undefined && this.#cacheAffinity?.target === "provider-option") {
|
|
427
|
+
throw new Error(`${this.#source}: native provider-option cache affinity requires an AI SDK model`);
|
|
428
|
+
}
|
|
429
|
+
if (this.#languageModel !== undefined && this.#cacheAffinity?.target === "body") {
|
|
430
|
+
throw new Error(`${this.#source}: body cache affinity requires an OpenAI-compatible URL`);
|
|
431
|
+
}
|
|
432
|
+
if (this.#languageModel === undefined && this.#systemCacheProviderOptions !== undefined) {
|
|
433
|
+
throw new Error(`${this.#source}: system cache provider options require an AI SDK model`);
|
|
434
|
+
}
|
|
435
|
+
if (this.#languageModel === undefined && this.#reasoningResponseProviderOptions !== undefined) {
|
|
436
|
+
throw new Error(`${this.#source}: reasoning response provider options require an AI SDK model`);
|
|
437
|
+
}
|
|
438
|
+
if (this.#cacheAffinity?.target === "provider-option"
|
|
439
|
+
&& Object.hasOwn(
|
|
440
|
+
this.#reasoningResponseProviderOptions?.[this.#cacheAffinity.provider] ?? {},
|
|
441
|
+
this.#cacheAffinity.name,
|
|
442
|
+
)) {
|
|
443
|
+
throw new Error(`${this.#source}: reasoning response options conflict with cache-affinity option ${this.#cacheAffinity.provider}.${this.#cacheAffinity.name}`);
|
|
444
|
+
}
|
|
330
445
|
this.#serviceTier = config.serviceTier;
|
|
331
446
|
this.#gbnfDebug = config.gbnfDebug ?? false;
|
|
332
447
|
this.#streaming = config.streaming ?? true;
|
|
@@ -337,18 +452,42 @@ export default class AiSdkProvider implements Provider {
|
|
|
337
452
|
this.#supportsSlotPinning = config.supportsSlotPinning ?? false;
|
|
338
453
|
this.#slotCount = config.slotCount ?? null;
|
|
339
454
|
this.#topLogprobs = config.topLogprobs ?? null;
|
|
340
|
-
this.#reasoningReserve = config.reasoningReserve;
|
|
341
|
-
this.#completionReserve = config.completionReserve;
|
|
342
455
|
this.#tuningFloors = config.tuningFloors ?? true;
|
|
343
456
|
this.#rawBody = config.rawBody ?? false;
|
|
344
457
|
this.#servedModel = config.servedModel;
|
|
345
|
-
this.#
|
|
346
|
-
const
|
|
347
|
-
|
|
458
|
+
this.#requiresOutputBudget = config.requiresOutputBudget;
|
|
459
|
+
for (const [name, value] of [
|
|
460
|
+
["contextWindow", this.#contextWindow],
|
|
461
|
+
["maxInputTokens", this.#maxInputTokens],
|
|
462
|
+
["maxOutputTokens", this.#maxOutputTokens],
|
|
463
|
+
["outputBudget", this.#outputBudget],
|
|
464
|
+
["reasoningBudget", this.#reasoningBudget],
|
|
465
|
+
] as const) {
|
|
466
|
+
if (value !== null && (!Number.isSafeInteger(value) || value <= 0)) {
|
|
467
|
+
throw new Error(`${this.#source}: ${name} must be a positive safe integer or null`);
|
|
468
|
+
}
|
|
469
|
+
}
|
|
470
|
+
if (this.#reasoningBudget !== null
|
|
471
|
+
&& this.#outputBudget !== null
|
|
472
|
+
&& this.#reasoningBudget >= this.#outputBudget) {
|
|
473
|
+
throw new Error(`${this.#source}: reasoningBudget must be smaller than the total outputBudget`);
|
|
474
|
+
}
|
|
475
|
+
if (this.#reasoning.budget !== this.#reasoningBudget) {
|
|
476
|
+
throw new Error(`${this.#source}: reasoning intent and generation envelope disagree on reasoningBudget`);
|
|
477
|
+
}
|
|
478
|
+
if (this.#reasoningStyle === "anthropic"
|
|
479
|
+
&& this.#reasoning.mode === "on"
|
|
480
|
+
&& this.#reasoning.budget === null
|
|
481
|
+
&& this.#reasoningBudget === null) {
|
|
482
|
+
throw new Error(`${this.#source}: explicit Anthropic reasoning requires PLURNK_PROVIDERS_REASONING_BUDGET`);
|
|
483
|
+
}
|
|
484
|
+
if (this.#additiveReasoningProvider !== undefined
|
|
348
485
|
&& this.#reasoning.mode === "on"
|
|
349
|
-
&&
|
|
350
|
-
|
|
351
|
-
|
|
486
|
+
&& this.#reasoningBudget === null) {
|
|
487
|
+
throw new Error(`${this.#source}: explicit ${this.#additiveReasoningProvider} reasoning requires PLURNK_PROVIDERS_REASONING_BUDGET so the total output budget remains bounded`);
|
|
488
|
+
}
|
|
489
|
+
if (this.#requiresOutputBudget === true && this.#outputBudget === null) {
|
|
490
|
+
throw new Error(`${this.#source}: this backend requires a resolved PLURNK_PROVIDERS_OUTPUT_BUDGET`);
|
|
352
491
|
}
|
|
353
492
|
const { tokenizeUrl } = config;
|
|
354
493
|
if (tokenizeUrl !== undefined) {
|
|
@@ -357,7 +496,9 @@ export default class AiSdkProvider implements Provider {
|
|
|
357
496
|
method: "POST",
|
|
358
497
|
headers: { "Content-Type": "application/json", ...this.#headers },
|
|
359
498
|
body: JSON.stringify({ content: text }),
|
|
360
|
-
|
|
499
|
+
...(this.#fetchTimeoutMs > 0
|
|
500
|
+
? { signal: AbortSignal.timeout(this.#fetchTimeoutMs) }
|
|
501
|
+
: {}),
|
|
361
502
|
});
|
|
362
503
|
if (!res.ok) throw new Error(`${this.#source}: tokenize endpoint returned ${res.status}`);
|
|
363
504
|
const { tokens } = (await res.json()) as { tokens?: unknown };
|
|
@@ -370,20 +511,22 @@ export default class AiSdkProvider implements Provider {
|
|
|
370
511
|
}
|
|
371
512
|
|
|
372
513
|
get contextWindow(): number | null { return this.#contextWindow; }
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
return
|
|
514
|
+
get maxInputTokens(): number | null { return this.#maxInputTokens; }
|
|
515
|
+
get maxOutputTokens(): number | null { return this.#maxOutputTokens; }
|
|
516
|
+
get outputBudget(): number | null { return this.#outputBudget; }
|
|
517
|
+
get reasoningBudget(): number | null { return this.#reasoningBudget; }
|
|
518
|
+
get inputCapacity(): number | null {
|
|
519
|
+
return effectiveInputCapacity({
|
|
520
|
+
contextWindow: this.#contextWindow,
|
|
521
|
+
maxInputTokens: this.#maxInputTokens,
|
|
522
|
+
outputBudget: this.#outputBudget,
|
|
523
|
+
});
|
|
379
524
|
}
|
|
380
|
-
get reasoningReserve(): number | null { return this.#resolveReserve(this.#reasoningReserve); }
|
|
381
|
-
get completionReserve(): number | null { return this.#resolveReserve(this.#completionReserve); }
|
|
382
525
|
get model(): string { return this.#model; }
|
|
383
526
|
// Backend's self-reported served id; undefined when unprobed/unknown.
|
|
384
527
|
get servedModel(): string | undefined { return this.#servedModel; }
|
|
385
528
|
// Resolved "decodes unbounded without a cap" fact; undefined = no claim.
|
|
386
|
-
get
|
|
529
|
+
get requiresOutputBudget(): boolean | undefined { return this.#requiresOutputBudget; }
|
|
387
530
|
// Resolved capability: will a transported grammar actually constrain
|
|
388
531
|
// this backend's decode? Introspectable so a consumer can verify the rails
|
|
389
532
|
// are LIVE without spending a generation on a forcing-grammar probe.
|
|
@@ -402,7 +545,14 @@ export default class AiSdkProvider implements Provider {
|
|
|
402
545
|
|
|
403
546
|
signal?.throwIfAborted();
|
|
404
547
|
try {
|
|
405
|
-
const timeout =
|
|
548
|
+
const timeout = this.#fetchTimeoutMs > 0
|
|
549
|
+
? AbortSignal.timeout(this.#fetchTimeoutMs)
|
|
550
|
+
: undefined;
|
|
551
|
+
const requestSignal = signal === undefined
|
|
552
|
+
? timeout
|
|
553
|
+
: timeout === undefined
|
|
554
|
+
? signal
|
|
555
|
+
: AbortSignal.any([signal, timeout]);
|
|
406
556
|
const response = await this.#fetch(this.#promptTokensUrl, {
|
|
407
557
|
method: "POST",
|
|
408
558
|
headers: { "Content-Type": "application/json", ...this.#headers },
|
|
@@ -411,7 +561,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
411
561
|
messages,
|
|
412
562
|
...this.#reasoningBody(),
|
|
413
563
|
}),
|
|
414
|
-
|
|
564
|
+
...(requestSignal === undefined ? {} : { signal: requestSignal }),
|
|
415
565
|
});
|
|
416
566
|
if (!response.ok) {
|
|
417
567
|
return estimatePromptTokens(
|
|
@@ -439,23 +589,46 @@ export default class AiSdkProvider implements Provider {
|
|
|
439
589
|
);
|
|
440
590
|
}
|
|
441
591
|
}
|
|
442
|
-
calculateCost(usage: ProviderUsage): number { return this.#calculateCost(usage); }
|
|
443
|
-
calculateCharge(usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }> {
|
|
444
|
-
return this.#calculateCharge?.(usage)
|
|
445
|
-
?? { kind: "unknown", reason: "no provider rate or settled charge is available" };
|
|
446
|
-
}
|
|
447
592
|
|
|
593
|
+
async assessRequestCapacity(
|
|
594
|
+
messages: readonly ChatMessage[],
|
|
595
|
+
maxOutputTokens?: number,
|
|
596
|
+
signal?: AbortSignal,
|
|
597
|
+
): Promise<ProviderRequestCapacity> {
|
|
598
|
+
const outputBudget = effectiveOutputBudget({
|
|
599
|
+
requested: maxOutputTokens,
|
|
600
|
+
configured: this.#outputBudget,
|
|
601
|
+
maxOutputTokens: this.#maxOutputTokens,
|
|
602
|
+
contextWindow: this.#contextWindow,
|
|
603
|
+
});
|
|
604
|
+
const reasoningBudget = effectiveReasoningBudget({
|
|
605
|
+
configured: this.#reasoningBudget,
|
|
606
|
+
outputBudget,
|
|
607
|
+
});
|
|
608
|
+
return assessRequestCapacity({
|
|
609
|
+
contextWindow: this.#contextWindow,
|
|
610
|
+
maxInputTokens: this.#maxInputTokens,
|
|
611
|
+
maxOutputTokens: this.#maxOutputTokens,
|
|
612
|
+
outputBudget,
|
|
613
|
+
reasoningBudget,
|
|
614
|
+
measurement: await this.countPromptTokens(messages, signal),
|
|
615
|
+
});
|
|
616
|
+
}
|
|
448
617
|
// Reasoning activation and allowance are independent of grammar transport;
|
|
449
618
|
// only the response representation becomes lossless when evidence is needed.
|
|
450
619
|
// The llama-server template mapping is owned by {§llama-reasoning-request}.
|
|
451
|
-
#reasoningBody(
|
|
452
|
-
|
|
620
|
+
#reasoningBody(
|
|
621
|
+
preserveGrammarSentence = false,
|
|
622
|
+
reasoningBudget = this.#reasoningBudget,
|
|
623
|
+
): Record<string, unknown> {
|
|
624
|
+
const { mode } = this.#reasoning;
|
|
625
|
+
const budget = reasoningBudget;
|
|
453
626
|
const on = mode !== "off";
|
|
454
627
|
switch (this.#reasoningStyle) {
|
|
455
628
|
case "template": {
|
|
456
629
|
const allowance = mode === "off"
|
|
457
630
|
? 0
|
|
458
|
-
: mode === "on" ? budget : this
|
|
631
|
+
: mode === "on" && budget !== null ? budget : this.#reasoningBudget;
|
|
459
632
|
return {
|
|
460
633
|
chat_template_kwargs: { enable_thinking: on },
|
|
461
634
|
reasoning_format: preserveGrammarSentence ? "none" : "auto",
|
|
@@ -464,9 +637,9 @@ export default class AiSdkProvider implements Provider {
|
|
|
464
637
|
}
|
|
465
638
|
case "think": return on ? { think: true } : {};
|
|
466
639
|
case "include_reasoning": return on ? { include_reasoning: true } : {};
|
|
467
|
-
//
|
|
468
|
-
//
|
|
469
|
-
case "effort": return mode === "on" ? { reasoning_effort:
|
|
640
|
+
// Explicit on uses the portable enabled posture or a tier derived
|
|
641
|
+
// from an explicit budget; off/adaptive omit the field.
|
|
642
|
+
case "effort": return mode === "on" ? { reasoning_effort: effortFromReasoning({ mode, budget }) } : {};
|
|
470
643
|
// Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
|
|
471
644
|
// reason-by-default model (DeepSeek V4: default 'high') reasoning.
|
|
472
645
|
// ADAPTIVE omits the field: the backend's own default posture IS the
|
|
@@ -476,19 +649,25 @@ export default class AiSdkProvider implements Provider {
|
|
|
476
649
|
// efforts 400.
|
|
477
650
|
case "effort_explicit": return mode === "off"
|
|
478
651
|
? { reasoning_effort: "none" }
|
|
479
|
-
: mode === "on" ? { reasoning_effort:
|
|
652
|
+
: mode === "on" ? { reasoning_effort: effortFromReasoning({ mode, budget }) } : {};
|
|
480
653
|
// {§deepseek-reasoning-request}
|
|
481
654
|
case "thinking_effort": return mode === "off"
|
|
482
655
|
? { thinking: { type: "disabled" } }
|
|
483
656
|
: mode === "on" ? {
|
|
484
657
|
thinking: { type: "enabled" },
|
|
485
|
-
reasoning_effort: effortFromBudget(budget
|
|
658
|
+
...(budget === null ? {} : { reasoning_effort: effortFromBudget(budget) }),
|
|
486
659
|
} : {};
|
|
487
660
|
// Anthropic compat: explicit thinking object. off → disabled; on →
|
|
488
|
-
// enabled with
|
|
661
|
+
// enabled with the explicit reasoning subset; adaptive →
|
|
662
|
+
// omit (the API default).
|
|
489
663
|
case "anthropic": return mode === "off"
|
|
490
664
|
? { thinking: { type: "disabled" } }
|
|
491
|
-
: mode === "on" ? {
|
|
665
|
+
: mode === "on" ? {
|
|
666
|
+
thinking: {
|
|
667
|
+
type: "enabled",
|
|
668
|
+
budget_tokens: budget!,
|
|
669
|
+
},
|
|
670
|
+
} : {};
|
|
492
671
|
case "none": return {};
|
|
493
672
|
}
|
|
494
673
|
}
|
|
@@ -555,7 +734,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
555
734
|
}
|
|
556
735
|
}
|
|
557
736
|
|
|
558
|
-
// First-party telemetry headers ({§provider-request-authority}): forwarded only when the spec
|
|
737
|
+
// First-party telemetry headers ({§provider-request-authority} {§provider-call-kind}): forwarded only when the spec
|
|
559
738
|
// opted in (the plurnk endpoint). The gate is here, not at the call site, so
|
|
560
739
|
// attributions/client/strikes can never reach a third-party backend even if
|
|
561
740
|
// the consumer passes them to the wrong provider. Empty values emit no header
|
|
@@ -563,7 +742,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
563
742
|
// absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
|
|
564
743
|
// ride HTTP headers only — the packet never carries them (the model must
|
|
565
744
|
// never see strike state; engine accounting is not a metric to game).
|
|
566
|
-
#metadataHeaders(attributions: string[] | undefined, client: string | undefined, strikes: number | undefined, workerId: string, primaryWorkerId: string | undefined, workspaceId: string | undefined, loop: number | undefined, turn: number | undefined): Record<string, string> {
|
|
745
|
+
#metadataHeaders(attributions: string[] | undefined, client: string | undefined, strikes: number | undefined, workerId: string, primaryWorkerId: string | undefined, workspaceId: string | undefined, loop: number | undefined, turn: number | undefined, callKind: ProviderCallKind | undefined): Record<string, string> {
|
|
567
746
|
if (!this.#firstPartyMetadata) return {};
|
|
568
747
|
const h: Record<string, string> = {};
|
|
569
748
|
if (attributions !== undefined && attributions.length > 0) h["Plurnk-Attribution"] = JSON.stringify(attributions);
|
|
@@ -588,6 +767,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
588
767
|
if (workspaceId !== undefined && workspaceId.length > 0) h["Plurnk-Workspace-Id"] = workspaceId;
|
|
589
768
|
if (loop !== undefined && Number.isInteger(loop) && loop >= 1) h["Plurnk-Loop"] = String(loop);
|
|
590
769
|
if (turn !== undefined && Number.isInteger(turn) && turn >= 1) h["Plurnk-Turn"] = String(turn);
|
|
770
|
+
if (callKind !== undefined) h["Plurnk-Call-Kind"] = callKind;
|
|
591
771
|
return h;
|
|
592
772
|
}
|
|
593
773
|
|
|
@@ -627,9 +807,69 @@ export default class AiSdkProvider implements Provider {
|
|
|
627
807
|
return out;
|
|
628
808
|
}
|
|
629
809
|
|
|
630
|
-
|
|
810
|
+
#requestProviderOptions(
|
|
811
|
+
workerId: string,
|
|
812
|
+
reasoningBudget: number | null,
|
|
813
|
+
): AiSdkProviderOptions | undefined {
|
|
814
|
+
const responseOptions = this.#reasoning.mode === "off"
|
|
815
|
+
? undefined
|
|
816
|
+
: this.#reasoningResponseProviderOptions;
|
|
817
|
+
const nativeReasoning = this.#reasoning.mode === "on" && reasoningBudget !== null
|
|
818
|
+
? this.#additiveReasoningProvider === "anthropic"
|
|
819
|
+
? { anthropic: { thinking: { type: "enabled", budgetTokens: reasoningBudget } } }
|
|
820
|
+
: this.#additiveReasoningProvider === "bedrock"
|
|
821
|
+
? { bedrock: { reasoningConfig: { type: "enabled", budgetTokens: reasoningBudget } } }
|
|
822
|
+
: undefined
|
|
823
|
+
: undefined;
|
|
824
|
+
const options: AiSdkProviderOptions = {};
|
|
825
|
+
for (const part of [responseOptions, nativeReasoning]) {
|
|
826
|
+
for (const [provider, values] of Object.entries(part ?? {})) {
|
|
827
|
+
options[provider] = { ...options[provider], ...values };
|
|
828
|
+
}
|
|
829
|
+
}
|
|
830
|
+
if (this.#cacheAffinity?.target === "provider-option") {
|
|
831
|
+
const { provider, name } = this.#cacheAffinity;
|
|
832
|
+
options[provider] = { ...options[provider], [name]: workerId };
|
|
833
|
+
}
|
|
834
|
+
return Object.keys(options).length === 0 ? undefined : options;
|
|
835
|
+
}
|
|
836
|
+
|
|
837
|
+
#nativeMaxOutputTokens(
|
|
838
|
+
outputBudget: number | null,
|
|
839
|
+
reasoningBudget: number | null,
|
|
840
|
+
): number | undefined {
|
|
841
|
+
if (outputBudget === null) return undefined;
|
|
842
|
+
return this.#additiveReasoningProvider !== undefined
|
|
843
|
+
&& this.#reasoning.mode === "on"
|
|
844
|
+
&& reasoningBudget !== null
|
|
845
|
+
? outputBudget - reasoningBudget
|
|
846
|
+
: outputBudget;
|
|
847
|
+
}
|
|
848
|
+
|
|
849
|
+
#accounting(
|
|
850
|
+
outcome: ProviderRequestAccounting["outcome"],
|
|
851
|
+
usage: ProviderUsage | undefined,
|
|
852
|
+
evidence: Parameters<ProviderCostNormalizer>[0],
|
|
853
|
+
status?: number,
|
|
854
|
+
): ProviderRequestAccounting {
|
|
855
|
+
const knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
|
|
856
|
+
const direct = this.#normalizeCost?.(evidence);
|
|
857
|
+
return validateProviderRequestAccounting({
|
|
858
|
+
provider: this.#source,
|
|
859
|
+
model: this.#model,
|
|
860
|
+
outcome,
|
|
861
|
+
...(status === undefined ? {} : { status }),
|
|
862
|
+
...(knownUsage === undefined ? {} : { usage: knownUsage }),
|
|
863
|
+
cost: resolveProviderCost(direct, this.#estimateCost(knownUsage)),
|
|
864
|
+
});
|
|
865
|
+
}
|
|
866
|
+
|
|
867
|
+
async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxOutputTokens, attributions, client, strikes, workspaceId, loop, turn, sampling, observeRequest, callKind }: ProviderGenerateArgs): Promise<ProviderResponse> {
|
|
631
868
|
// {§provider-interface} The worker identity is required.
|
|
632
869
|
if (workerId === undefined || workerId.length === 0) throw new Error("generate: workerId is required — the worker's stable, opaque identity");
|
|
870
|
+
if (callKind !== undefined && callKind !== "emission" && callKind !== "bare") {
|
|
871
|
+
throw new Error(`generate: unsupported callKind ${JSON.stringify(callKind)}`);
|
|
872
|
+
}
|
|
633
873
|
// Reject before any wire call when already aborted
|
|
634
874
|
// ({§provider-failure-normalization}).
|
|
635
875
|
signal?.throwIfAborted();
|
|
@@ -642,6 +882,20 @@ export default class AiSdkProvider implements Provider {
|
|
|
642
882
|
const preserveGrammarSentence = wantGrammar
|
|
643
883
|
&& this.#reasoningStyle === "template";
|
|
644
884
|
|
|
885
|
+
const capacity = await this.assessRequestCapacity(messages, maxOutputTokens, signal);
|
|
886
|
+
if (capacity.decision === "reject") {
|
|
887
|
+
if (capacity.prompt.kind !== "exact") {
|
|
888
|
+
throw new TypeError(`${this.#source}: only an exact prompt measurement may reject capacity`);
|
|
889
|
+
}
|
|
890
|
+
throw new ProviderError(
|
|
891
|
+
this.#source,
|
|
892
|
+
"capacity_exceeded",
|
|
893
|
+
`The exact provider request uses ${capacity.prompt.tokens} input tokens, exceeding its ${capacity.inputCapacity} token input capacity.`,
|
|
894
|
+
{ capacity, extensions: { capacityStage: "preflight", capacity } },
|
|
895
|
+
);
|
|
896
|
+
}
|
|
897
|
+
const effectiveMaxOutputTokens = capacity.outputBudget ?? undefined;
|
|
898
|
+
|
|
645
899
|
// Assembly order = precedence: the family's sampling DEFAULTS
|
|
646
900
|
// (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
|
|
647
901
|
// paths and the name promises every request) < the caller's `sampling`
|
|
@@ -654,79 +908,195 @@ export default class AiSdkProvider implements Provider {
|
|
|
654
908
|
...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
|
|
655
909
|
model: this.#model,
|
|
656
910
|
messages,
|
|
657
|
-
...this.#reasoningBody(preserveGrammarSentence),
|
|
911
|
+
...this.#reasoningBody(preserveGrammarSentence, capacity.reasoningBudget),
|
|
658
912
|
...this.#grammarBody(sendGrammar),
|
|
659
|
-
...(
|
|
913
|
+
...(effectiveMaxOutputTokens !== undefined ? { max_tokens: effectiveMaxOutputTokens } : {}),
|
|
660
914
|
// Request per-token logprobs only when enabled (managed field —
|
|
661
915
|
// reserved from caller sampling; the env flag is the single control).
|
|
662
916
|
...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
|
|
663
917
|
...this.#slotBody(workerId),
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
|
|
667
|
-
...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
|
|
918
|
+
...(this.#cacheAffinity?.target === "body"
|
|
919
|
+
? { [this.#cacheAffinity.name]: workerId }
|
|
920
|
+
: {}),
|
|
668
921
|
};
|
|
669
922
|
|
|
670
923
|
// Per-request headers = static auth/routing + any first-party telemetry.
|
|
671
|
-
const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
|
|
672
|
-
const headers =
|
|
673
|
-
|
|
674
|
-
|
|
675
|
-
|
|
676
|
-
|
|
677
|
-
|
|
678
|
-
|
|
679
|
-
|
|
924
|
+
const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind);
|
|
925
|
+
const headers = new Headers(this.#headers);
|
|
926
|
+
if (this.#cacheAffinity?.target === "header") {
|
|
927
|
+
headers.set(this.#cacheAffinity.name, workerId);
|
|
928
|
+
}
|
|
929
|
+
for (const [name, value] of Object.entries(metaHeaders)) headers.set(name, value);
|
|
930
|
+
const requestHeaders = Object.fromEntries(headers.entries());
|
|
931
|
+
const accounting: ProviderRequestAccounting[] = [];
|
|
932
|
+
const operationTimeout = this.#operationTimeoutMs > 0
|
|
933
|
+
? AbortSignal.timeout(this.#operationTimeoutMs)
|
|
934
|
+
: undefined;
|
|
935
|
+
const operationSignal = signal === undefined
|
|
936
|
+
? operationTimeout
|
|
937
|
+
: operationTimeout === undefined
|
|
938
|
+
? signal
|
|
939
|
+
: AbortSignal.any([signal, operationTimeout]);
|
|
940
|
+
const executeRequest = async () => {
|
|
941
|
+
let settle: ProviderRequestSettlement | undefined;
|
|
942
|
+
try {
|
|
943
|
+
settle = await observeRequest?.({
|
|
944
|
+
provider: this.#source,
|
|
680
945
|
model: this.#model,
|
|
681
|
-
headers,
|
|
682
|
-
body,
|
|
683
|
-
messages,
|
|
684
|
-
signal,
|
|
685
|
-
fetch: this.#fetch,
|
|
686
|
-
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
687
|
-
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
688
|
-
retryAttempts: this.#retryAttempts,
|
|
689
|
-
streaming: this.#streaming,
|
|
690
|
-
captureRawBody: this.#rawBody,
|
|
691
|
-
})
|
|
692
|
-
: await executeAiSdkModel({
|
|
693
|
-
languageModel: this.#languageModel,
|
|
694
|
-
headers,
|
|
695
|
-
messages,
|
|
696
|
-
signal,
|
|
697
|
-
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
698
|
-
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
699
|
-
retryAttempts: this.#retryAttempts,
|
|
700
|
-
streaming: this.#streaming,
|
|
701
|
-
captureRawBody: this.#rawBody,
|
|
702
|
-
temperature: this.#tuningFloors
|
|
703
|
-
? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
|
|
704
|
-
: typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
|
|
705
|
-
topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
|
|
706
|
-
topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
|
|
707
|
-
presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
|
|
708
|
-
frequencyPenalty: typeof sampling?.frequency_penalty === "number"
|
|
709
|
-
? sampling.frequency_penalty
|
|
710
|
-
: this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
|
|
711
|
-
stopSequences: typeof sampling?.stop === "string"
|
|
712
|
-
? [sampling.stop]
|
|
713
|
-
: Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
|
|
714
|
-
? sampling.stop
|
|
715
|
-
: undefined,
|
|
716
|
-
seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
|
|
717
|
-
maxOutputTokens: maxTokens,
|
|
718
|
-
reasoning: this.#reasoning.mode === "off"
|
|
719
|
-
? "none"
|
|
720
|
-
: this.#reasoning.mode === "adaptive"
|
|
721
|
-
? "provider-default"
|
|
722
|
-
: effortFromBudget(this.#reasoning.budget!),
|
|
723
946
|
});
|
|
947
|
+
} catch (cause) {
|
|
948
|
+
throw new ProviderRequestObserverError(cause);
|
|
949
|
+
}
|
|
950
|
+
const settleAccounting = async (
|
|
951
|
+
outcome: ProviderRequestAccounting["outcome"],
|
|
952
|
+
usage: ProviderUsage | undefined,
|
|
953
|
+
evidence: Parameters<ProviderCostNormalizer>[0],
|
|
954
|
+
status?: number,
|
|
955
|
+
): Promise<ProviderRequestAccounting> => {
|
|
956
|
+
let requestAccounting: ProviderRequestAccounting;
|
|
957
|
+
let normalizationFailure: { cause: unknown } | undefined;
|
|
958
|
+
try {
|
|
959
|
+
requestAccounting = this.#accounting(outcome, usage, evidence, status);
|
|
960
|
+
} catch (cause) {
|
|
961
|
+
normalizationFailure = { cause };
|
|
962
|
+
let knownUsage: ProviderUsage | undefined;
|
|
963
|
+
try {
|
|
964
|
+
knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
|
|
965
|
+
} catch {
|
|
966
|
+
knownUsage = undefined;
|
|
967
|
+
}
|
|
968
|
+
requestAccounting = validateProviderRequestAccounting({
|
|
969
|
+
provider: this.#source,
|
|
970
|
+
model: this.#model,
|
|
971
|
+
outcome,
|
|
972
|
+
...(status === undefined ? {} : { status }),
|
|
973
|
+
...(knownUsage === undefined ? {} : { usage: knownUsage }),
|
|
974
|
+
cost: {
|
|
975
|
+
kind: "unknown",
|
|
976
|
+
reason: "provider request accounting could not be normalized after physical I/O",
|
|
977
|
+
},
|
|
978
|
+
});
|
|
979
|
+
}
|
|
980
|
+
accounting.push(requestAccounting);
|
|
981
|
+
try {
|
|
982
|
+
await settle?.(requestAccounting);
|
|
983
|
+
} catch (cause) {
|
|
984
|
+
throw new ProviderRequestObserverError(cause);
|
|
985
|
+
}
|
|
986
|
+
if (normalizationFailure !== undefined) {
|
|
987
|
+
throw new ProviderRequestAccountingError(normalizationFailure.cause);
|
|
988
|
+
}
|
|
989
|
+
return requestAccounting;
|
|
990
|
+
};
|
|
991
|
+
let response;
|
|
992
|
+
try {
|
|
993
|
+
response = this.#languageModel === undefined
|
|
994
|
+
? await executeOpenAICompatible({
|
|
995
|
+
url: this.#url!,
|
|
996
|
+
model: this.#model,
|
|
997
|
+
headers: requestHeaders,
|
|
998
|
+
body,
|
|
999
|
+
messages,
|
|
1000
|
+
signal: operationSignal,
|
|
1001
|
+
fetch: this.#fetch,
|
|
1002
|
+
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
1003
|
+
firstContentTimeoutMs: this.#firstContentTimeoutMs,
|
|
1004
|
+
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
1005
|
+
streaming: this.#streaming,
|
|
1006
|
+
captureRawBody: this.#rawBody,
|
|
1007
|
+
})
|
|
1008
|
+
: await executeAiSdkModel({
|
|
1009
|
+
languageModel: this.#languageModel,
|
|
1010
|
+
headers: requestHeaders,
|
|
1011
|
+
providerOptions: this.#requestProviderOptions(workerId, capacity.reasoningBudget),
|
|
1012
|
+
systemProviderOptions: this.#systemCacheProviderOptions,
|
|
1013
|
+
messages,
|
|
1014
|
+
signal: operationSignal,
|
|
1015
|
+
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
1016
|
+
firstContentTimeoutMs: this.#firstContentTimeoutMs,
|
|
1017
|
+
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
1018
|
+
streaming: this.#streaming,
|
|
1019
|
+
captureRawBody: this.#rawBody,
|
|
1020
|
+
temperature: this.#tuningFloors
|
|
1021
|
+
? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
|
|
1022
|
+
: typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
|
|
1023
|
+
topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
|
|
1024
|
+
topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
|
|
1025
|
+
presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
|
|
1026
|
+
frequencyPenalty: typeof sampling?.frequency_penalty === "number"
|
|
1027
|
+
? sampling.frequency_penalty
|
|
1028
|
+
: this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
|
|
1029
|
+
stopSequences: typeof sampling?.stop === "string"
|
|
1030
|
+
? [sampling.stop]
|
|
1031
|
+
: Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
|
|
1032
|
+
? sampling.stop
|
|
1033
|
+
: undefined,
|
|
1034
|
+
seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
|
|
1035
|
+
maxOutputTokens: this.#nativeMaxOutputTokens(capacity.outputBudget, capacity.reasoningBudget),
|
|
1036
|
+
reasoning: this.#reasoning.mode === "off"
|
|
1037
|
+
? "none"
|
|
1038
|
+
: this.#reasoning.mode === "adaptive"
|
|
1039
|
+
? "provider-default"
|
|
1040
|
+
: this.#additiveReasoningProvider !== undefined && capacity.reasoningBudget !== null
|
|
1041
|
+
? "provider-default"
|
|
1042
|
+
: effortFromReasoning({
|
|
1043
|
+
mode: this.#reasoning.mode,
|
|
1044
|
+
budget: capacity.reasoningBudget,
|
|
1045
|
+
}),
|
|
1046
|
+
});
|
|
1047
|
+
} catch (error) {
|
|
1048
|
+
const failure = transportFailureEvidence(error);
|
|
1049
|
+
await settleAccounting(
|
|
1050
|
+
"error",
|
|
1051
|
+
failure.usage,
|
|
1052
|
+
failure.chargeEvidence,
|
|
1053
|
+
failure.status,
|
|
1054
|
+
);
|
|
1055
|
+
throw error;
|
|
1056
|
+
}
|
|
1057
|
+
await settleAccounting(
|
|
1058
|
+
"response",
|
|
1059
|
+
response.usage,
|
|
1060
|
+
response.chargeEvidence,
|
|
1061
|
+
);
|
|
1062
|
+
return response;
|
|
1063
|
+
};
|
|
1064
|
+
|
|
1065
|
+
let raw;
|
|
1066
|
+
try {
|
|
1067
|
+
const { retry } = prepareRetries({
|
|
1068
|
+
maxRetries: this.#retryAttempts,
|
|
1069
|
+
abortSignal: operationSignal,
|
|
1070
|
+
});
|
|
1071
|
+
raw = await retry(executeRequest);
|
|
724
1072
|
} catch (err) {
|
|
1073
|
+
if (err instanceof ProviderRequestObserverError
|
|
1074
|
+
|| err instanceof ProviderRequestAccountingError) throw err.cause;
|
|
725
1075
|
if (signal?.aborted) throw err;
|
|
726
|
-
|
|
1076
|
+
if (operationTimeout?.aborted) {
|
|
1077
|
+
const timeout = new ProviderTimeoutError("operation", this.#operationTimeoutMs, err);
|
|
1078
|
+
throw new ProviderError(this.#source, "deadline_exceeded", timeout.message, {
|
|
1079
|
+
status: 504,
|
|
1080
|
+
cause: timeout,
|
|
1081
|
+
retryable: false,
|
|
1082
|
+
extensions: {
|
|
1083
|
+
timeoutPhase: timeout.phase,
|
|
1084
|
+
timeoutMs: timeout.timeoutMs,
|
|
1085
|
+
},
|
|
1086
|
+
accounting,
|
|
1087
|
+
capacity,
|
|
1088
|
+
});
|
|
1089
|
+
}
|
|
1090
|
+
const pe = toProviderError(err, this.#source, this.#errorDetailLimit, capacity);
|
|
727
1091
|
if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
|
|
728
|
-
throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, {
|
|
1092
|
+
throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, {
|
|
1093
|
+
status: pe.status,
|
|
1094
|
+
cause: err,
|
|
1095
|
+
accounting,
|
|
1096
|
+
capacity,
|
|
1097
|
+
});
|
|
729
1098
|
}
|
|
1099
|
+
pe.prependAccounting(accounting);
|
|
730
1100
|
throw pe;
|
|
731
1101
|
}
|
|
732
1102
|
|
|
@@ -747,10 +1117,10 @@ export default class AiSdkProvider implements Provider {
|
|
|
747
1117
|
this.#reasoningResponseStyle,
|
|
748
1118
|
);
|
|
749
1119
|
|
|
750
|
-
// Preserve the exact
|
|
751
|
-
//
|
|
752
|
-
//
|
|
753
|
-
// supply independent
|
|
1120
|
+
// Preserve the exact pre-projection response. Constrained template turns
|
|
1121
|
+
// request `reasoning_format: "none"`, so even an empty channel and any
|
|
1122
|
+
// template-provided opener remain observable. An unexpectedly projected
|
|
1123
|
+
// response cannot supply independent evidence.
|
|
754
1124
|
let grammarEvidence: GrammarEvidence | undefined;
|
|
755
1125
|
if (wantGrammar) {
|
|
756
1126
|
if (preserveGrammarSentence) {
|
|
@@ -779,12 +1149,13 @@ export default class AiSdkProvider implements Provider {
|
|
|
779
1149
|
if (projectedReasoning.projected) {
|
|
780
1150
|
raw.content = projectedReasoning.content;
|
|
781
1151
|
raw.reasoning = projectedReasoning.reasoning;
|
|
782
|
-
raw.usage = attributeUnitemizedReasoning(raw.usage, raw.reasoning, raw.content);
|
|
783
1152
|
}
|
|
784
1153
|
|
|
785
1154
|
let notices: ProviderNotice[] | undefined;
|
|
786
1155
|
const usage = raw.usage;
|
|
787
|
-
if (sendGrammar !== undefined
|
|
1156
|
+
if (sendGrammar !== undefined
|
|
1157
|
+
&& this.tokenize !== undefined
|
|
1158
|
+
&& usage?.outputTokens !== undefined) {
|
|
788
1159
|
// Channel-escape detector: completion tokens
|
|
789
1160
|
// billed far beyond every visible channel mean the decode ESCAPED into
|
|
790
1161
|
// a server-discarded reasoning block mid-emission. This diagnostic
|
|
@@ -795,12 +1166,12 @@ export default class AiSdkProvider implements Provider {
|
|
|
795
1166
|
this.tokenize(raw.reasoning),
|
|
796
1167
|
]);
|
|
797
1168
|
const visible = contentTokens.length + reasoningTokens.length;
|
|
798
|
-
if (usage.
|
|
1169
|
+
if (usage.outputTokens > visible + 64) {
|
|
799
1170
|
(notices ??= []).push({
|
|
800
1171
|
source: this.#source,
|
|
801
1172
|
kind: "grammar_unenforced",
|
|
802
1173
|
level: "warn",
|
|
803
|
-
message: `decode escaped the grammar: ${usage.
|
|
1174
|
+
message: `decode escaped the grammar: ${usage.outputTokens} output tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
|
|
804
1175
|
position: [...raw.content].length,
|
|
805
1176
|
});
|
|
806
1177
|
}
|
|
@@ -824,22 +1195,40 @@ export default class AiSdkProvider implements Provider {
|
|
|
824
1195
|
...(raw.reasoningEncrypted.length > 0
|
|
825
1196
|
? { reasoningEncrypted: raw.reasoningEncrypted }
|
|
826
1197
|
: {}),
|
|
827
|
-
usage,
|
|
828
1198
|
model: raw.model,
|
|
829
1199
|
...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
|
|
830
1200
|
};
|
|
831
|
-
const normalizedCharge = this.#normalizeCharge?.(raw.chargeEvidence);
|
|
832
|
-
const charge = normalizedCharge === undefined
|
|
833
|
-
? undefined
|
|
834
|
-
: validateAuthoritativeCharge(normalizedCharge);
|
|
835
1201
|
const evidence = {
|
|
836
1202
|
assistantRaw: raw,
|
|
837
|
-
|
|
1203
|
+
accounting,
|
|
1204
|
+
capacity,
|
|
838
1205
|
...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
|
|
839
1206
|
...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
|
|
840
1207
|
...(meta !== undefined ? { meta } : {}),
|
|
841
1208
|
...(notices !== undefined ? { notices } : {}),
|
|
842
1209
|
};
|
|
1210
|
+
if (capacity.outputBudget !== null
|
|
1211
|
+
&& usage?.outputTokens !== undefined
|
|
1212
|
+
&& usage.outputTokens > capacity.outputBudget) {
|
|
1213
|
+
const attempt: ProviderAttempt = {
|
|
1214
|
+
assistant: { ...assistant, finishReason: raw.finishReason },
|
|
1215
|
+
...evidence,
|
|
1216
|
+
};
|
|
1217
|
+
throw new ProviderError(
|
|
1218
|
+
this.#source,
|
|
1219
|
+
"invalid_response",
|
|
1220
|
+
`The provider reported ${usage.outputTokens} output tokens after receiving a total output budget of ${capacity.outputBudget}.`,
|
|
1221
|
+
{
|
|
1222
|
+
attempt,
|
|
1223
|
+
accounting,
|
|
1224
|
+
extensions: {
|
|
1225
|
+
stage: "provider-response",
|
|
1226
|
+
outputBudget: capacity.outputBudget,
|
|
1227
|
+
reportedOutputTokens: usage.outputTokens,
|
|
1228
|
+
},
|
|
1229
|
+
},
|
|
1230
|
+
);
|
|
1231
|
+
}
|
|
843
1232
|
if (raw.finishReason === "resource_interrupted") {
|
|
844
1233
|
const attempt: ProviderResponse<"resource_interrupted"> = {
|
|
845
1234
|
assistant: { ...assistant, finishReason: raw.finishReason },
|
|
@@ -851,6 +1240,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
851
1240
|
"The provider interrupted generation because inference resources were unavailable.",
|
|
852
1241
|
{
|
|
853
1242
|
attempt,
|
|
1243
|
+
accounting,
|
|
854
1244
|
extensions: {
|
|
855
1245
|
stage: "provider-response",
|
|
856
1246
|
finishReason: "resource_interrupted",
|