@plurnk/plurnk-providers 1.4.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +40 -34
- package/README.md +3 -0
- package/SPEC.md +153 -62
- package/dist/AiSdkProvider.d.ts +19 -25
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +353 -120
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +7 -13
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +36 -8
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +2 -21
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +19 -14
- package/dist/Pool.js.map +1 -1
- package/dist/accounting.d.ts +6 -0
- package/dist/accounting.d.ts.map +1 -0
- package/dist/accounting.js +168 -0
- package/dist/accounting.js.map +1 -0
- package/dist/aiSdkTransport.d.ts +11 -3
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +198 -29
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +7 -2
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +32 -26
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +18 -7
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +10 -10
- package/dist/cost.d.ts.map +1 -1
- package/dist/cost.js +88 -43
- package/dist/cost.js.map +1 -1
- package/dist/env.d.ts +5 -7
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +30 -32
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +14 -2
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +60 -2
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +4 -4
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -2
- package/dist/index.js.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/sdkModels.d.ts +6 -0
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +46 -3
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +40 -29
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +21 -4
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +188 -74
- package/dist/usage.js.map +1 -1
- package/package.json +9 -7
- package/src/AiSdkProvider.test.ts +1039 -182
- package/src/AiSdkProvider.ts +428 -141
- package/src/Mock.test.ts +37 -12
- package/src/Mock.ts +46 -12
- package/src/Pool.test.ts +19 -6
- package/src/Pool.ts +20 -16
- package/src/ProviderRegistry.test.ts +16 -11
- package/src/accounting.test.ts +94 -0
- package/src/accounting.ts +190 -0
- package/src/aiSdkTransport.test.ts +42 -49
- package/src/aiSdkTransport.ts +218 -32
- package/src/boundaries.test.ts +2 -0
- package/src/catalogProvider.test.ts +271 -24
- package/src/catalogProvider.ts +44 -28
- package/src/compatibleProvider.test.ts +6 -3
- package/src/compatibleProvider.ts +20 -7
- package/src/cost.test.ts +55 -35
- package/src/cost.ts +110 -54
- package/src/defaults.test.ts +13 -3
- package/src/env.test.ts +50 -26
- package/src/env.ts +43 -42
- package/src/errors.test.ts +47 -2
- package/src/errors.ts +68 -3
- package/src/index.ts +21 -5
- package/src/ollama.test.ts +4 -1
- package/src/ollama.ts +3 -3
- package/src/sdkModels.test.ts +94 -3
- package/src/sdkModels.ts +53 -3
- package/src/types.ts +91 -33
- package/src/usage.test.ts +112 -108
- package/src/usage.ts +233 -84
package/src/AiSdkProvider.ts
CHANGED
|
@@ -6,18 +6,40 @@
|
|
|
6
6
|
// ordinary vendor protocol. The compatible URL path remains only for PLURNK
|
|
7
7
|
// extensions and local endpoint probes the SDK cannot represent.
|
|
8
8
|
|
|
9
|
-
import type {
|
|
9
|
+
import type {
|
|
10
|
+
ChatMessage,
|
|
11
|
+
GrammarEvidence,
|
|
12
|
+
PromptTokenMeasurement,
|
|
13
|
+
Provider,
|
|
14
|
+
ProviderCostNormalizer,
|
|
15
|
+
ProviderCallKind,
|
|
16
|
+
ProviderGenerateArgs,
|
|
17
|
+
ProviderRequestAccounting,
|
|
18
|
+
ProviderRequestObserver,
|
|
19
|
+
ProviderRequestSettlement,
|
|
20
|
+
ProviderResponse,
|
|
21
|
+
ProviderUsage,
|
|
22
|
+
} from "./types.ts";
|
|
10
23
|
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
24
|
+
import type { JSONValue } from "ai";
|
|
25
|
+
import { MAX_PROVIDER_TIMEOUT_MS } from "./env.ts";
|
|
11
26
|
import type { Reasoning, ReasoningResponseStyle, ReserveSpec } from "./env.ts";
|
|
12
|
-
import {
|
|
27
|
+
import {
|
|
28
|
+
executeAiSdkModel,
|
|
29
|
+
executeOpenAICompatible,
|
|
30
|
+
transportFailureEvidence,
|
|
31
|
+
} from "./aiSdkTransport.ts";
|
|
13
32
|
import type { LanguageModel } from "ai";
|
|
14
|
-
import {
|
|
15
|
-
import {
|
|
33
|
+
import { prepareRetries } from "ai/internal";
|
|
34
|
+
import { toProviderError, ProviderError, ProviderTimeoutError } from "./errors.ts";
|
|
16
35
|
import type { ProviderNotice } from "./notices.ts";
|
|
17
36
|
import { validateGbnf } from "@plurnk/gbnf";
|
|
18
37
|
import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.ts";
|
|
19
38
|
import { emitWarningOnce } from "./warnings.ts";
|
|
20
39
|
import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
|
|
40
|
+
import { resolveProviderCost } from "./cost.ts";
|
|
41
|
+
import { validateProviderRequestAccounting } from "./accounting.ts";
|
|
42
|
+
import { validateProviderUsage } from "./usage.ts";
|
|
21
43
|
|
|
22
44
|
export type ProviderFetch = typeof globalThis.fetch;
|
|
23
45
|
|
|
@@ -29,29 +51,40 @@ export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" |
|
|
|
29
51
|
// service-managed constrained sampling; endpoint-owned settings are not inferred.
|
|
30
52
|
export type GrammarStyle = "none" | "llamacpp";
|
|
31
53
|
|
|
54
|
+
export type CacheAffinity =
|
|
55
|
+
| { readonly target: "header" | "body"; readonly name: string }
|
|
56
|
+
| { readonly target: "provider-option"; readonly provider: string; readonly name: string };
|
|
57
|
+
|
|
58
|
+
export type AiSdkProviderOptions = Record<string, Record<string, JSONValue | undefined>>;
|
|
59
|
+
|
|
32
60
|
export type AiSdkProviderConfig = {
|
|
33
61
|
model: string;
|
|
34
62
|
url?: string; // OpenAI-compatible chat-completions URL
|
|
35
63
|
languageModel?: LanguageModel; // native AI SDK provider model
|
|
36
64
|
attributions?: (context: PluginAttributionContext) => PluginAttribution;
|
|
37
|
-
fetchTimeoutMs: number;
|
|
38
|
-
|
|
65
|
+
fetchTimeoutMs: number; // one physical generation attempt; zero disables
|
|
66
|
+
operationTimeoutMs: number; // complete logical call across retries/backoff; zero disables
|
|
67
|
+
firstContentTimeoutMs: number; // first semantic streamed content; zero disables
|
|
68
|
+
streamIdleTimeoutMs?: number; // semantic streamed-content idle deadline; zero/unset disables
|
|
39
69
|
headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
|
|
40
70
|
fetch?: ProviderFetch; // per-instance request executor; default globalThis.fetch
|
|
41
71
|
contextWindow?: number | null; // default null; caller resolves-or-fails, narrows to required with the interface
|
|
42
72
|
reasoningStyle?: ReasoningStyle; // default "none"
|
|
43
73
|
reasoningResponseStyle?: ReasoningResponseStyle; // {§provider-tagged-reasoning}; default "verbatim"
|
|
44
74
|
countPromptTokens?: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
|
|
45
|
-
|
|
46
|
-
|
|
75
|
+
estimateCost?: (usage: ProviderUsage | undefined) => ProviderCost;
|
|
76
|
+
normalizeCost?: ProviderCostNormalizer;
|
|
47
77
|
source?: string; // notice/problem source, e.g. "provider:openai"; default "provider"
|
|
48
78
|
grammarStyle?: GrammarStyle; // how a GBNF grammar is carried; default "none" (not sent)
|
|
49
|
-
//
|
|
50
|
-
//
|
|
51
|
-
|
|
52
|
-
//
|
|
53
|
-
//
|
|
54
|
-
|
|
79
|
+
// {§provider-cache-affinity} Provider routes own the exact documented
|
|
80
|
+
// projection; the common transport only applies it as managed request state.
|
|
81
|
+
cacheAffinity?: CacheAffinity;
|
|
82
|
+
// {§provider-cache-write-policy} Already policy-gated by provider construction.
|
|
83
|
+
// The transport attaches it to only the final leading system instruction.
|
|
84
|
+
systemCacheProviderOptions?: AiSdkProviderOptions;
|
|
85
|
+
// {§provider-readable-reasoning} Route-owned native option needed to expose
|
|
86
|
+
// readable reasoning. Applied only when the effective posture is not off.
|
|
87
|
+
reasoningResponseProviderOptions?: AiSdkProviderOptions;
|
|
55
88
|
// Optional provider-configured service tier. Unlike caller sampling, this is
|
|
56
89
|
// a fixed deployment choice and therefore wins on every request.
|
|
57
90
|
serviceTier?: string;
|
|
@@ -81,9 +114,9 @@ export type AiSdkProviderConfig = {
|
|
|
81
114
|
requiresMaxTokens?: boolean;
|
|
82
115
|
// The side-channel reasoning intent — REQUIRED, no in-code default
|
|
83
116
|
// (PLURNK_PROVIDERS_REASONING + _BUDGET, read via reasoningFromEnv):
|
|
84
|
-
// { mode: off|adaptive|on, budget:
|
|
85
|
-
// backend's mechanism via reasoningStyle; budget is only ever
|
|
86
|
-
// never a hidden activation flag.
|
|
117
|
+
// { mode: off|adaptive|on, budget: optional when on }. The provider maps it
|
|
118
|
+
// to the backend's mechanism via reasoningStyle; budget is only ever an
|
|
119
|
+
// explicit magnitude, never a hidden activation flag.
|
|
87
120
|
reasoning: Reasoning;
|
|
88
121
|
// Decode tuning: no in-code defaults; the canonical measured values (0.2 /
|
|
89
122
|
// 1.15) ship as the floor in .env.defaults (alias-scopable). `temperature` is the
|
|
@@ -138,6 +171,20 @@ export type AiSdkProviderConfig = {
|
|
|
138
171
|
tuningFloors?: boolean;
|
|
139
172
|
};
|
|
140
173
|
|
|
174
|
+
class ProviderRequestObserverError extends Error {
|
|
175
|
+
constructor(cause: unknown) {
|
|
176
|
+
super("provider request accounting could not be durably settled", { cause });
|
|
177
|
+
this.name = "ProviderRequestObserverError";
|
|
178
|
+
}
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
class ProviderRequestAccountingError extends Error {
|
|
182
|
+
constructor(cause: unknown) {
|
|
183
|
+
super("provider request accounting could not be normalized", { cause });
|
|
184
|
+
this.name = "ProviderRequestAccountingError";
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
|
|
141
188
|
// Drop trailing occurrences of a server-rendered EOG marker. llama-server
|
|
142
189
|
// under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
|
|
143
190
|
// trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
|
|
@@ -157,19 +204,15 @@ type TaggedReasoningProjection = {
|
|
|
157
204
|
readonly contentStart: number;
|
|
158
205
|
};
|
|
159
206
|
|
|
160
|
-
|
|
161
|
-
// one exact leading envelope. Parsing after stream assembly keeps SSE and JSON
|
|
162
|
-
// on one path and leaves later literal tags in the visible suffix untouched.
|
|
163
|
-
const projectTaggedReasoning = (
|
|
207
|
+
const projectLeadingReasoning = (
|
|
164
208
|
content: string,
|
|
165
209
|
structuredReasoning: string,
|
|
166
|
-
|
|
210
|
+
opening: string,
|
|
211
|
+
closing: string,
|
|
167
212
|
): TaggedReasoningProjection => {
|
|
168
|
-
|
|
169
|
-
if (style !== "think-tags" || structuredReasoning.length > 0 || !content.startsWith(opening)) {
|
|
213
|
+
if (structuredReasoning.length > 0 || !content.startsWith(opening)) {
|
|
170
214
|
return { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
|
|
171
215
|
}
|
|
172
|
-
const closing = "</think>";
|
|
173
216
|
const closingIndex = content.indexOf(closing, opening.length);
|
|
174
217
|
if (closingIndex === -1) {
|
|
175
218
|
return {
|
|
@@ -188,6 +231,31 @@ const projectTaggedReasoning = (
|
|
|
188
231
|
};
|
|
189
232
|
};
|
|
190
233
|
|
|
234
|
+
// {§provider-tagged-reasoning} Only the model-contract position is structural:
|
|
235
|
+
// one exact leading envelope. Parsing after stream assembly keeps SSE and JSON
|
|
236
|
+
// on one path and leaves later literal tags in the visible suffix untouched.
|
|
237
|
+
const projectTaggedReasoning = (
|
|
238
|
+
content: string,
|
|
239
|
+
structuredReasoning: string,
|
|
240
|
+
style: ReasoningResponseStyle,
|
|
241
|
+
): TaggedReasoningProjection => style === "think-tags"
|
|
242
|
+
? projectLeadingReasoning(content, structuredReasoning, "<think>", "</think>")
|
|
243
|
+
: { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
|
|
244
|
+
|
|
245
|
+
// llama-server's template reasoning parser can project either supported leading
|
|
246
|
+
// reasoning envelope out of the OpenAI-compatible response. Grammar evidence
|
|
247
|
+
// needs the sentence before that lossy projection, so constrained template turns
|
|
248
|
+
// request it verbatim and split the observed enclosure here.
|
|
249
|
+
const projectTemplateReasoning = (content: string): TaggedReasoningProjection => {
|
|
250
|
+
for (const [opening, closing] of [
|
|
251
|
+
["<|channel>thought\n", "<channel|>"],
|
|
252
|
+
["<think>\n", "</think>"],
|
|
253
|
+
] as const) {
|
|
254
|
+
if (content.startsWith(opening)) return projectLeadingReasoning(content, "", opening, closing);
|
|
255
|
+
}
|
|
256
|
+
return { content, reasoning: "", projected: false, contentStart: 0 };
|
|
257
|
+
};
|
|
258
|
+
|
|
191
259
|
// Shared budget→effort breakpoints (xai and google had identical copies).
|
|
192
260
|
export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
|
|
193
261
|
if (budget <= 1000) return "low";
|
|
@@ -195,6 +263,13 @@ export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
|
|
|
195
263
|
return "high";
|
|
196
264
|
};
|
|
197
265
|
|
|
266
|
+
// AI SDK's portable reasoning control has no boolean-enabled value. `medium`
|
|
267
|
+
// is the neutral activation projection for an explicit, unqualified `on`; it
|
|
268
|
+
// changes no PLURNK token reserve. An operator budget, when present, remains
|
|
269
|
+
// the only input to the existing magnitude-to-tier projection.
|
|
270
|
+
const effortFromReasoning = (reasoning: Reasoning): "low" | "medium" | "high" =>
|
|
271
|
+
reasoning.budget === null ? "medium" : effortFromBudget(reasoning.budget);
|
|
272
|
+
|
|
198
273
|
// Body keys the provider owns — a caller's `sampling` passthrough may not set
|
|
199
274
|
// these. Two families:
|
|
200
275
|
// transport/managed — grammar transport, the stream/JSON choice, slot pinning,
|
|
@@ -222,6 +297,8 @@ export default class AiSdkProvider implements Provider {
|
|
|
222
297
|
#url: string | undefined;
|
|
223
298
|
#languageModel: LanguageModel | undefined;
|
|
224
299
|
#fetchTimeoutMs: number;
|
|
300
|
+
#operationTimeoutMs: number;
|
|
301
|
+
#firstContentTimeoutMs: number;
|
|
225
302
|
#streamIdleTimeoutMs: number | undefined;
|
|
226
303
|
#headers: Record<string, string>;
|
|
227
304
|
#fetch: ProviderFetch;
|
|
@@ -241,11 +318,13 @@ export default class AiSdkProvider implements Provider {
|
|
|
241
318
|
#reasoningResponseStyle: ReasoningResponseStyle;
|
|
242
319
|
#countPromptTokens: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
|
|
243
320
|
#promptTokensUrl: string | undefined;
|
|
244
|
-
#
|
|
245
|
-
#
|
|
321
|
+
#estimateCost: (usage: ProviderUsage | undefined) => ProviderCost;
|
|
322
|
+
#normalizeCost?: ProviderCostNormalizer;
|
|
246
323
|
#source: string;
|
|
247
324
|
#grammarStyle: GrammarStyle;
|
|
248
|
-
#
|
|
325
|
+
#cacheAffinity: CacheAffinity | undefined;
|
|
326
|
+
#systemCacheProviderOptions: AiSdkProviderOptions | undefined;
|
|
327
|
+
#reasoningResponseProviderOptions: AiSdkProviderOptions | undefined;
|
|
249
328
|
#serviceTier: string | undefined;
|
|
250
329
|
#gbnfDebug: boolean;
|
|
251
330
|
#streaming: boolean;
|
|
@@ -268,7 +347,6 @@ export default class AiSdkProvider implements Provider {
|
|
|
268
347
|
// tokenizeUrl (llama-server), so `provider.tokenize === undefined` remains
|
|
269
348
|
// the honest capability signal for every other backend.
|
|
270
349
|
tokenize?: (text: string) => Promise<number[]>;
|
|
271
|
-
|
|
272
350
|
constructor(config: AiSdkProviderConfig) {
|
|
273
351
|
this.#model = config.model;
|
|
274
352
|
this.#url = config.url;
|
|
@@ -277,7 +355,19 @@ export default class AiSdkProvider implements Provider {
|
|
|
277
355
|
if ((this.#url === undefined) === (this.#languageModel === undefined)) {
|
|
278
356
|
throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
|
|
279
357
|
}
|
|
358
|
+
for (const [name, value] of [
|
|
359
|
+
["fetchTimeoutMs", config.fetchTimeoutMs],
|
|
360
|
+
["operationTimeoutMs", config.operationTimeoutMs],
|
|
361
|
+
["firstContentTimeoutMs", config.firstContentTimeoutMs],
|
|
362
|
+
["streamIdleTimeoutMs", config.streamIdleTimeoutMs],
|
|
363
|
+
] as const) {
|
|
364
|
+
if (value !== undefined && (!Number.isInteger(value) || value < 0 || value > MAX_PROVIDER_TIMEOUT_MS)) {
|
|
365
|
+
throw new Error(`${config.source ?? "provider"}: ${name} must be an integer from 0 through ${MAX_PROVIDER_TIMEOUT_MS} milliseconds`);
|
|
366
|
+
}
|
|
367
|
+
}
|
|
280
368
|
this.#fetchTimeoutMs = config.fetchTimeoutMs;
|
|
369
|
+
this.#operationTimeoutMs = config.operationTimeoutMs;
|
|
370
|
+
this.#firstContentTimeoutMs = config.firstContentTimeoutMs;
|
|
281
371
|
this.#streamIdleTimeoutMs = config.streamIdleTimeoutMs;
|
|
282
372
|
this.#headers = config.headers ?? {};
|
|
283
373
|
this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
|
|
@@ -305,11 +395,36 @@ export default class AiSdkProvider implements Provider {
|
|
|
305
395
|
}
|
|
306
396
|
this.#countPromptTokens = config.countPromptTokens ?? ((messages) => estimatePromptTokens(messages));
|
|
307
397
|
this.#promptTokensUrl = config.promptTokensUrl;
|
|
308
|
-
this.#
|
|
309
|
-
|
|
398
|
+
this.#estimateCost = config.estimateCost
|
|
399
|
+
?? (() => ({
|
|
400
|
+
kind: "unknown",
|
|
401
|
+
reason: "the request reported no direct cost and no model rate is configured",
|
|
402
|
+
}));
|
|
403
|
+
this.#normalizeCost = config.normalizeCost;
|
|
310
404
|
this.#source = config.source ?? "provider";
|
|
311
405
|
this.#grammarStyle = config.grammarStyle ?? "none";
|
|
312
|
-
this.#
|
|
406
|
+
this.#cacheAffinity = config.cacheAffinity;
|
|
407
|
+
this.#systemCacheProviderOptions = config.systemCacheProviderOptions;
|
|
408
|
+
this.#reasoningResponseProviderOptions = config.reasoningResponseProviderOptions;
|
|
409
|
+
if (this.#languageModel === undefined && this.#cacheAffinity?.target === "provider-option") {
|
|
410
|
+
throw new Error(`${this.#source}: native provider-option cache affinity requires an AI SDK model`);
|
|
411
|
+
}
|
|
412
|
+
if (this.#languageModel !== undefined && this.#cacheAffinity?.target === "body") {
|
|
413
|
+
throw new Error(`${this.#source}: body cache affinity requires an OpenAI-compatible URL`);
|
|
414
|
+
}
|
|
415
|
+
if (this.#languageModel === undefined && this.#systemCacheProviderOptions !== undefined) {
|
|
416
|
+
throw new Error(`${this.#source}: system cache provider options require an AI SDK model`);
|
|
417
|
+
}
|
|
418
|
+
if (this.#languageModel === undefined && this.#reasoningResponseProviderOptions !== undefined) {
|
|
419
|
+
throw new Error(`${this.#source}: reasoning response provider options require an AI SDK model`);
|
|
420
|
+
}
|
|
421
|
+
if (this.#cacheAffinity?.target === "provider-option"
|
|
422
|
+
&& Object.hasOwn(
|
|
423
|
+
this.#reasoningResponseProviderOptions?.[this.#cacheAffinity.provider] ?? {},
|
|
424
|
+
this.#cacheAffinity.name,
|
|
425
|
+
)) {
|
|
426
|
+
throw new Error(`${this.#source}: reasoning response options conflict with cache-affinity option ${this.#cacheAffinity.provider}.${this.#cacheAffinity.name}`);
|
|
427
|
+
}
|
|
313
428
|
this.#serviceTier = config.serviceTier;
|
|
314
429
|
this.#gbnfDebug = config.gbnfDebug ?? false;
|
|
315
430
|
this.#streaming = config.streaming ?? true;
|
|
@@ -329,10 +444,17 @@ export default class AiSdkProvider implements Provider {
|
|
|
329
444
|
const reasoningReserve = this.reasoningReserve;
|
|
330
445
|
if (this.#reasoningStyle === "template"
|
|
331
446
|
&& this.#reasoning.mode === "on"
|
|
447
|
+
&& this.#reasoning.budget !== null
|
|
332
448
|
&& reasoningReserve !== null
|
|
333
|
-
&& this.#reasoning.budget
|
|
449
|
+
&& this.#reasoning.budget > reasoningReserve) {
|
|
334
450
|
throw new Error(`${this.#source}: PLURNK_PROVIDERS_REASONING_BUDGET (${this.#reasoning.budget}) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE (${reasoningReserve})`);
|
|
335
451
|
}
|
|
452
|
+
if (this.#reasoningStyle === "anthropic"
|
|
453
|
+
&& this.#reasoning.mode === "on"
|
|
454
|
+
&& this.#reasoning.budget === null
|
|
455
|
+
&& reasoningReserve === null) {
|
|
456
|
+
throw new Error(`${this.#source}: explicit Anthropic reasoning requires a resolved reasoning reserve or PLURNK_PROVIDERS_REASONING_BUDGET`);
|
|
457
|
+
}
|
|
336
458
|
const { tokenizeUrl } = config;
|
|
337
459
|
if (tokenizeUrl !== undefined) {
|
|
338
460
|
this.tokenize = async (text: string): Promise<number[]> => {
|
|
@@ -340,7 +462,9 @@ export default class AiSdkProvider implements Provider {
|
|
|
340
462
|
method: "POST",
|
|
341
463
|
headers: { "Content-Type": "application/json", ...this.#headers },
|
|
342
464
|
body: JSON.stringify({ content: text }),
|
|
343
|
-
|
|
465
|
+
...(this.#fetchTimeoutMs > 0
|
|
466
|
+
? { signal: AbortSignal.timeout(this.#fetchTimeoutMs) }
|
|
467
|
+
: {}),
|
|
344
468
|
});
|
|
345
469
|
if (!res.ok) throw new Error(`${this.#source}: tokenize endpoint returned ${res.status}`);
|
|
346
470
|
const { tokens } = (await res.json()) as { tokens?: unknown };
|
|
@@ -385,7 +509,14 @@ export default class AiSdkProvider implements Provider {
|
|
|
385
509
|
|
|
386
510
|
signal?.throwIfAborted();
|
|
387
511
|
try {
|
|
388
|
-
const timeout =
|
|
512
|
+
const timeout = this.#fetchTimeoutMs > 0
|
|
513
|
+
? AbortSignal.timeout(this.#fetchTimeoutMs)
|
|
514
|
+
: undefined;
|
|
515
|
+
const requestSignal = signal === undefined
|
|
516
|
+
? timeout
|
|
517
|
+
: timeout === undefined
|
|
518
|
+
? signal
|
|
519
|
+
: AbortSignal.any([signal, timeout]);
|
|
389
520
|
const response = await this.#fetch(this.#promptTokensUrl, {
|
|
390
521
|
method: "POST",
|
|
391
522
|
headers: { "Content-Type": "application/json", ...this.#headers },
|
|
@@ -394,7 +525,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
394
525
|
messages,
|
|
395
526
|
...this.#reasoningBody(),
|
|
396
527
|
}),
|
|
397
|
-
|
|
528
|
+
...(requestSignal === undefined ? {} : { signal: requestSignal }),
|
|
398
529
|
});
|
|
399
530
|
if (!response.ok) {
|
|
400
531
|
return estimatePromptTokens(
|
|
@@ -422,33 +553,28 @@ export default class AiSdkProvider implements Provider {
|
|
|
422
553
|
);
|
|
423
554
|
}
|
|
424
555
|
}
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
}
|
|
430
|
-
|
|
431
|
-
// Reasoning intent maps independently of grammar transport. The llama-server
|
|
432
|
-
// template mapping is owned by {§llama-reasoning-request}.
|
|
433
|
-
#reasoningBody(): Record<string, unknown> {
|
|
556
|
+
// Reasoning activation and allowance are independent of grammar transport;
|
|
557
|
+
// only the response representation becomes lossless when evidence is needed.
|
|
558
|
+
// The llama-server template mapping is owned by {§llama-reasoning-request}.
|
|
559
|
+
#reasoningBody(preserveGrammarSentence = false): Record<string, unknown> {
|
|
434
560
|
const { mode, budget } = this.#reasoning;
|
|
435
561
|
const on = mode !== "off";
|
|
436
562
|
switch (this.#reasoningStyle) {
|
|
437
563
|
case "template": {
|
|
438
564
|
const allowance = mode === "off"
|
|
439
565
|
? 0
|
|
440
|
-
: mode === "on" ? budget : this.reasoningReserve;
|
|
566
|
+
: mode === "on" && budget !== null ? budget : this.reasoningReserve;
|
|
441
567
|
return {
|
|
442
568
|
chat_template_kwargs: { enable_thinking: on },
|
|
443
|
-
reasoning_format: "auto",
|
|
569
|
+
reasoning_format: preserveGrammarSentence ? "none" : "auto",
|
|
444
570
|
...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
|
|
445
571
|
};
|
|
446
572
|
}
|
|
447
573
|
case "think": return on ? { think: true } : {};
|
|
448
574
|
case "include_reasoning": return on ? { include_reasoning: true } : {};
|
|
449
|
-
//
|
|
450
|
-
//
|
|
451
|
-
case "effort": return mode === "on" ? { reasoning_effort:
|
|
575
|
+
// Explicit on uses the portable enabled posture or a tier derived
|
|
576
|
+
// from an explicit budget; off/adaptive omit the field.
|
|
577
|
+
case "effort": return mode === "on" ? { reasoning_effort: effortFromReasoning(this.#reasoning) } : {};
|
|
452
578
|
// Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
|
|
453
579
|
// reason-by-default model (DeepSeek V4: default 'high') reasoning.
|
|
454
580
|
// ADAPTIVE omits the field: the backend's own default posture IS the
|
|
@@ -458,19 +584,25 @@ export default class AiSdkProvider implements Provider {
|
|
|
458
584
|
// efforts 400.
|
|
459
585
|
case "effort_explicit": return mode === "off"
|
|
460
586
|
? { reasoning_effort: "none" }
|
|
461
|
-
: mode === "on" ? { reasoning_effort:
|
|
587
|
+
: mode === "on" ? { reasoning_effort: effortFromReasoning(this.#reasoning) } : {};
|
|
462
588
|
// {§deepseek-reasoning-request}
|
|
463
589
|
case "thinking_effort": return mode === "off"
|
|
464
590
|
? { thinking: { type: "disabled" } }
|
|
465
591
|
: mode === "on" ? {
|
|
466
592
|
thinking: { type: "enabled" },
|
|
467
|
-
reasoning_effort: effortFromBudget(budget
|
|
593
|
+
...(budget === null ? {} : { reasoning_effort: effortFromBudget(budget) }),
|
|
468
594
|
} : {};
|
|
469
595
|
// Anthropic compat: explicit thinking object. off → disabled; on →
|
|
470
|
-
// enabled with
|
|
596
|
+
// enabled with the explicit budget or resolved reserve; adaptive →
|
|
597
|
+
// omit (the API default).
|
|
471
598
|
case "anthropic": return mode === "off"
|
|
472
599
|
? { thinking: { type: "disabled" } }
|
|
473
|
-
: mode === "on" ? {
|
|
600
|
+
: mode === "on" ? {
|
|
601
|
+
thinking: {
|
|
602
|
+
type: "enabled",
|
|
603
|
+
budget_tokens: budget ?? this.reasoningReserve!,
|
|
604
|
+
},
|
|
605
|
+
} : {};
|
|
474
606
|
case "none": return {};
|
|
475
607
|
}
|
|
476
608
|
}
|
|
@@ -537,7 +669,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
537
669
|
}
|
|
538
670
|
}
|
|
539
671
|
|
|
540
|
-
// First-party telemetry headers ({§provider-request-authority}): forwarded only when the spec
|
|
672
|
+
// First-party telemetry headers ({§provider-request-authority} {§provider-call-kind}): forwarded only when the spec
|
|
541
673
|
// opted in (the plurnk endpoint). The gate is here, not at the call site, so
|
|
542
674
|
// attributions/client/strikes can never reach a third-party backend even if
|
|
543
675
|
// the consumer passes them to the wrong provider. Empty values emit no header
|
|
@@ -545,7 +677,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
545
677
|
// absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
|
|
546
678
|
// ride HTTP headers only — the packet never carries them (the model must
|
|
547
679
|
// never see strike state; engine accounting is not a metric to game).
|
|
548
|
-
#metadataHeaders(attributions: string[] | undefined, client: string | undefined, strikes: number | undefined, workerId: string, primaryWorkerId: string | undefined, workspaceId: string | undefined, loop: number | undefined, turn: number | undefined): Record<string, string> {
|
|
680
|
+
#metadataHeaders(attributions: string[] | undefined, client: string | undefined, strikes: number | undefined, workerId: string, primaryWorkerId: string | undefined, workspaceId: string | undefined, loop: number | undefined, turn: number | undefined, callKind: ProviderCallKind | undefined): Record<string, string> {
|
|
549
681
|
if (!this.#firstPartyMetadata) return {};
|
|
550
682
|
const h: Record<string, string> = {};
|
|
551
683
|
if (attributions !== undefined && attributions.length > 0) h["Plurnk-Attribution"] = JSON.stringify(attributions);
|
|
@@ -570,6 +702,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
570
702
|
if (workspaceId !== undefined && workspaceId.length > 0) h["Plurnk-Workspace-Id"] = workspaceId;
|
|
571
703
|
if (loop !== undefined && Number.isInteger(loop) && loop >= 1) h["Plurnk-Loop"] = String(loop);
|
|
572
704
|
if (turn !== undefined && Number.isInteger(turn) && turn >= 1) h["Plurnk-Turn"] = String(turn);
|
|
705
|
+
if (callKind !== undefined) h["Plurnk-Call-Kind"] = callKind;
|
|
573
706
|
return h;
|
|
574
707
|
}
|
|
575
708
|
|
|
@@ -609,9 +742,45 @@ export default class AiSdkProvider implements Provider {
|
|
|
609
742
|
return out;
|
|
610
743
|
}
|
|
611
744
|
|
|
612
|
-
|
|
745
|
+
#requestProviderOptions(workerId: string): AiSdkProviderOptions | undefined {
|
|
746
|
+
const reasoningOptions = this.#reasoning.mode === "off"
|
|
747
|
+
? undefined
|
|
748
|
+
: this.#reasoningResponseProviderOptions;
|
|
749
|
+
if (this.#cacheAffinity?.target !== "provider-option") return reasoningOptions;
|
|
750
|
+
const { provider, name } = this.#cacheAffinity;
|
|
751
|
+
return {
|
|
752
|
+
...reasoningOptions,
|
|
753
|
+
[provider]: {
|
|
754
|
+
...reasoningOptions?.[provider],
|
|
755
|
+
[name]: workerId,
|
|
756
|
+
},
|
|
757
|
+
};
|
|
758
|
+
}
|
|
759
|
+
|
|
760
|
+
#accounting(
|
|
761
|
+
outcome: ProviderRequestAccounting["outcome"],
|
|
762
|
+
usage: ProviderUsage | undefined,
|
|
763
|
+
evidence: Parameters<ProviderCostNormalizer>[0],
|
|
764
|
+
status?: number,
|
|
765
|
+
): ProviderRequestAccounting {
|
|
766
|
+
const knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
|
|
767
|
+
const direct = this.#normalizeCost?.(evidence);
|
|
768
|
+
return validateProviderRequestAccounting({
|
|
769
|
+
provider: this.#source,
|
|
770
|
+
model: this.#model,
|
|
771
|
+
outcome,
|
|
772
|
+
...(status === undefined ? {} : { status }),
|
|
773
|
+
...(knownUsage === undefined ? {} : { usage: knownUsage }),
|
|
774
|
+
cost: resolveProviderCost(direct, this.#estimateCost(knownUsage)),
|
|
775
|
+
});
|
|
776
|
+
}
|
|
777
|
+
|
|
778
|
+
async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling, observeRequest, callKind }: ProviderGenerateArgs): Promise<ProviderResponse> {
|
|
613
779
|
// {§provider-interface} The worker identity is required.
|
|
614
780
|
if (workerId === undefined || workerId.length === 0) throw new Error("generate: workerId is required — the worker's stable, opaque identity");
|
|
781
|
+
if (callKind !== undefined && callKind !== "emission" && callKind !== "bare") {
|
|
782
|
+
throw new Error(`generate: unsupported callKind ${JSON.stringify(callKind)}`);
|
|
783
|
+
}
|
|
615
784
|
// Reject before any wire call when already aborted
|
|
616
785
|
// ({§provider-failure-normalization}).
|
|
617
786
|
signal?.throwIfAborted();
|
|
@@ -621,6 +790,8 @@ export default class AiSdkProvider implements Provider {
|
|
|
621
790
|
const wantGrammar = grammar !== undefined && this.#grammarStyle !== "none";
|
|
622
791
|
if (wantGrammar && this.#gbnfDebug) this.#assertGrammarValid(grammar!);
|
|
623
792
|
const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
|
|
793
|
+
const preserveGrammarSentence = wantGrammar
|
|
794
|
+
&& this.#reasoningStyle === "template";
|
|
624
795
|
|
|
625
796
|
// Assembly order = precedence: the family's sampling DEFAULTS
|
|
626
797
|
// (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
|
|
@@ -634,77 +805,188 @@ export default class AiSdkProvider implements Provider {
|
|
|
634
805
|
...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
|
|
635
806
|
model: this.#model,
|
|
636
807
|
messages,
|
|
637
|
-
...this.#reasoningBody(),
|
|
808
|
+
...this.#reasoningBody(preserveGrammarSentence),
|
|
638
809
|
...this.#grammarBody(sendGrammar),
|
|
639
810
|
...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
|
|
640
811
|
// Request per-token logprobs only when enabled (managed field —
|
|
641
812
|
// reserved from caller sampling; the env flag is the single control).
|
|
642
813
|
...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
|
|
643
814
|
...this.#slotBody(workerId),
|
|
644
|
-
|
|
645
|
-
|
|
646
|
-
|
|
647
|
-
...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
|
|
815
|
+
...(this.#cacheAffinity?.target === "body"
|
|
816
|
+
? { [this.#cacheAffinity.name]: workerId }
|
|
817
|
+
: {}),
|
|
648
818
|
};
|
|
649
819
|
|
|
650
820
|
// Per-request headers = static auth/routing + any first-party telemetry.
|
|
651
|
-
const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
|
|
652
|
-
const headers =
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
|
|
821
|
+
const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind);
|
|
822
|
+
const headers = new Headers(this.#headers);
|
|
823
|
+
if (this.#cacheAffinity?.target === "header") {
|
|
824
|
+
headers.set(this.#cacheAffinity.name, workerId);
|
|
825
|
+
}
|
|
826
|
+
for (const [name, value] of Object.entries(metaHeaders)) headers.set(name, value);
|
|
827
|
+
const requestHeaders = Object.fromEntries(headers.entries());
|
|
828
|
+
const accounting: ProviderRequestAccounting[] = [];
|
|
829
|
+
const operationTimeout = this.#operationTimeoutMs > 0
|
|
830
|
+
? AbortSignal.timeout(this.#operationTimeoutMs)
|
|
831
|
+
: undefined;
|
|
832
|
+
const operationSignal = signal === undefined
|
|
833
|
+
? operationTimeout
|
|
834
|
+
: operationTimeout === undefined
|
|
835
|
+
? signal
|
|
836
|
+
: AbortSignal.any([signal, operationTimeout]);
|
|
837
|
+
const executeRequest = async () => {
|
|
838
|
+
let settle: ProviderRequestSettlement | undefined;
|
|
839
|
+
try {
|
|
840
|
+
settle = await observeRequest?.({
|
|
841
|
+
provider: this.#source,
|
|
658
842
|
model: this.#model,
|
|
659
|
-
headers,
|
|
660
|
-
body,
|
|
661
|
-
messages,
|
|
662
|
-
signal,
|
|
663
|
-
fetch: this.#fetch,
|
|
664
|
-
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
665
|
-
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
666
|
-
retryAttempts: this.#retryAttempts,
|
|
667
|
-
streaming: this.#streaming,
|
|
668
|
-
captureRawBody: this.#rawBody,
|
|
669
|
-
})
|
|
670
|
-
: await executeAiSdkModel({
|
|
671
|
-
languageModel: this.#languageModel,
|
|
672
|
-
headers,
|
|
673
|
-
messages,
|
|
674
|
-
signal,
|
|
675
|
-
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
676
|
-
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
677
|
-
retryAttempts: this.#retryAttempts,
|
|
678
|
-
streaming: this.#streaming,
|
|
679
|
-
captureRawBody: this.#rawBody,
|
|
680
|
-
temperature: this.#tuningFloors
|
|
681
|
-
? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
|
|
682
|
-
: typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
|
|
683
|
-
topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
|
|
684
|
-
topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
|
|
685
|
-
presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
|
|
686
|
-
frequencyPenalty: typeof sampling?.frequency_penalty === "number"
|
|
687
|
-
? sampling.frequency_penalty
|
|
688
|
-
: this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
|
|
689
|
-
stopSequences: typeof sampling?.stop === "string"
|
|
690
|
-
? [sampling.stop]
|
|
691
|
-
: Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
|
|
692
|
-
? sampling.stop
|
|
693
|
-
: undefined,
|
|
694
|
-
seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
|
|
695
|
-
maxOutputTokens: maxTokens,
|
|
696
|
-
reasoning: this.#reasoning.mode === "off"
|
|
697
|
-
? "none"
|
|
698
|
-
: this.#reasoning.mode === "adaptive"
|
|
699
|
-
? "provider-default"
|
|
700
|
-
: effortFromBudget(this.#reasoning.budget!),
|
|
701
843
|
});
|
|
844
|
+
} catch (cause) {
|
|
845
|
+
throw new ProviderRequestObserverError(cause);
|
|
846
|
+
}
|
|
847
|
+
const settleAccounting = async (
|
|
848
|
+
outcome: ProviderRequestAccounting["outcome"],
|
|
849
|
+
usage: ProviderUsage | undefined,
|
|
850
|
+
evidence: Parameters<ProviderCostNormalizer>[0],
|
|
851
|
+
status?: number,
|
|
852
|
+
): Promise<ProviderRequestAccounting> => {
|
|
853
|
+
let requestAccounting: ProviderRequestAccounting;
|
|
854
|
+
let normalizationFailure: { cause: unknown } | undefined;
|
|
855
|
+
try {
|
|
856
|
+
requestAccounting = this.#accounting(outcome, usage, evidence, status);
|
|
857
|
+
} catch (cause) {
|
|
858
|
+
normalizationFailure = { cause };
|
|
859
|
+
let knownUsage: ProviderUsage | undefined;
|
|
860
|
+
try {
|
|
861
|
+
knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
|
|
862
|
+
} catch {
|
|
863
|
+
knownUsage = undefined;
|
|
864
|
+
}
|
|
865
|
+
requestAccounting = validateProviderRequestAccounting({
|
|
866
|
+
provider: this.#source,
|
|
867
|
+
model: this.#model,
|
|
868
|
+
outcome,
|
|
869
|
+
...(status === undefined ? {} : { status }),
|
|
870
|
+
...(knownUsage === undefined ? {} : { usage: knownUsage }),
|
|
871
|
+
cost: {
|
|
872
|
+
kind: "unknown",
|
|
873
|
+
reason: "provider request accounting could not be normalized after physical I/O",
|
|
874
|
+
},
|
|
875
|
+
});
|
|
876
|
+
}
|
|
877
|
+
accounting.push(requestAccounting);
|
|
878
|
+
try {
|
|
879
|
+
await settle?.(requestAccounting);
|
|
880
|
+
} catch (cause) {
|
|
881
|
+
throw new ProviderRequestObserverError(cause);
|
|
882
|
+
}
|
|
883
|
+
if (normalizationFailure !== undefined) {
|
|
884
|
+
throw new ProviderRequestAccountingError(normalizationFailure.cause);
|
|
885
|
+
}
|
|
886
|
+
return requestAccounting;
|
|
887
|
+
};
|
|
888
|
+
let response;
|
|
889
|
+
try {
|
|
890
|
+
response = this.#languageModel === undefined
|
|
891
|
+
? await executeOpenAICompatible({
|
|
892
|
+
url: this.#url!,
|
|
893
|
+
model: this.#model,
|
|
894
|
+
headers: requestHeaders,
|
|
895
|
+
body,
|
|
896
|
+
messages,
|
|
897
|
+
signal: operationSignal,
|
|
898
|
+
fetch: this.#fetch,
|
|
899
|
+
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
900
|
+
firstContentTimeoutMs: this.#firstContentTimeoutMs,
|
|
901
|
+
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
902
|
+
streaming: this.#streaming,
|
|
903
|
+
captureRawBody: this.#rawBody,
|
|
904
|
+
})
|
|
905
|
+
: await executeAiSdkModel({
|
|
906
|
+
languageModel: this.#languageModel,
|
|
907
|
+
headers: requestHeaders,
|
|
908
|
+
providerOptions: this.#requestProviderOptions(workerId),
|
|
909
|
+
systemProviderOptions: this.#systemCacheProviderOptions,
|
|
910
|
+
messages,
|
|
911
|
+
signal: operationSignal,
|
|
912
|
+
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
913
|
+
firstContentTimeoutMs: this.#firstContentTimeoutMs,
|
|
914
|
+
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
915
|
+
streaming: this.#streaming,
|
|
916
|
+
captureRawBody: this.#rawBody,
|
|
917
|
+
temperature: this.#tuningFloors
|
|
918
|
+
? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
|
|
919
|
+
: typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
|
|
920
|
+
topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
|
|
921
|
+
topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
|
|
922
|
+
presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
|
|
923
|
+
frequencyPenalty: typeof sampling?.frequency_penalty === "number"
|
|
924
|
+
? sampling.frequency_penalty
|
|
925
|
+
: this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
|
|
926
|
+
stopSequences: typeof sampling?.stop === "string"
|
|
927
|
+
? [sampling.stop]
|
|
928
|
+
: Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
|
|
929
|
+
? sampling.stop
|
|
930
|
+
: undefined,
|
|
931
|
+
seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
|
|
932
|
+
maxOutputTokens: maxTokens,
|
|
933
|
+
reasoning: this.#reasoning.mode === "off"
|
|
934
|
+
? "none"
|
|
935
|
+
: this.#reasoning.mode === "adaptive"
|
|
936
|
+
? "provider-default"
|
|
937
|
+
: effortFromReasoning(this.#reasoning),
|
|
938
|
+
});
|
|
939
|
+
} catch (error) {
|
|
940
|
+
const failure = transportFailureEvidence(error);
|
|
941
|
+
await settleAccounting(
|
|
942
|
+
"error",
|
|
943
|
+
failure.usage,
|
|
944
|
+
failure.chargeEvidence,
|
|
945
|
+
failure.status,
|
|
946
|
+
);
|
|
947
|
+
throw error;
|
|
948
|
+
}
|
|
949
|
+
await settleAccounting(
|
|
950
|
+
"response",
|
|
951
|
+
response.usage,
|
|
952
|
+
response.chargeEvidence,
|
|
953
|
+
);
|
|
954
|
+
return response;
|
|
955
|
+
};
|
|
956
|
+
|
|
957
|
+
let raw;
|
|
958
|
+
try {
|
|
959
|
+
const { retry } = prepareRetries({
|
|
960
|
+
maxRetries: this.#retryAttempts,
|
|
961
|
+
abortSignal: operationSignal,
|
|
962
|
+
});
|
|
963
|
+
raw = await retry(executeRequest);
|
|
702
964
|
} catch (err) {
|
|
965
|
+
if (err instanceof ProviderRequestObserverError
|
|
966
|
+
|| err instanceof ProviderRequestAccountingError) throw err.cause;
|
|
703
967
|
if (signal?.aborted) throw err;
|
|
968
|
+
if (operationTimeout?.aborted) {
|
|
969
|
+
const timeout = new ProviderTimeoutError("operation", this.#operationTimeoutMs, err);
|
|
970
|
+
throw new ProviderError(this.#source, "deadline_exceeded", timeout.message, {
|
|
971
|
+
status: 504,
|
|
972
|
+
cause: timeout,
|
|
973
|
+
retryable: false,
|
|
974
|
+
extensions: {
|
|
975
|
+
timeoutPhase: timeout.phase,
|
|
976
|
+
timeoutMs: timeout.timeoutMs,
|
|
977
|
+
},
|
|
978
|
+
accounting,
|
|
979
|
+
});
|
|
980
|
+
}
|
|
704
981
|
const pe = toProviderError(err, this.#source, this.#errorDetailLimit);
|
|
705
982
|
if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
|
|
706
|
-
throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, {
|
|
983
|
+
throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, {
|
|
984
|
+
status: pe.status,
|
|
985
|
+
cause: err,
|
|
986
|
+
accounting,
|
|
987
|
+
});
|
|
707
988
|
}
|
|
989
|
+
pe.prependAccounting(accounting);
|
|
708
990
|
throw pe;
|
|
709
991
|
}
|
|
710
992
|
|
|
@@ -716,51 +998,54 @@ export default class AiSdkProvider implements Provider {
|
|
|
716
998
|
// wire text for forensics.
|
|
717
999
|
if (this.#eosText !== undefined) raw.content = stripTrailingSpecial(raw.content, this.#eosText);
|
|
718
1000
|
|
|
719
|
-
const
|
|
720
|
-
|
|
721
|
-
raw.
|
|
722
|
-
|
|
723
|
-
|
|
1001
|
+
const grammarInput = raw.content;
|
|
1002
|
+
const projectedReasoning = preserveGrammarSentence && !raw.reasoningProjected
|
|
1003
|
+
? projectTemplateReasoning(raw.content)
|
|
1004
|
+
: projectTaggedReasoning(
|
|
1005
|
+
raw.content,
|
|
1006
|
+
raw.reasoning,
|
|
1007
|
+
this.#reasoningResponseStyle,
|
|
1008
|
+
);
|
|
724
1009
|
|
|
725
|
-
// Preserve the exact
|
|
726
|
-
// `reasoning_format: "
|
|
727
|
-
//
|
|
728
|
-
//
|
|
1010
|
+
// Preserve the exact pre-projection response. Constrained template turns
|
|
1011
|
+
// request `reasoning_format: "none"`, so even an empty channel and any
|
|
1012
|
+
// template-provided opener remain observable. An unexpectedly projected
|
|
1013
|
+
// response cannot supply independent evidence.
|
|
729
1014
|
let grammarEvidence: GrammarEvidence | undefined;
|
|
730
1015
|
if (wantGrammar) {
|
|
731
|
-
if (
|
|
732
|
-
|
|
733
|
-
input: raw.content,
|
|
734
|
-
contentStart: taggedReasoning.contentStart,
|
|
735
|
-
transported: sendGrammar !== undefined,
|
|
736
|
-
};
|
|
737
|
-
} else if (this.#reasoningStyle === "template" && this.#reasoning.mode !== "off") {
|
|
738
|
-
if (raw.reasoningProjected) {
|
|
739
|
-
const prefix = `<|channel>thought\n${raw.reasoning}<channel|>`;
|
|
1016
|
+
if (preserveGrammarSentence) {
|
|
1017
|
+
if (!raw.reasoningProjected) {
|
|
740
1018
|
grammarEvidence = {
|
|
741
|
-
input:
|
|
742
|
-
contentStart:
|
|
1019
|
+
input: grammarInput,
|
|
1020
|
+
contentStart: projectedReasoning.projected ? projectedReasoning.contentStart : 0,
|
|
743
1021
|
transported: sendGrammar !== undefined,
|
|
744
1022
|
};
|
|
745
1023
|
}
|
|
1024
|
+
} else if (projectedReasoning.projected) {
|
|
1025
|
+
grammarEvidence = {
|
|
1026
|
+
input: grammarInput,
|
|
1027
|
+
contentStart: projectedReasoning.contentStart,
|
|
1028
|
+
transported: sendGrammar !== undefined,
|
|
1029
|
+
};
|
|
746
1030
|
} else {
|
|
747
1031
|
grammarEvidence = {
|
|
748
|
-
input:
|
|
1032
|
+
input: grammarInput,
|
|
749
1033
|
contentStart: 0,
|
|
750
1034
|
transported: sendGrammar !== undefined,
|
|
751
1035
|
};
|
|
752
1036
|
}
|
|
753
1037
|
}
|
|
754
1038
|
|
|
755
|
-
if (
|
|
756
|
-
raw.content =
|
|
757
|
-
raw.reasoning =
|
|
758
|
-
raw.usage = attributeUnitemizedReasoning(raw.usage, raw.reasoning, raw.content);
|
|
1039
|
+
if (projectedReasoning.projected) {
|
|
1040
|
+
raw.content = projectedReasoning.content;
|
|
1041
|
+
raw.reasoning = projectedReasoning.reasoning;
|
|
759
1042
|
}
|
|
760
1043
|
|
|
761
1044
|
let notices: ProviderNotice[] | undefined;
|
|
762
1045
|
const usage = raw.usage;
|
|
763
|
-
if (sendGrammar !== undefined
|
|
1046
|
+
if (sendGrammar !== undefined
|
|
1047
|
+
&& this.tokenize !== undefined
|
|
1048
|
+
&& usage?.outputTokens !== undefined) {
|
|
764
1049
|
// Channel-escape detector: completion tokens
|
|
765
1050
|
// billed far beyond every visible channel mean the decode ESCAPED into
|
|
766
1051
|
// a server-discarded reasoning block mid-emission. This diagnostic
|
|
@@ -771,12 +1056,12 @@ export default class AiSdkProvider implements Provider {
|
|
|
771
1056
|
this.tokenize(raw.reasoning),
|
|
772
1057
|
]);
|
|
773
1058
|
const visible = contentTokens.length + reasoningTokens.length;
|
|
774
|
-
if (usage.
|
|
1059
|
+
if (usage.outputTokens > visible + 64) {
|
|
775
1060
|
(notices ??= []).push({
|
|
776
1061
|
source: this.#source,
|
|
777
1062
|
kind: "grammar_unenforced",
|
|
778
1063
|
level: "warn",
|
|
779
|
-
message: `decode escaped the grammar: ${usage.
|
|
1064
|
+
message: `decode escaped the grammar: ${usage.outputTokens} output tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
|
|
780
1065
|
position: [...raw.content].length,
|
|
781
1066
|
});
|
|
782
1067
|
}
|
|
@@ -800,12 +1085,12 @@ export default class AiSdkProvider implements Provider {
|
|
|
800
1085
|
...(raw.reasoningEncrypted.length > 0
|
|
801
1086
|
? { reasoningEncrypted: raw.reasoningEncrypted }
|
|
802
1087
|
: {}),
|
|
803
|
-
usage,
|
|
804
1088
|
model: raw.model,
|
|
805
1089
|
...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
|
|
806
1090
|
};
|
|
807
1091
|
const evidence = {
|
|
808
1092
|
assistantRaw: raw,
|
|
1093
|
+
accounting,
|
|
809
1094
|
...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
|
|
810
1095
|
...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
|
|
811
1096
|
...(meta !== undefined ? { meta } : {}),
|
|
@@ -822,6 +1107,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
822
1107
|
"The provider interrupted generation because inference resources were unavailable.",
|
|
823
1108
|
{
|
|
824
1109
|
attempt,
|
|
1110
|
+
accounting,
|
|
825
1111
|
extensions: {
|
|
826
1112
|
stage: "provider-response",
|
|
827
1113
|
finishReason: "resource_interrupted",
|
|
@@ -837,4 +1123,5 @@ export default class AiSdkProvider implements Provider {
|
|
|
837
1123
|
...evidence,
|
|
838
1124
|
};
|
|
839
1125
|
}
|
|
1126
|
+
|
|
840
1127
|
}
|