@plurnk/plurnk-providers 1.14.2 → 1.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/SPEC.md +14 -0
- package/dist/AiSdkProvider.d.ts +3 -0
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +19 -338
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/AiSdkRequestBody.d.ts +40 -0
- package/dist/AiSdkRequestBody.d.ts.map +1 -0
- package/dist/AiSdkRequestBody.js +361 -0
- package/dist/AiSdkRequestBody.js.map +1 -0
- package/dist/Mock.d.ts +5 -1
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +9 -2
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +2 -1
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +5 -0
- package/dist/Pool.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +1 -1
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +21 -2
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/capacity.d.ts +0 -1
- package/dist/capacity.d.ts.map +1 -1
- package/dist/capacity.js +1 -1
- package/dist/capacity.js.map +1 -1
- package/dist/catalogProvider.d.ts +2 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +6 -0
- package/dist/catalogProvider.js.map +1 -1
- package/dist/index.d.ts +3 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -0
- package/dist/index.js.map +1 -1
- package/dist/promptTokens.d.ts.map +1 -1
- package/dist/promptTokens.js +2 -1
- package/dist/promptTokens.js.map +1 -1
- package/dist/reasoning-effort.d.ts +4 -0
- package/dist/reasoning-effort.d.ts.map +1 -0
- package/dist/reasoning-effort.js +14 -0
- package/dist/reasoning-effort.js.map +1 -0
- package/dist/types.d.ts +17 -1
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +5 -0
- package/dist/types.js.map +1 -1
- package/package.json +6 -6
- package/src/AiSdkProvider.test.ts +4 -3
- package/src/AiSdkProvider.ts +22 -374
- package/src/AiSdkRequestBody.ts +419 -0
- package/src/Mock.ts +10 -2
- package/src/Pool.test.ts +1 -0
- package/src/Pool.ts +6 -1
- package/src/aiSdkTransport.ts +23 -4
- package/src/boundaries.test.ts +2 -0
- package/src/capacity.ts +1 -1
- package/src/catalogProvider.ts +9 -1
- package/src/index.ts +3 -0
- package/src/inputModalities.test.ts +53 -0
- package/src/promptTokens.ts +2 -1
- package/src/reasoning-effort.ts +15 -0
- package/src/types.ts +18 -1
package/SPEC.md
CHANGED
|
@@ -412,6 +412,20 @@ names without values. Construction rejects the same missing requirements at
|
|
|
412
412
|
the provider boundary instead of deferring a known configuration failure to a
|
|
413
413
|
model request.
|
|
414
414
|
|
|
415
|
+
### §provider-input-modalities Native input parts
|
|
416
|
+
|
|
417
|
+
A provider declares `inputModalities`, the set of native non-text inputs its model
|
|
418
|
+
accepts, from the catalog's input modalities (Models.dev `modalities.input` minus
|
|
419
|
+
`text`, kept to the vocabulary `image`, `pdf`, `audio`, `video`; empty when the
|
|
420
|
+
model is unknown). A user `ChatMessage` may then carry content parts: text beside
|
|
421
|
+
`{ type: "image", image: bytes, mediaType }` and `{ type: "file", data: bytes, mediaType }`,
|
|
422
|
+
which the AI SDK transport forwards as the model's native image and file input;
|
|
423
|
+
system and assistant messages stay text, and prompt-token estimates count text
|
|
424
|
+
only, the provider's reported usage owning each part's cost. A pool declares a
|
|
425
|
+
modality only when every backend does; the Mock declares them by option and
|
|
426
|
+
records every request it receives. Which parts actually ride a request is the
|
|
427
|
+
service's decision per attachment ({§packet-attachment-parts} in the core specification).
|
|
428
|
+
|
|
415
429
|
### §model-fact-resolution Model fact precedence
|
|
416
430
|
|
|
417
431
|
Provider and model facts resolve independently:
|
package/dist/AiSdkProvider.d.ts
CHANGED
|
@@ -2,6 +2,7 @@ import type { ChatMessage, PromptTokenMeasurement, Provider, ProviderCostNormali
|
|
|
2
2
|
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
3
3
|
import type { JSONValue } from "ai";
|
|
4
4
|
import { type Reasoning, type ReasoningResponseStyle } from "./env.ts";
|
|
5
|
+
import type { InputModality } from "./types.ts";
|
|
5
6
|
import type { LanguageModel } from "ai";
|
|
6
7
|
import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
|
|
7
8
|
export type ProviderFetch = typeof globalThis.fetch;
|
|
@@ -30,6 +31,7 @@ export type AiSdkProviderConfig = {
|
|
|
30
31
|
headers?: Record<string, string>;
|
|
31
32
|
fetch?: ProviderFetch;
|
|
32
33
|
contextWindow?: number | null;
|
|
34
|
+
inputModalities?: ReadonlySet<InputModality>;
|
|
33
35
|
maxInputTokens?: number | null;
|
|
34
36
|
maxOutputTokens?: number | null;
|
|
35
37
|
outputBudget?: number | null;
|
|
@@ -83,6 +85,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
83
85
|
tokenize?: (text: string) => Promise<number[]>;
|
|
84
86
|
constructor(config: AiSdkProviderConfig);
|
|
85
87
|
get contextWindow(): number | null;
|
|
88
|
+
get inputModalities(): ReadonlySet<InputModality>;
|
|
86
89
|
get maxInputTokens(): number | null;
|
|
87
90
|
get maxOutputTokens(): number | null;
|
|
88
91
|
get outputBudget(): number | null;
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"AiSdkProvider.d.ts","sourceRoot":"","sources":["../src/AiSdkProvider.ts"],"names":[],"mappings":"AAQA,OAAO,KAAK,
|
|
1
|
+
{"version":3,"file":"AiSdkProvider.d.ts","sourceRoot":"","sources":["../src/AiSdkProvider.ts"],"names":[],"mappings":"AAQA,OAAO,KAAK,EAAE,WAAW,EAAmB,sBAAsB,EAAE,QAAQ,EAAmB,sBAAsB,EAAE,oBAAoB,EAA6B,uBAAuB,EAA6B,gBAAgB,EAAE,aAAa,EAAE,eAAe,EAAE,MAAM,YAAY,CAAC;AACjS,OAAO,KAAK,EAAE,YAAY,EAAE,MAAM,0BAA0B,CAAC;AAE7D,OAAO,KAAK,EAAe,SAAS,EAAE,MAAM,IAAI,CAAC;AACjD,OAAO,EAA2B,KAAK,SAAS,EAAE,KAAK,sBAAsB,EAAE,MAAM,UAAU,CAAC;AAEhG,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,YAAY,CAAC;AAEhD,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,IAAI,CAAC;AAMxC,OAAO,KAAK,EAAE,iBAAiB,EAAE,wBAAwB,EAAE,MAAM,qBAAqB,CAAC;AAQvF,MAAM,MAAM,aAAa,GAAG,OAAO,UAAU,CAAC,KAAK,CAAC;AAIpD,MAAM,MAAM,cAAc,GAAG,MAAM,GAAG,OAAO,GAAG,mBAAmB,GAAG,QAAQ,GAAG,iBAAiB,GAAG,iBAAiB,GAAG,iBAAiB,GAAG,UAAU,GAAG,WAAW,CAAC;AAEtK,MAAM,MAAM,qBAAqB,GAAG,SAAS,GAAG,KAAK,GAAG,QAAQ,GAAG,MAAM,GAAG,OAAO,CAAC;AACpF,MAAM,MAAM,yBAAyB,GAAG,qBAAqB,GAAG,KAAK,CAAC;AAItE,MAAM,MAAM,YAAY,GAAG,MAAM,GAAG,UAAU,CAAC;AAE/C,MAAM,MAAM,aAAa,GACnB;IAAE,QAAQ,CAAC,MAAM,EAAE,QAAQ,GAAG,MAAM,CAAC;IAAC,QAAQ,CAAC,IAAI,EAAE,MAAM,CAAA;CAAE,GAC7D;IAAE,QAAQ,CAAC,MAAM,EAAE,iBAAiB,CAAC;IAAC,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAAC,QAAQ,CAAC,IAAI,EAAE,MAAM,CAAA;CAAE,CAAC;AAE/F,MAAM,MAAM,oBAAoB,GAAG,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,MAAM,EAAE,SAAS,GAAG,SAAS,CAAC,CAAC,CAAC;AAIzF,MAAM,MAAM,mBAAmB,GAAG;IAC9B,KAAK,EAAE,MAAM,CAAC;IACd,GAAG,CAAC,EAAE,MAAM,CAAC;IACb,aAAa,CAAC,EAAE,aAAa,CAAC;IAC9B,YAAY,CAAC,EAAE,CAAC,OAAO,EAAE,wBAAwB,KAAK,iBAAiB,CAAC;IACxE,cAAc,EAAE,MAAM,CAAC;IACvB,kBAAkB,EAAE,MAAM,CAAC;IAC3B,qBAAqB,EAAE,MAAM,CAAC;IAC9B,mBAAmB,CAAC,EAAE,MAAM,CAAC;IAC7B,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACjC,KAAK,CAAC,EAAE,aAAa,CAAC;IACtB,aAAa,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC9B,eAAe,CAAC,EAAE,WAAW,CAAC,aAAa,CAAC,CAAC;IAC7C,cAAc,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC/B,eAAe,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAChC,YAAY,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC7B,eAAe,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAChC,0BAA0B,CAAC,EAAE,SAAS,eAAe,EAAE,CAAC;IAIxD,iBAAiB,CAAC,EAAE,qBAAqB,GAAG,kBAAkB,CAAC;IAE/D,eAAe,CAAC,EAAE,OAAO,CAAC;IAI1B,2BAA2B,CAAC,EAAE,yBAAyB,GAAG,kBAAkB,CAAC;IAC7E,sBAAsB,CAAC,EAAE,MAAM,CAAC;IAChC,gCAAgC,CAAC,EAAE,oBAAoB,CAAC;IAKxD,yBAAyB,CAAC,EAAE,WAAW,GAAG,SAAS,CAAC;IACpD,cAAc,CAAC,EAAE,cAAc,CAAC;IAChC,sBAAsB,CAAC,EAAE,sBAAsB,CAAC;IAChD,iBAAiB,CAAC,EAAE,CAAC,QAAQ,EAAE,SAAS,WAAW,EAAE,EAAE,MAAM,CAAC,EAAE,WAAW,KAAK,sBAAsB,GAAG,OAAO,CAAC,sBAAsB,CAAC,CAAC;IACzI,YAAY,CAAC,EAAE,CAAC,KAAK,EAAE,aAAa,GAAG,SAAS,KAAK,YAAY,CAAC;IAClE,aAAa,CAAC,EAAE,sBAAsB,CAAC;IACvC,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,YAAY,CAAC,EAAE,YAAY,CAAC;IAG5B,aAAa,CAAC,EAAE,aAAa,CAAC;IAG9B,0BAA0B,CAAC,EAAE,oBAAoB,CAAC;IAGlD,gCAAgC,CAAC,EAAE,oBAAoB,CAAC;IAGxD,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,SAAS,CAAC,EAAE,OAAO,CAAC;IACpB,SAAS,CAAC,EAAE,OAAO,CAAC;IACpB,kBAAkB,CAAC,EAAE,OAAO,CAAC;IAC7B,qBAAqB,CAAC,EAAE,MAAM,CAAC;IAC/B,OAAO,CAAC,EAAE,MAAM,CAAC;IAEjB,mBAAmB,CAAC,EAAE,OAAO,CAAC;IAC9B,SAAS,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAI1B,WAAW,CAAC,EAAE,MAAM,CAAC;IAGrB,eAAe,CAAC,EAAE,MAAM,CAAC;IAKzB,WAAW,CAAC,EAAE,MAAM,CAAC;IAIrB,oBAAoB,CAAC,EAAE,OAAO,CAAC;IAM/B,SAAS,EAAE,SAAS,CAAC;IAOrB,WAAW,EAAE,MAAM,GAAG,IAAI,CAAC;IAC3B,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;IAK7B,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAK1B,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,WAAW,CAAC,EAAE,MAAM,CAAC;IAKrB,aAAa,EAAE,MAAM,CAAC;IAItB,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAQ1B,WAAW,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC5B,OAAO,CAAC,EAAE,OAAO,CAAC;IAUlB,YAAY,CAAC,EAAE,OAAO,CAAC;CAC1B,CAAC;AA0GF,MAAM,CAAC,OAAO,OAAO,aAAc,YAAW,QAAQ;;IA0DlD,QAAQ,CAAC,YAAY,CAAC,EAAE,CAAC,OAAO,EAAE,wBAAwB,KAAK,iBAAiB,CAAC;IAMjF,QAAQ,CAAC,EAAE,CAAC,IAAI,EAAE,MAAM,KAAK,OAAO,CAAC,MAAM,EAAE,CAAC,CAAC;IAE/C,YAAY,MAAM,EAAE,mBAAmB,EAqKtC;IAED,IAAI,aAAa,IAAI,MAAM,GAAG,IAAI,CAAgC;IAClE,IAAI,eAAe,IAAI,WAAW,CAAC,aAAa,CAAC,CAAkC;IACnF,IAAI,cAAc,IAAI,MAAM,GAAG,IAAI,CAAiC;IACpE,IAAI,eAAe,IAAI,MAAM,GAAG,IAAI,CAAkC;IACtE,IAAI,YAAY,IAAI,MAAM,GAAG,IAAI,CAA+B;IAChE,IAAI,eAAe,IAAI,MAAM,GAAG,IAAI,CAAkC;IACtE,IAAI,0BAA0B,IAAI,SAAS,eAAe,EAAE,CAA6C;IACzG,IAAI,aAAa,IAAI,MAAM,GAAG,IAAI,CAMjC;IACD,IAAI,KAAK,IAAI,MAAM,CAAwB;IAE3C,IAAI,WAAW,IAAI,MAAM,GAAG,SAAS,CAA8B;IAEnE,IAAI,oBAAoB,IAAI,OAAO,GAAG,SAAS,CAAuC;IAItF,IAAI,gBAAgB,IAAI,OAAO,CAA0C;IAEnE,iBAAiB,CACnB,QAAQ,EAAE,SAAS,WAAW,EAAE,EAChC,MAAM,CAAC,EAAE,WAAW,GACrB,OAAO,CAAC,sBAAsB,CAAC,CAqDjC;IAEK,qBAAqB,CACvB,QAAQ,EAAE,SAAS,WAAW,EAAE,EAChC,eAAe,CAAC,EAAE,MAAM,EACxB,MAAM,CAAC,EAAE,WAAW,GACrB,OAAO,CAAC,uBAAuB,CAAC,CAmBlC;IA4BK,QAAQ,CAAC,EAAE,QAAQ,EAAE,QAAQ,EAAE,eAAe,EAAE,MAAM,EAAE,OAAO,EAAE,eAAe,EAAE,YAAY,EAAE,MAAM,EAAE,OAAO,EAAE,WAAW,EAAE,IAAI,EAAE,IAAI,EAAE,QAAQ,EAAE,cAAc,EAAE,gBAAgB,EAAE,QAAQ,EAAE,EAAE,oBAAoB,GAAG,OAAO,CAAC,gBAAgB,CAAC,CAkcvP;CAEJ"}
|
package/dist/AiSdkProvider.js
CHANGED
|
@@ -8,27 +8,17 @@
|
|
|
8
8
|
import { REASONING_POLICIES } from "@plurnk/plurnk-contracts";
|
|
9
9
|
import { MAX_PROVIDER_TIMEOUT_MS } from "./env.js";
|
|
10
10
|
import { UnsupportedReasoningPolicyError } from "./types.js";
|
|
11
|
-
import { executeAiSdkModel, executeOpenAICompatible, transportFailureOutputObserved, transportFailureEvidence
|
|
11
|
+
import { executeAiSdkModel, executeOpenAICompatible, transportFailureOutputObserved, transportFailureEvidence } from "./aiSdkTransport.js";
|
|
12
12
|
import { prepareRetries } from "ai/internal";
|
|
13
13
|
import { toProviderError, ProviderError, ProviderTimeoutError } from "./errors.js";
|
|
14
|
-
import { validateGbnf } from "@plurnk/gbnf";
|
|
15
14
|
import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.js";
|
|
16
15
|
import { emitWarningOnce } from "./warnings.js";
|
|
17
16
|
import { resolveProviderCost } from "./cost.js";
|
|
18
17
|
import { validateProviderRequestAccounting } from "./accounting.js";
|
|
19
18
|
import { validateProviderUsage } from "./usage.js";
|
|
20
19
|
import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.js";
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
const leftValue = left[key];
|
|
24
|
-
const rightValue = right[key];
|
|
25
|
-
return [
|
|
26
|
-
key,
|
|
27
|
-
isJsonObject(leftValue) && isJsonObject(rightValue)
|
|
28
|
-
? mergeJsonObjects(leftValue, rightValue)
|
|
29
|
-
: rightValue ?? leftValue,
|
|
30
|
-
];
|
|
31
|
-
}));
|
|
20
|
+
import { nativeFixedEffort } from "./reasoning-effort.js";
|
|
21
|
+
import AiSdkRequestBody from "./AiSdkRequestBody.js";
|
|
32
22
|
class ProviderRequestObserverError extends Error {
|
|
33
23
|
constructor(cause) {
|
|
34
24
|
super("provider request accounting could not be durably settled", { cause });
|
|
@@ -101,32 +91,6 @@ const projectTemplateReasoning = (content) => {
|
|
|
101
91
|
}
|
|
102
92
|
return { content, reasoning: "", projected: false, contentStart: 0 };
|
|
103
93
|
};
|
|
104
|
-
const fixedEffort = (mode) => {
|
|
105
|
-
if (mode === "low" || mode === "medium" || mode === "high" || mode === "xhigh" || mode === "max")
|
|
106
|
-
return mode;
|
|
107
|
-
throw new TypeError(`reasoning policy '${mode}' is not a fixed effort`);
|
|
108
|
-
};
|
|
109
|
-
// The native SDK effort surface tops at xhigh; admission never grants a native
|
|
110
|
-
// route "max", so reaching it here is a contract violation, not a fallback site.
|
|
111
|
-
const nativeFixedEffort = (mode) => {
|
|
112
|
-
const effort = fixedEffort(mode);
|
|
113
|
-
if (effort === "max")
|
|
114
|
-
throw new TypeError(`reasoning policy 'max' has no native SDK effort surface`);
|
|
115
|
-
return effort;
|
|
116
|
-
};
|
|
117
|
-
// Anthropic's older manual-reasoning protocol needs an absolute allowance while
|
|
118
|
-
// PLURNK's durable contract names an effort. These fractions match the native
|
|
119
|
-
// SDK's policy projection, but apply to PLURNK's total envelope rather than the
|
|
120
|
-
// model's physical maximum. The minimum is imposed by the provider protocol.
|
|
121
|
-
const MANUAL_REASONING_FRACTIONS = Object.freeze({
|
|
122
|
-
adaptive: 0.6,
|
|
123
|
-
low: 0.1,
|
|
124
|
-
medium: 0.3,
|
|
125
|
-
high: 0.6,
|
|
126
|
-
xhigh: 0.75,
|
|
127
|
-
max: 0.85,
|
|
128
|
-
});
|
|
129
|
-
const MANUAL_REASONING_MINIMUM = 1024;
|
|
130
94
|
const providerWarningMessage = (warning) => {
|
|
131
95
|
switch (warning.type) {
|
|
132
96
|
case "unsupported":
|
|
@@ -136,27 +100,6 @@ const providerWarningMessage = (warning) => {
|
|
|
136
100
|
case "other": return warning.message;
|
|
137
101
|
}
|
|
138
102
|
};
|
|
139
|
-
// Body keys the provider owns — a caller's `sampling` passthrough may not set
|
|
140
|
-
// these. Two families:
|
|
141
|
-
// transport/managed — grammar transport, the stream/JSON choice, slot pinning,
|
|
142
|
-
// data capture ({§provider-evidence}: backend-specific fields never cross the contract);
|
|
143
|
-
// contract invariants — `n` (atomic single completion: choices[0] is the
|
|
144
|
-
// response; n>1 = paid, dropped output), the tool-calling family (tools-in-
|
|
145
|
-
// body doctrine, §2: native tool_calls return null content = a broken turn),
|
|
146
|
-
// modalities/audio (text-only contract), prediction (decode semantics, not
|
|
147
|
-
// sampling), and the token caps (the envelope is the managed maxOutputTokens —
|
|
148
|
-
// sampling must not bypass the consumer's cap).
|
|
149
|
-
// Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
|
|
150
|
-
// platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
|
|
151
|
-
// metadata, store, verbosity) pass through; the managed floors spread UNDER
|
|
152
|
-
// sampling stay deliberately caller-overridable.
|
|
153
|
-
const RESERVED_BODY_KEYS = new Set([
|
|
154
|
-
"model", "messages", "stream", "stream_options", "grammar", "response_format", "id_slot", "logprobs", "top_logprobs",
|
|
155
|
-
"reasoning_format", "reasoning_effort", "thinking", "think", "include_reasoning", "chat_template_kwargs", "thinking_budget_tokens", // lexicon-allow: backend wire fields
|
|
156
|
-
"n", "tools", "tool_choice", "functions", "function_call", "parallel_tool_calls",
|
|
157
|
-
"modalities", "audio", "prediction", "max_tokens", "max_completion_tokens",
|
|
158
|
-
"prompt_cache_key",
|
|
159
|
-
]);
|
|
160
103
|
export default class AiSdkProvider {
|
|
161
104
|
#model;
|
|
162
105
|
#url;
|
|
@@ -171,6 +114,7 @@ export default class AiSdkProvider {
|
|
|
171
114
|
#apiKeyRejectedMessage;
|
|
172
115
|
#eosText;
|
|
173
116
|
#contextWindow;
|
|
117
|
+
#inputModalities;
|
|
174
118
|
#maxInputTokens;
|
|
175
119
|
#maxOutputTokens;
|
|
176
120
|
#outputBudget;
|
|
@@ -220,6 +164,7 @@ export default class AiSdkProvider {
|
|
|
220
164
|
// tokenizeUrl (llama-server), so `provider.tokenize === undefined` remains
|
|
221
165
|
// the honest capability signal for every other backend.
|
|
222
166
|
tokenize;
|
|
167
|
+
#requestBody;
|
|
223
168
|
constructor(config) {
|
|
224
169
|
this.#model = config.model;
|
|
225
170
|
this.#url = config.url;
|
|
@@ -245,6 +190,7 @@ export default class AiSdkProvider {
|
|
|
245
190
|
this.#headers = config.headers ?? {};
|
|
246
191
|
this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
|
|
247
192
|
this.#contextWindow = config.contextWindow ?? null;
|
|
193
|
+
this.#inputModalities = config.inputModalities ?? new Set();
|
|
248
194
|
this.#maxInputTokens = config.maxInputTokens ?? null;
|
|
249
195
|
this.#maxOutputTokens = config.maxOutputTokens ?? null;
|
|
250
196
|
this.#outputBudget = config.outputBudget ?? null;
|
|
@@ -377,8 +323,10 @@ export default class AiSdkProvider {
|
|
|
377
323
|
return tokens;
|
|
378
324
|
};
|
|
379
325
|
}
|
|
326
|
+
this.#requestBody = new AiSdkRequestBody({ reasoningBudget: this.#reasoningBudget, additiveReasoningProvider: this.#additiveReasoningProvider, reasoning: this.#reasoning, reasoningToggle: this.#reasoningToggle, compatibleAdaptiveReasoning: this.#compatibleAdaptiveReasoning, compatibleOffReasoning: this.#compatibleOffReasoning, adaptiveReasoningProviderOptions: this.#adaptiveReasoningProviderOptions, repeatPenalty: this.#repeatPenalty, frequencyPenalty: this.#frequencyPenalty, dryMultiplier: this.#dryMultiplier, dryBase: this.#dryBase, dryAllowedLength: this.#dryAllowedLength, repeatLastN: this.#repeatLastN, reasoningStyle: this.#reasoningStyle, source: this.#source, grammarStyle: this.#grammarStyle, cacheAffinity: this.#cacheAffinity, reasoningResponseProviderOptions: this.#reasoningResponseProviderOptions, firstPartyMetadata: this.#firstPartyMetadata, supportsSlotPinning: this.#supportsSlotPinning, slotCount: this.#slotCount });
|
|
380
327
|
}
|
|
381
328
|
get contextWindow() { return this.#contextWindow; }
|
|
329
|
+
get inputModalities() { return this.#inputModalities; }
|
|
382
330
|
get maxInputTokens() { return this.#maxInputTokens; }
|
|
383
331
|
get maxOutputTokens() { return this.#maxOutputTokens; }
|
|
384
332
|
get outputBudget() { return this.#outputBudget; }
|
|
@@ -420,7 +368,7 @@ export default class AiSdkProvider {
|
|
|
420
368
|
body: JSON.stringify({
|
|
421
369
|
model: this.#model,
|
|
422
370
|
messages,
|
|
423
|
-
...this.#reasoningBody(),
|
|
371
|
+
...this.#requestBody.reasoningBody(),
|
|
424
372
|
}),
|
|
425
373
|
...(requestSignal === undefined ? {} : { signal: requestSignal }),
|
|
426
374
|
});
|
|
@@ -462,208 +410,6 @@ export default class AiSdkProvider {
|
|
|
462
410
|
measurement: await this.countPromptTokens(messages, signal),
|
|
463
411
|
});
|
|
464
412
|
}
|
|
465
|
-
// Reasoning activation and allowance are independent of grammar transport;
|
|
466
|
-
// only the response representation becomes lossless when evidence is needed.
|
|
467
|
-
// The llama-server template mapping is owned by {§llama-reasoning-request}.
|
|
468
|
-
#reasoningBody(preserveGrammarSentence = false, reasoningBudget = this.#reasoningBudget) {
|
|
469
|
-
const { mode } = this.#reasoning;
|
|
470
|
-
const budget = reasoningBudget;
|
|
471
|
-
const on = mode !== "off";
|
|
472
|
-
switch (this.#reasoningStyle) {
|
|
473
|
-
case "template": {
|
|
474
|
-
const allowance = mode === "off"
|
|
475
|
-
? 0
|
|
476
|
-
: budget;
|
|
477
|
-
// A fixed effort rides into the template as its own variable; adaptive
|
|
478
|
-
// and off send none and leave the template's default in force.
|
|
479
|
-
const templateEffort = mode === "off" || mode === "adaptive" ? {} : { reasoning_effort: fixedEffort(mode) };
|
|
480
|
-
return {
|
|
481
|
-
chat_template_kwargs: { enable_thinking: on, ...templateEffort },
|
|
482
|
-
reasoning_format: preserveGrammarSentence ? "none" : "auto",
|
|
483
|
-
...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
|
|
484
|
-
};
|
|
485
|
-
}
|
|
486
|
-
case "think": return on ? { think: true } : {};
|
|
487
|
-
case "include_reasoning": return on ? { include_reasoning: true } : {};
|
|
488
|
-
case "effort": return mode === "off"
|
|
489
|
-
? this.#compatibleOffReasoning === undefined
|
|
490
|
-
? {}
|
|
491
|
-
: { reasoning_effort: this.#compatibleOffReasoning }
|
|
492
|
-
: mode === "adaptive"
|
|
493
|
-
? this.#compatibleAdaptiveReasoning === "provider-default"
|
|
494
|
-
? {}
|
|
495
|
-
: { reasoning_effort: this.#compatibleAdaptiveReasoning }
|
|
496
|
-
: { reasoning_effort: fixedEffort(mode) };
|
|
497
|
-
// Graded reasoning is mandatory when the route advertises an effort
|
|
498
|
-
// value. Cataloged routes supply the exact strongest legal value;
|
|
499
|
-
// construction rejects an unsupported off or fixed policy.
|
|
500
|
-
case "effort_required": {
|
|
501
|
-
if (mode === "off") {
|
|
502
|
-
if (this.#compatibleOffReasoning === undefined) {
|
|
503
|
-
throw new TypeError(`${this.#source}: required reasoning effort has no off projection`);
|
|
504
|
-
}
|
|
505
|
-
return { reasoning_effort: this.#compatibleOffReasoning };
|
|
506
|
-
}
|
|
507
|
-
if (mode === "adaptive") {
|
|
508
|
-
return this.#compatibleAdaptiveReasoning === "provider-default"
|
|
509
|
-
? {}
|
|
510
|
-
: { reasoning_effort: this.#compatibleAdaptiveReasoning };
|
|
511
|
-
}
|
|
512
|
-
return { reasoning_effort: fixedEffort(mode) };
|
|
513
|
-
}
|
|
514
|
-
// Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
|
|
515
|
-
// reason-by-default model (DeepSeek V4: default 'high') reasoning.
|
|
516
|
-
// ADAPTIVE omits the field UNLESS the catalog declares a toggle control:
|
|
517
|
-
// toggle routes (nemotron-lightning) default reasoning OFF, so adaptive
|
|
518
|
-
// sends the documented Fireworks Boolean enable (#457). The literal
|
|
519
|
-
// "adaptive" is MiniMax-M3-only — Fireworks 400s it for every other
|
|
520
|
-
// model (wire-verified; the 1.0.2 adaptive default refused to boot on
|
|
521
|
-
// it). V4 gotcha: integer efforts 400.
|
|
522
|
-
case "effort_explicit": return mode === "off"
|
|
523
|
-
? { reasoning_effort: "none" }
|
|
524
|
-
: mode === "adaptive"
|
|
525
|
-
? this.#reasoningToggle ? { reasoning_effort: true } : {}
|
|
526
|
-
: { reasoning_effort: fixedEffort(mode) };
|
|
527
|
-
// {§deepseek-reasoning-request}
|
|
528
|
-
case "thinking_effort": return mode === "off"
|
|
529
|
-
? { thinking: { type: "disabled" } }
|
|
530
|
-
: mode === "adaptive" ? { thinking: { type: "enabled" } } : {
|
|
531
|
-
thinking: { type: "enabled" },
|
|
532
|
-
reasoning_effort: fixedEffort(mode),
|
|
533
|
-
};
|
|
534
|
-
// Anthropic-compatible native dynamic or manual budget mode.
|
|
535
|
-
case "anthropic": return mode === "off"
|
|
536
|
-
? { thinking: { type: "disabled" } }
|
|
537
|
-
: mode === "adaptive" ? { thinking: { type: "adaptive" } } : {
|
|
538
|
-
thinking: {
|
|
539
|
-
type: "enabled",
|
|
540
|
-
budget_tokens: budget,
|
|
541
|
-
},
|
|
542
|
-
};
|
|
543
|
-
case "none": return {};
|
|
544
|
-
}
|
|
545
|
-
}
|
|
546
|
-
// Per-worker slot affinity: the consumer passes which worker this is; the
|
|
547
|
-
// provider owns WHICH slot serves it. Sticky per workerId, round-robin across
|
|
548
|
-
// new runs (distinct runs → distinct slots while slots last), LRU-bounded
|
|
549
|
-
// bookkeeping so a long-lived daemon never grows the map unboundedly —
|
|
550
|
-
// an evicted-and-returning run simply re-pins, worst case one cold prefill.
|
|
551
|
-
#runSlots = new Map();
|
|
552
|
-
#nextSlot = 0;
|
|
553
|
-
#slotBody(workerId) {
|
|
554
|
-
if (!this.#supportsSlotPinning || this.#slotCount === null || this.#slotCount < 1)
|
|
555
|
-
return {};
|
|
556
|
-
let slot = this.#runSlots.get(workerId);
|
|
557
|
-
if (slot === undefined) {
|
|
558
|
-
slot = this.#nextSlot++ % this.#slotCount;
|
|
559
|
-
if (this.#runSlots.size >= this.#slotCount * 8) {
|
|
560
|
-
this.#runSlots.delete(this.#runSlots.keys().next().value);
|
|
561
|
-
}
|
|
562
|
-
}
|
|
563
|
-
else {
|
|
564
|
-
this.#runSlots.delete(workerId); // re-insert to refresh LRU recency
|
|
565
|
-
}
|
|
566
|
-
this.#runSlots.set(workerId, slot);
|
|
567
|
-
return { id_slot: slot };
|
|
568
|
-
}
|
|
569
|
-
// Optional local llama-server GBNF transport ({§gbnf-response-observation}). Unsupported
|
|
570
|
-
// backends receive no grammar-related field.
|
|
571
|
-
#grammarBody(grammar) {
|
|
572
|
-
if (grammar === undefined)
|
|
573
|
-
return {};
|
|
574
|
-
switch (this.#grammarStyle) {
|
|
575
|
-
// Grammar-constrained decoding can loop under the mask; a configured
|
|
576
|
-
// per-alias repeat_penalty is the measured remedy ({§provider-sampling-passthrough}).
|
|
577
|
-
case "llamacpp": return { grammar, ...(this.#repeatPenalty !== null ? { repeat_penalty: this.#repeatPenalty } : {}) };
|
|
578
|
-
case "none": return {};
|
|
579
|
-
}
|
|
580
|
-
}
|
|
581
|
-
// Anti-degeneration default on every request, keyed to the backend's wire
|
|
582
|
-
// convention - NOT grammar-bound. GBNF is a local constraint, so a cloud
|
|
583
|
-
// alias runs the sampler bare: firefast (deepseek/fireworks) ran 4/86 bench turns
|
|
584
|
-
// straight to the token cap on pure looped repetition (run52). Ships next to
|
|
585
|
-
// temperature so caller `sampling` can tune it; the grammar path re-asserts it as a
|
|
586
|
-
// managed FLOOR in #grammarBody. llama.cpp takes the repeat_penalty
|
|
587
|
-
// MULTIPLIER; the plain cloud path ("none") can't, so it gets
|
|
588
|
-
// frequency_penalty - OpenAI-standard, accepted by every OpenAI-compat backend (verified
|
|
589
|
-
// live: together/deepinfra/fireworks; it is OpenAI's own param, so real OpenAI takes it too).
|
|
590
|
-
#repetitionPenaltyBody() {
|
|
591
|
-
switch (this.#grammarStyle) {
|
|
592
|
-
// repeat_penalty + optional DRY (repeated-sequence penalty) + a wider
|
|
593
|
-
// repeat_last_n window — the loop-breaking tools a llama.cpp backend serves.
|
|
594
|
-
// Each rides only when its operator knob is set; absent = the box's default.
|
|
595
|
-
case "llamacpp": return {
|
|
596
|
-
...(this.#repeatPenalty !== null ? { repeat_penalty: this.#repeatPenalty } : {}),
|
|
597
|
-
...(this.#repeatLastN !== undefined ? { repeat_last_n: this.#repeatLastN } : {}),
|
|
598
|
-
...(this.#dryMultiplier !== undefined && this.#dryMultiplier > 0 ? {
|
|
599
|
-
dry_multiplier: this.#dryMultiplier,
|
|
600
|
-
...(this.#dryBase !== undefined ? { dry_base: this.#dryBase } : {}),
|
|
601
|
-
...(this.#dryAllowedLength !== undefined ? { dry_allowed_length: this.#dryAllowedLength } : {}),
|
|
602
|
-
} : {}),
|
|
603
|
-
};
|
|
604
|
-
case "none": return this.#frequencyPenalty > 0 ? { frequency_penalty: this.#frequencyPenalty } : {};
|
|
605
|
-
}
|
|
606
|
-
}
|
|
607
|
-
// First-party telemetry headers ({§provider-request-authority} {§provider-call-kind}): forwarded only when the spec
|
|
608
|
-
// opted in (the plurnk endpoint). The gate is here, not at the call site, so
|
|
609
|
-
// attributions/client/strikes can never reach a third-party backend even if
|
|
610
|
-
// the consumer passes them to the wrong provider. Empty values emit no header
|
|
611
|
-
// — EXCEPT strikes, where 0 is a real value (clean streak) distinct from
|
|
612
|
-
// absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
|
|
613
|
-
// ride HTTP headers only — the packet never carries them (the model must
|
|
614
|
-
// never see strike state; engine accounting is not a metric to game).
|
|
615
|
-
#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind) {
|
|
616
|
-
if (!this.#firstPartyMetadata)
|
|
617
|
-
return {};
|
|
618
|
-
const h = {};
|
|
619
|
-
if (attributions !== undefined && attributions.length > 0)
|
|
620
|
-
h["Plurnk-Attribution"] = JSON.stringify(attributions);
|
|
621
|
-
if (client !== undefined && client.length > 0)
|
|
622
|
-
h["Plurnk-Client"] = client;
|
|
623
|
-
if (strikes !== undefined && Number.isInteger(strikes) && strikes >= 0)
|
|
624
|
-
h["Plurnk-Strikes"] = String(strikes);
|
|
625
|
-
// Worker identity: the opaque workerId
|
|
626
|
-
// the consumer already supplies, forwarded so the endpoint can key
|
|
627
|
-
// per-worker affinity/telemetry — same gate as every first-party signal.
|
|
628
|
-
h["Plurnk-Worker-Id"] = workerId;
|
|
629
|
-
// Root worker of the lineage ({§worker-primary}): the no-parent ancestor of this turn's
|
|
630
|
-
// worker tree. The consumer classifies primary-vs-spawned by equality
|
|
631
|
-
// (primaryWorkerId == workerId ⇒ the primary/root worker). The provider
|
|
632
|
-
// EMITS what the consumer supplies and never invents a primary; the
|
|
633
|
-
// consumer's contract is to stamp it EVERY turn (including the primary's
|
|
634
|
-
// own, where it equals workerId). Absence is the consumer's violation for
|
|
635
|
-
// the endpoint to surface, not a provider default.
|
|
636
|
-
if (primaryWorkerId !== undefined && primaryWorkerId.length > 0)
|
|
637
|
-
h["Plurnk-Worker-Primary"] = primaryWorkerId;
|
|
638
|
-
// Turn coordinate ({§lifecycle-terms}): workspace/loop/turn, the
|
|
639
|
-
// daemon-side sequence the endpoint can never scrape from the wire.
|
|
640
|
-
// Coordinates are 1-based — 0 is not a real value, so no strikes-style
|
|
641
|
-
// zero exception; absent/empty/0 emits no header.
|
|
642
|
-
if (workspaceId !== undefined && workspaceId.length > 0)
|
|
643
|
-
h["Plurnk-Workspace-Id"] = workspaceId;
|
|
644
|
-
if (loop !== undefined && Number.isInteger(loop) && loop >= 1)
|
|
645
|
-
h["Plurnk-Loop"] = String(loop);
|
|
646
|
-
if (turn !== undefined && Number.isInteger(turn) && turn >= 1)
|
|
647
|
-
h["Plurnk-Turn"] = String(turn);
|
|
648
|
-
if (callKind !== undefined)
|
|
649
|
-
h["Plurnk-Call-Kind"] = callKind;
|
|
650
|
-
return h;
|
|
651
|
-
}
|
|
652
|
-
// PLURNK_PROVIDERS_GBNF_DEBUG ({§gbnf-response-observation}): validate the supplied GBNF locally and fail
|
|
653
|
-
// hard if it's malformed, BEFORE any wire call — and the grammar is NOT
|
|
654
|
-
// transported, so the request runs unconstrained. A debug aid to catch invalid
|
|
655
|
-
// grammars (e.g. while editing the plurnk grammar) without a model round-trip;
|
|
656
|
-
// off in production. `validateGbnf(grammar, "")` parses the grammar + resolves
|
|
657
|
-
// its root, throwing iff the grammar itself is invalid (the empty input's
|
|
658
|
-
// verdict is irrelevant — we only care that parsing succeeded).
|
|
659
|
-
#assertGrammarValid(grammar) {
|
|
660
|
-
try {
|
|
661
|
-
validateGbnf(grammar, "");
|
|
662
|
-
}
|
|
663
|
-
catch (cause) {
|
|
664
|
-
throw new Error(`grammar validation (PLURNK_PROVIDERS_GBNF_DEBUG): invalid GBNF — ${cause.message}`, { cause });
|
|
665
|
-
}
|
|
666
|
-
}
|
|
667
413
|
// Per-turn metadata bag: pass the backend's non-standard top-level fields
|
|
668
414
|
// through verbatim. Providers do not reinterpret vendor currency or account
|
|
669
415
|
// metadata; a monetary value carries its own amount and currency.
|
|
@@ -671,71 +417,6 @@ export default class AiSdkProvider {
|
|
|
671
417
|
const meta = { ...chunkMetadata };
|
|
672
418
|
return Object.keys(meta).length > 0 ? meta : undefined;
|
|
673
419
|
}
|
|
674
|
-
// Caller-supplied OpenAI-compat sampling params (temperature, top_p, top_k,
|
|
675
|
-
// penalties, stop, seed, …) merged UNDER the managed body: model, messages,
|
|
676
|
-
// reasoning, grammar (+ its repeat-penalty floor), max_tokens and slot always
|
|
677
|
-
// win, and reserved transport/protocol keys are stripped so the passthrough
|
|
678
|
-
// can't smuggle a grammar, a stream toggle, or a backend slot
|
|
679
|
-
// ({§provider-request-authority}).
|
|
680
|
-
#samplingBody(sampling) {
|
|
681
|
-
if (sampling === undefined)
|
|
682
|
-
return {};
|
|
683
|
-
const out = {};
|
|
684
|
-
for (const [k, v] of Object.entries(sampling))
|
|
685
|
-
if (!RESERVED_BODY_KEYS.has(k))
|
|
686
|
-
out[k] = v;
|
|
687
|
-
return out;
|
|
688
|
-
}
|
|
689
|
-
#requestProviderOptions(workerId, nativeReasoningBudget) {
|
|
690
|
-
const responseOptions = this.#reasoning.mode === "off"
|
|
691
|
-
? undefined
|
|
692
|
-
: this.#reasoningResponseProviderOptions;
|
|
693
|
-
const adaptiveOptions = this.#reasoning.mode === "adaptive"
|
|
694
|
-
&& nativeReasoningBudget === null
|
|
695
|
-
? this.#adaptiveReasoningProviderOptions
|
|
696
|
-
: undefined;
|
|
697
|
-
const nativeReasoning = nativeReasoningBudget !== null
|
|
698
|
-
? this.#additiveReasoningProvider === "anthropic"
|
|
699
|
-
? { anthropic: { thinking: { type: "enabled", budgetTokens: nativeReasoningBudget } } }
|
|
700
|
-
: this.#additiveReasoningProvider === "bedrock"
|
|
701
|
-
? { bedrock: { reasoningConfig: { type: "enabled", budgetTokens: nativeReasoningBudget } } }
|
|
702
|
-
: undefined
|
|
703
|
-
: undefined;
|
|
704
|
-
const options = {};
|
|
705
|
-
for (const part of [responseOptions, adaptiveOptions, nativeReasoning]) {
|
|
706
|
-
for (const [provider, values] of Object.entries(part ?? {})) {
|
|
707
|
-
options[provider] = mergeJsonObjects(options[provider] ?? {}, values);
|
|
708
|
-
}
|
|
709
|
-
}
|
|
710
|
-
if (this.#cacheAffinity?.target === "provider-option") {
|
|
711
|
-
const { provider, name } = this.#cacheAffinity;
|
|
712
|
-
options[provider] = { ...options[provider], [name]: workerId };
|
|
713
|
-
}
|
|
714
|
-
return Object.keys(options).length === 0 ? undefined : options;
|
|
715
|
-
}
|
|
716
|
-
#nativeMaxOutputTokens(outputBudget, nativeReasoningBudget) {
|
|
717
|
-
if (outputBudget === null)
|
|
718
|
-
return undefined;
|
|
719
|
-
return nativeReasoningBudget !== null
|
|
720
|
-
? outputBudget - nativeReasoningBudget
|
|
721
|
-
: outputBudget;
|
|
722
|
-
}
|
|
723
|
-
#nativeReasoningBudget(outputBudget, configuredReasoningBudget) {
|
|
724
|
-
if (this.#additiveReasoningProvider === undefined || this.#reasoning.mode === "off")
|
|
725
|
-
return null;
|
|
726
|
-
if (configuredReasoningBudget !== null)
|
|
727
|
-
return configuredReasoningBudget;
|
|
728
|
-
if (this.#adaptiveReasoningProviderOptions !== undefined)
|
|
729
|
-
return null;
|
|
730
|
-
if (outputBudget === null) {
|
|
731
|
-
throw new TypeError(`${this.#source}: manual provider reasoning requires a resolved total output budget`);
|
|
732
|
-
}
|
|
733
|
-
if (outputBudget <= MANUAL_REASONING_MINIMUM) {
|
|
734
|
-
throw new TypeError(`${this.#source}: total output budget must exceed the provider's ${MANUAL_REASONING_MINIMUM}-token minimum reasoning allowance`);
|
|
735
|
-
}
|
|
736
|
-
const fraction = MANUAL_REASONING_FRACTIONS[this.#reasoning.mode];
|
|
737
|
-
return Math.min(outputBudget - 1, Math.max(MANUAL_REASONING_MINIMUM, Math.round(outputBudget * fraction)));
|
|
738
|
-
}
|
|
739
420
|
#accounting(outcome, usage, evidence, status) {
|
|
740
421
|
const knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
|
|
741
422
|
const direct = this.#normalizeCost?.(evidence);
|
|
@@ -762,7 +443,7 @@ export default class AiSdkProvider {
|
|
|
762
443
|
// supplied grammar before the call but withholds it from the backend.
|
|
763
444
|
const wantGrammar = grammar !== undefined && this.#grammarStyle !== "none";
|
|
764
445
|
if (wantGrammar && this.#gbnfDebug)
|
|
765
|
-
this.#assertGrammarValid(grammar);
|
|
446
|
+
this.#requestBody.assertGrammarValid(grammar);
|
|
766
447
|
const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
|
|
767
448
|
const preserveGrammarSentence = wantGrammar
|
|
768
449
|
&& this.#reasoningStyle === "template";
|
|
@@ -776,7 +457,7 @@ export default class AiSdkProvider {
|
|
|
776
457
|
// {§provider-flexed-allowance} (#482): the wire grants the flexed
|
|
777
458
|
// allowance — the floor, or the exactly-measured slack above it.
|
|
778
459
|
const effectiveMaxOutputTokens = capacity.responseMax ?? capacity.outputBudget ?? undefined;
|
|
779
|
-
const nativeReasoningBudget = this.#nativeReasoningBudget(capacity.outputBudget, capacity.reasoningBudget);
|
|
460
|
+
const nativeReasoningBudget = this.#requestBody.nativeReasoningBudget(capacity.outputBudget, capacity.reasoningBudget);
|
|
780
461
|
// Assembly order = precedence: the family's sampling DEFAULTS
|
|
781
462
|
// (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
|
|
782
463
|
// paths and the name promises every request) < the caller's `sampling`
|
|
@@ -784,24 +465,24 @@ export default class AiSdkProvider {
|
|
|
784
465
|
const body = {
|
|
785
466
|
// Floors are suppressed on router-owned-tuning providers (plurnk) —
|
|
786
467
|
// the router's per-model tuning must not be overridden by client floors.
|
|
787
|
-
...(this.#tuningFloors ? { ...(this.#temperature !== null ? { temperature: this.#temperature } : {}), ...this.#repetitionPenaltyBody() } : {}),
|
|
788
|
-
...this.#samplingBody(sampling),
|
|
468
|
+
...(this.#tuningFloors ? { ...(this.#temperature !== null ? { temperature: this.#temperature } : {}), ...this.#requestBody.repetitionPenaltyBody() } : {}),
|
|
469
|
+
...this.#requestBody.samplingBody(sampling),
|
|
789
470
|
...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
|
|
790
471
|
model: this.#model,
|
|
791
472
|
messages,
|
|
792
|
-
...this.#reasoningBody(preserveGrammarSentence, capacity.reasoningBudget),
|
|
793
|
-
...this.#grammarBody(sendGrammar),
|
|
473
|
+
...this.#requestBody.reasoningBody(preserveGrammarSentence, capacity.reasoningBudget),
|
|
474
|
+
...this.#requestBody.grammarBody(sendGrammar),
|
|
794
475
|
...(effectiveMaxOutputTokens !== undefined ? { max_tokens: effectiveMaxOutputTokens } : {}),
|
|
795
476
|
// Request per-token logprobs only when enabled (managed field —
|
|
796
477
|
// reserved from caller sampling; the env flag is the single control).
|
|
797
478
|
...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
|
|
798
|
-
...this.#slotBody(workerId),
|
|
479
|
+
...this.#requestBody.slotBody(workerId),
|
|
799
480
|
...(this.#cacheAffinity?.target === "body"
|
|
800
481
|
? { [this.#cacheAffinity.name]: workerId }
|
|
801
482
|
: {}),
|
|
802
483
|
};
|
|
803
484
|
// Per-request headers = static auth/routing + any first-party telemetry.
|
|
804
|
-
const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind);
|
|
485
|
+
const metaHeaders = this.#requestBody.metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind);
|
|
805
486
|
const headers = new Headers(this.#headers);
|
|
806
487
|
if (this.#cacheAffinity?.target === "header") {
|
|
807
488
|
headers.set(this.#cacheAffinity.name, workerId);
|
|
@@ -910,7 +591,7 @@ export default class AiSdkProvider {
|
|
|
910
591
|
: await executeAiSdkModel({
|
|
911
592
|
languageModel: this.#languageModel,
|
|
912
593
|
headers: requestHeaders,
|
|
913
|
-
providerOptions: this.#requestProviderOptions(workerId, nativeReasoningBudget),
|
|
594
|
+
providerOptions: this.#requestBody.requestProviderOptions(workerId, nativeReasoningBudget),
|
|
914
595
|
systemProviderOptions: this.#systemCacheProviderOptions,
|
|
915
596
|
messages,
|
|
916
597
|
signal: operationSignal,
|
|
@@ -935,7 +616,7 @@ export default class AiSdkProvider {
|
|
|
935
616
|
? sampling.stop
|
|
936
617
|
: undefined,
|
|
937
618
|
seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
|
|
938
|
-
maxOutputTokens: this.#nativeMaxOutputTokens(capacity.outputBudget, nativeReasoningBudget),
|
|
619
|
+
maxOutputTokens: this.#requestBody.nativeMaxOutputTokens(capacity.outputBudget, nativeReasoningBudget),
|
|
939
620
|
reasoning: this.#reasoning.mode === "off"
|
|
940
621
|
? "none"
|
|
941
622
|
: this.#reasoning.mode === "adaptive"
|