@plurnk/plurnk-providers 1.14.2 → 1.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +1 -1
- package/SPEC.md +16 -2
- package/dist/AiSdkProvider.d.ts +3 -0
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +50 -339
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/AiSdkRequestBody.d.ts +40 -0
- package/dist/AiSdkRequestBody.d.ts.map +1 -0
- package/dist/AiSdkRequestBody.js +361 -0
- package/dist/AiSdkRequestBody.js.map +1 -0
- package/dist/Mock.d.ts +5 -1
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +9 -2
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +2 -1
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +5 -0
- package/dist/Pool.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +1 -1
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +21 -2
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/capacity.d.ts +0 -1
- package/dist/capacity.d.ts.map +1 -1
- package/dist/capacity.js +1 -1
- package/dist/capacity.js.map +1 -1
- package/dist/catalogProvider.d.ts +2 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +6 -0
- package/dist/catalogProvider.js.map +1 -1
- package/dist/index.d.ts +3 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -0
- package/dist/index.js.map +1 -1
- package/dist/promptTokens.d.ts.map +1 -1
- package/dist/promptTokens.js +2 -1
- package/dist/promptTokens.js.map +1 -1
- package/dist/reasoning-effort.d.ts +4 -0
- package/dist/reasoning-effort.d.ts.map +1 -0
- package/dist/reasoning-effort.js +14 -0
- package/dist/reasoning-effort.js.map +1 -0
- package/dist/types.d.ts +17 -1
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +5 -0
- package/dist/types.js.map +1 -1
- package/package.json +6 -6
- package/src/AiSdkProvider.test.ts +22 -3
- package/src/AiSdkProvider.ts +51 -375
- package/src/AiSdkRequestBody.ts +419 -0
- package/src/Mock.ts +10 -2
- package/src/Pool.test.ts +1 -0
- package/src/Pool.ts +6 -1
- package/src/aiSdkTransport.ts +23 -4
- package/src/boundaries.test.ts +2 -0
- package/src/capacity.ts +1 -1
- package/src/catalogProvider.ts +9 -1
- package/src/index.ts +3 -0
- package/src/inputModalities.test.ts +53 -0
- package/src/promptTokens.ts +2 -1
- package/src/reasoning-effort.ts +15 -0
- package/src/types.ts +18 -1
|
@@ -0,0 +1,419 @@
|
|
|
1
|
+
// The provider-specific request body and headers one generate call sends: reasoning, grammar, sampling, repetition, slots, metadata. Split out of AiSdkProvider; every knob it reads is injected.
|
|
2
|
+
import type { ProviderCallKind, ReasoningPolicy } from "./types.ts";
|
|
3
|
+
import type { JSONValue } from "ai";
|
|
4
|
+
import { type Reasoning } from "./env.ts";
|
|
5
|
+
import { validateGbnf } from "@plurnk/gbnf";
|
|
6
|
+
import { fixedEffort } from "./reasoning-effort.ts";
|
|
7
|
+
import type { ReasoningStyle, CompatibleReasoningEffort, GrammarStyle, CacheAffinity, AiSdkProviderOptions } from "./AiSdkProvider.ts";
|
|
8
|
+
|
|
9
|
+
const isJsonObject = (value: JSONValue | undefined): value is Record<string, JSONValue> =>
|
|
10
|
+
typeof value === "object" && value !== null && !Array.isArray(value);
|
|
11
|
+
|
|
12
|
+
const mergeJsonObjects = (
|
|
13
|
+
left: Record<string, JSONValue | undefined>,
|
|
14
|
+
right: Record<string, JSONValue | undefined>,
|
|
15
|
+
): Record<string, JSONValue | undefined> => Object.fromEntries(
|
|
16
|
+
[...new Set([...Object.keys(left), ...Object.keys(right)])].map((key) => {
|
|
17
|
+
const leftValue = left[key];
|
|
18
|
+
const rightValue = right[key];
|
|
19
|
+
return [
|
|
20
|
+
key,
|
|
21
|
+
isJsonObject(leftValue) && isJsonObject(rightValue)
|
|
22
|
+
? mergeJsonObjects(leftValue, rightValue)
|
|
23
|
+
: rightValue ?? leftValue,
|
|
24
|
+
];
|
|
25
|
+
}),
|
|
26
|
+
);
|
|
27
|
+
|
|
28
|
+
// Anthropic's older manual-reasoning protocol needs an absolute allowance while
|
|
29
|
+
// PLURNK's durable contract names an effort. These fractions match the native
|
|
30
|
+
// SDK's policy projection, but apply to PLURNK's total envelope rather than the
|
|
31
|
+
// model's physical maximum. The minimum is imposed by the provider protocol.
|
|
32
|
+
const MANUAL_REASONING_FRACTIONS = Object.freeze({
|
|
33
|
+
adaptive: 0.6,
|
|
34
|
+
low: 0.1,
|
|
35
|
+
medium: 0.3,
|
|
36
|
+
high: 0.6,
|
|
37
|
+
xhigh: 0.75,
|
|
38
|
+
max: 0.85,
|
|
39
|
+
} satisfies Record<Exclude<ReasoningPolicy, "off">, number>);
|
|
40
|
+
|
|
41
|
+
const MANUAL_REASONING_MINIMUM = 1024;
|
|
42
|
+
|
|
43
|
+
// Body keys the provider owns — a caller's `sampling` passthrough may not set
|
|
44
|
+
// these. Two families:
|
|
45
|
+
// transport/managed — grammar transport, the stream/JSON choice, slot pinning,
|
|
46
|
+
// data capture ({§provider-evidence}: backend-specific fields never cross the contract);
|
|
47
|
+
// contract invariants — `n` (atomic single completion: choices[0] is the
|
|
48
|
+
// response; n>1 = paid, dropped output), the tool-calling family (tools-in-
|
|
49
|
+
// body doctrine, §2: native tool_calls return null content = a broken turn),
|
|
50
|
+
// modalities/audio (text-only contract), prediction (decode semantics, not
|
|
51
|
+
// sampling), and the token caps (the envelope is the managed maxOutputTokens —
|
|
52
|
+
// sampling must not bypass the consumer's cap).
|
|
53
|
+
// Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
|
|
54
|
+
// platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
|
|
55
|
+
// metadata, store, verbosity) pass through; the managed floors spread UNDER
|
|
56
|
+
// sampling stay deliberately caller-overridable.
|
|
57
|
+
const RESERVED_BODY_KEYS: ReadonlySet<string> = new Set([
|
|
58
|
+
"model", "messages", "stream", "stream_options", "grammar", "response_format", "id_slot", "logprobs", "top_logprobs",
|
|
59
|
+
"reasoning_format", "reasoning_effort", "thinking", "think", "include_reasoning", "chat_template_kwargs", "thinking_budget_tokens", // lexicon-allow: backend wire fields
|
|
60
|
+
"n", "tools", "tool_choice", "functions", "function_call", "parallel_tool_calls",
|
|
61
|
+
"modalities", "audio", "prediction", "max_tokens", "max_completion_tokens",
|
|
62
|
+
"prompt_cache_key",
|
|
63
|
+
]);
|
|
64
|
+
|
|
65
|
+
export default class AiSdkRequestBody {
|
|
66
|
+
readonly #reasoningBudget: number | null;
|
|
67
|
+
readonly #additiveReasoningProvider: "anthropic" | "bedrock" | undefined;
|
|
68
|
+
readonly #reasoning: Reasoning;
|
|
69
|
+
readonly #reasoningToggle: boolean;
|
|
70
|
+
readonly #compatibleAdaptiveReasoning: CompatibleReasoningEffort | "provider-default";
|
|
71
|
+
readonly #compatibleOffReasoning: "none" | undefined;
|
|
72
|
+
readonly #adaptiveReasoningProviderOptions: AiSdkProviderOptions | undefined;
|
|
73
|
+
readonly #repeatPenalty: number | null;
|
|
74
|
+
readonly #frequencyPenalty: number;
|
|
75
|
+
readonly #dryMultiplier: number | undefined;
|
|
76
|
+
readonly #dryBase: number | undefined;
|
|
77
|
+
readonly #dryAllowedLength: number | undefined;
|
|
78
|
+
readonly #repeatLastN: number | undefined;
|
|
79
|
+
readonly #reasoningStyle: ReasoningStyle;
|
|
80
|
+
readonly #source: string;
|
|
81
|
+
readonly #grammarStyle: GrammarStyle;
|
|
82
|
+
readonly #cacheAffinity: CacheAffinity | undefined;
|
|
83
|
+
readonly #reasoningResponseProviderOptions: AiSdkProviderOptions | undefined;
|
|
84
|
+
readonly #firstPartyMetadata: boolean;
|
|
85
|
+
readonly #supportsSlotPinning: boolean;
|
|
86
|
+
readonly #slotCount: number | null;
|
|
87
|
+
#runSlots = new Map<string, number>();
|
|
88
|
+
#nextSlot = 0;
|
|
89
|
+
|
|
90
|
+
constructor({ reasoningBudget, additiveReasoningProvider, reasoning, reasoningToggle, compatibleAdaptiveReasoning, compatibleOffReasoning, adaptiveReasoningProviderOptions, repeatPenalty, frequencyPenalty, dryMultiplier, dryBase, dryAllowedLength, repeatLastN, reasoningStyle, source, grammarStyle, cacheAffinity, reasoningResponseProviderOptions, firstPartyMetadata, supportsSlotPinning, slotCount }: {
|
|
91
|
+
reasoningBudget: number | null;
|
|
92
|
+
additiveReasoningProvider: "anthropic" | "bedrock" | undefined;
|
|
93
|
+
reasoning: Reasoning;
|
|
94
|
+
reasoningToggle: boolean;
|
|
95
|
+
compatibleAdaptiveReasoning: CompatibleReasoningEffort | "provider-default";
|
|
96
|
+
compatibleOffReasoning: "none" | undefined;
|
|
97
|
+
adaptiveReasoningProviderOptions: AiSdkProviderOptions | undefined;
|
|
98
|
+
repeatPenalty: number | null;
|
|
99
|
+
frequencyPenalty: number;
|
|
100
|
+
dryMultiplier: number | undefined;
|
|
101
|
+
dryBase: number | undefined;
|
|
102
|
+
dryAllowedLength: number | undefined;
|
|
103
|
+
repeatLastN: number | undefined;
|
|
104
|
+
reasoningStyle: ReasoningStyle;
|
|
105
|
+
source: string;
|
|
106
|
+
grammarStyle: GrammarStyle;
|
|
107
|
+
cacheAffinity: CacheAffinity | undefined;
|
|
108
|
+
reasoningResponseProviderOptions: AiSdkProviderOptions | undefined;
|
|
109
|
+
firstPartyMetadata: boolean;
|
|
110
|
+
supportsSlotPinning: boolean;
|
|
111
|
+
slotCount: number | null;
|
|
112
|
+
}) {
|
|
113
|
+
this.#reasoningBudget = reasoningBudget;
|
|
114
|
+
this.#additiveReasoningProvider = additiveReasoningProvider;
|
|
115
|
+
this.#reasoning = reasoning;
|
|
116
|
+
this.#reasoningToggle = reasoningToggle;
|
|
117
|
+
this.#compatibleAdaptiveReasoning = compatibleAdaptiveReasoning;
|
|
118
|
+
this.#compatibleOffReasoning = compatibleOffReasoning;
|
|
119
|
+
this.#adaptiveReasoningProviderOptions = adaptiveReasoningProviderOptions;
|
|
120
|
+
this.#repeatPenalty = repeatPenalty;
|
|
121
|
+
this.#frequencyPenalty = frequencyPenalty;
|
|
122
|
+
this.#dryMultiplier = dryMultiplier;
|
|
123
|
+
this.#dryBase = dryBase;
|
|
124
|
+
this.#dryAllowedLength = dryAllowedLength;
|
|
125
|
+
this.#repeatLastN = repeatLastN;
|
|
126
|
+
this.#reasoningStyle = reasoningStyle;
|
|
127
|
+
this.#source = source;
|
|
128
|
+
this.#grammarStyle = grammarStyle;
|
|
129
|
+
this.#cacheAffinity = cacheAffinity;
|
|
130
|
+
this.#reasoningResponseProviderOptions = reasoningResponseProviderOptions;
|
|
131
|
+
this.#firstPartyMetadata = firstPartyMetadata;
|
|
132
|
+
this.#supportsSlotPinning = supportsSlotPinning;
|
|
133
|
+
this.#slotCount = slotCount;
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
// Reasoning activation and allowance are independent of grammar transport;
|
|
137
|
+
// only the response representation becomes lossless when evidence is needed.
|
|
138
|
+
// The llama-server template mapping is owned by {§llama-reasoning-request}.
|
|
139
|
+
reasoningBody(
|
|
140
|
+
preserveGrammarSentence = false,
|
|
141
|
+
reasoningBudget = this.#reasoningBudget,
|
|
142
|
+
): Record<string, unknown> {
|
|
143
|
+
const { mode } = this.#reasoning;
|
|
144
|
+
const budget = reasoningBudget;
|
|
145
|
+
const on = mode !== "off";
|
|
146
|
+
switch (this.#reasoningStyle) {
|
|
147
|
+
case "template": {
|
|
148
|
+
const allowance = mode === "off"
|
|
149
|
+
? 0
|
|
150
|
+
: budget;
|
|
151
|
+
// A fixed effort rides into the template as its own variable; adaptive
|
|
152
|
+
// and off send none and leave the template's default in force.
|
|
153
|
+
const templateEffort = mode === "off" || mode === "adaptive" ? {} : { reasoning_effort: fixedEffort(mode) };
|
|
154
|
+
return {
|
|
155
|
+
chat_template_kwargs: { enable_thinking: on, ...templateEffort },
|
|
156
|
+
reasoning_format: preserveGrammarSentence ? "none" : "auto",
|
|
157
|
+
...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
|
|
158
|
+
};
|
|
159
|
+
}
|
|
160
|
+
case "think": return on ? { think: true } : {};
|
|
161
|
+
case "include_reasoning": return on ? { include_reasoning: true } : {};
|
|
162
|
+
case "effort": return mode === "off"
|
|
163
|
+
? this.#compatibleOffReasoning === undefined
|
|
164
|
+
? {}
|
|
165
|
+
: { reasoning_effort: this.#compatibleOffReasoning }
|
|
166
|
+
: mode === "adaptive"
|
|
167
|
+
? this.#compatibleAdaptiveReasoning === "provider-default"
|
|
168
|
+
? {}
|
|
169
|
+
: { reasoning_effort: this.#compatibleAdaptiveReasoning }
|
|
170
|
+
: { reasoning_effort: fixedEffort(mode) };
|
|
171
|
+
// Graded reasoning is mandatory when the route advertises an effort
|
|
172
|
+
// value. Cataloged routes supply the exact strongest legal value;
|
|
173
|
+
// construction rejects an unsupported off or fixed policy.
|
|
174
|
+
case "effort_required": {
|
|
175
|
+
if (mode === "off") {
|
|
176
|
+
if (this.#compatibleOffReasoning === undefined) {
|
|
177
|
+
throw new TypeError(`${this.#source}: required reasoning effort has no off projection`);
|
|
178
|
+
}
|
|
179
|
+
return { reasoning_effort: this.#compatibleOffReasoning };
|
|
180
|
+
}
|
|
181
|
+
if (mode === "adaptive") {
|
|
182
|
+
return this.#compatibleAdaptiveReasoning === "provider-default"
|
|
183
|
+
? {}
|
|
184
|
+
: { reasoning_effort: this.#compatibleAdaptiveReasoning };
|
|
185
|
+
}
|
|
186
|
+
return { reasoning_effort: fixedEffort(mode) };
|
|
187
|
+
}
|
|
188
|
+
// Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
|
|
189
|
+
// reason-by-default model (DeepSeek V4: default 'high') reasoning.
|
|
190
|
+
// ADAPTIVE omits the field UNLESS the catalog declares a toggle control:
|
|
191
|
+
// toggle routes (nemotron-lightning) default reasoning OFF, so adaptive
|
|
192
|
+
// sends the documented Fireworks Boolean enable (#457). The literal
|
|
193
|
+
// "adaptive" is MiniMax-M3-only — Fireworks 400s it for every other
|
|
194
|
+
// model (wire-verified; the 1.0.2 adaptive default refused to boot on
|
|
195
|
+
// it). V4 gotcha: integer efforts 400.
|
|
196
|
+
case "effort_explicit": return mode === "off"
|
|
197
|
+
? { reasoning_effort: "none" }
|
|
198
|
+
: mode === "adaptive"
|
|
199
|
+
? this.#reasoningToggle ? { reasoning_effort: true } : {}
|
|
200
|
+
: { reasoning_effort: fixedEffort(mode) };
|
|
201
|
+
// {§deepseek-reasoning-request}
|
|
202
|
+
case "thinking_effort": return mode === "off"
|
|
203
|
+
? { thinking: { type: "disabled" } }
|
|
204
|
+
: mode === "adaptive" ? { thinking: { type: "enabled" } } : {
|
|
205
|
+
thinking: { type: "enabled" },
|
|
206
|
+
reasoning_effort: fixedEffort(mode),
|
|
207
|
+
};
|
|
208
|
+
// Anthropic-compatible native dynamic or manual budget mode.
|
|
209
|
+
case "anthropic": return mode === "off"
|
|
210
|
+
? { thinking: { type: "disabled" } }
|
|
211
|
+
: mode === "adaptive" ? { thinking: { type: "adaptive" } } : {
|
|
212
|
+
thinking: {
|
|
213
|
+
type: "enabled",
|
|
214
|
+
budget_tokens: budget!,
|
|
215
|
+
},
|
|
216
|
+
};
|
|
217
|
+
case "none": return {};
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
// Per-worker slot affinity: the consumer passes which worker this is; the
|
|
222
|
+
// provider owns WHICH slot serves it. Sticky per workerId, round-robin across
|
|
223
|
+
// new runs (distinct runs → distinct slots while slots last), LRU-bounded
|
|
224
|
+
// bookkeeping so a long-lived daemon never grows the map unboundedly —
|
|
225
|
+
// an evicted-and-returning run simply re-pins, worst case one cold prefill.
|
|
226
|
+
|
|
227
|
+
// Optional local llama-server GBNF transport ({§gbnf-response-observation}). Unsupported
|
|
228
|
+
// backends receive no grammar-related field.
|
|
229
|
+
grammarBody(grammar: string | undefined): Record<string, unknown> {
|
|
230
|
+
if (grammar === undefined) return {};
|
|
231
|
+
switch (this.#grammarStyle) {
|
|
232
|
+
// Grammar-constrained decoding can loop under the mask; a configured
|
|
233
|
+
// per-alias repeat_penalty is the measured remedy ({§provider-sampling-passthrough}).
|
|
234
|
+
case "llamacpp": return { grammar, ...(this.#repeatPenalty !== null ? { repeat_penalty: this.#repeatPenalty } : {}) };
|
|
235
|
+
case "none": return {};
|
|
236
|
+
}
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
// Anti-degeneration default on every request, keyed to the backend's wire
|
|
241
|
+
// convention - NOT grammar-bound. GBNF is a local constraint, so a cloud
|
|
242
|
+
// alias runs the sampler bare: firefast (deepseek/fireworks) ran 4/86 bench turns
|
|
243
|
+
// straight to the token cap on pure looped repetition (run52). Ships next to
|
|
244
|
+
// temperature so caller `sampling` can tune it; the grammar path re-asserts it as a
|
|
245
|
+
// managed FLOOR in #grammarBody. llama.cpp takes the repeat_penalty
|
|
246
|
+
// MULTIPLIER; the plain cloud path ("none") can't, so it gets
|
|
247
|
+
// frequency_penalty - OpenAI-standard, accepted by every OpenAI-compat backend (verified
|
|
248
|
+
// live: together/deepinfra/fireworks; it is OpenAI's own param, so real OpenAI takes it too).
|
|
249
|
+
repetitionPenaltyBody(): Record<string, unknown> {
|
|
250
|
+
switch (this.#grammarStyle) {
|
|
251
|
+
// repeat_penalty + optional DRY (repeated-sequence penalty) + a wider
|
|
252
|
+
// repeat_last_n window — the loop-breaking tools a llama.cpp backend serves.
|
|
253
|
+
// Each rides only when its operator knob is set; absent = the box's default.
|
|
254
|
+
case "llamacpp": return {
|
|
255
|
+
...(this.#repeatPenalty !== null ? { repeat_penalty: this.#repeatPenalty } : {}),
|
|
256
|
+
...(this.#repeatLastN !== undefined ? { repeat_last_n: this.#repeatLastN } : {}),
|
|
257
|
+
...(this.#dryMultiplier !== undefined && this.#dryMultiplier > 0 ? {
|
|
258
|
+
dry_multiplier: this.#dryMultiplier,
|
|
259
|
+
...(this.#dryBase !== undefined ? { dry_base: this.#dryBase } : {}),
|
|
260
|
+
...(this.#dryAllowedLength !== undefined ? { dry_allowed_length: this.#dryAllowedLength } : {}),
|
|
261
|
+
} : {}),
|
|
262
|
+
};
|
|
263
|
+
case "none": return this.#frequencyPenalty > 0 ? { frequency_penalty: this.#frequencyPenalty } : {};
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
// Caller-supplied OpenAI-compat sampling params (temperature, top_p, top_k,
|
|
269
|
+
// penalties, stop, seed, …) merged UNDER the managed body: model, messages,
|
|
270
|
+
// reasoning, grammar (+ its repeat-penalty floor), max_tokens and slot always
|
|
271
|
+
// win, and reserved transport/protocol keys are stripped so the passthrough
|
|
272
|
+
// can't smuggle a grammar, a stream toggle, or a backend slot
|
|
273
|
+
// ({§provider-request-authority}).
|
|
274
|
+
samplingBody(sampling: Record<string, unknown> | undefined): Record<string, unknown> {
|
|
275
|
+
if (sampling === undefined) return {};
|
|
276
|
+
const out: Record<string, unknown> = {};
|
|
277
|
+
for (const [k, v] of Object.entries(sampling)) if (!RESERVED_BODY_KEYS.has(k)) out[k] = v;
|
|
278
|
+
return out;
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
slotBody(workerId: string): Record<string, unknown> {
|
|
283
|
+
if (!this.#supportsSlotPinning || this.#slotCount === null || this.#slotCount < 1) return {};
|
|
284
|
+
let slot = this.#runSlots.get(workerId);
|
|
285
|
+
if (slot === undefined) {
|
|
286
|
+
slot = this.#nextSlot++ % this.#slotCount;
|
|
287
|
+
if (this.#runSlots.size >= this.#slotCount * 8) {
|
|
288
|
+
this.#runSlots.delete(this.#runSlots.keys().next().value as string);
|
|
289
|
+
}
|
|
290
|
+
} else {
|
|
291
|
+
this.#runSlots.delete(workerId); // re-insert to refresh LRU recency
|
|
292
|
+
}
|
|
293
|
+
this.#runSlots.set(workerId, slot);
|
|
294
|
+
return { id_slot: slot };
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
// First-party telemetry headers ({§provider-request-authority} {§provider-call-kind}): forwarded only when the spec
|
|
299
|
+
// opted in (the plurnk endpoint). The gate is here, not at the call site, so
|
|
300
|
+
// attributions/client/strikes can never reach a third-party backend even if
|
|
301
|
+
// the consumer passes them to the wrong provider. Empty values emit no header
|
|
302
|
+
// — EXCEPT strikes, where 0 is a real value (clean streak) distinct from
|
|
303
|
+
// absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
|
|
304
|
+
// ride HTTP headers only — the packet never carries them (the model must
|
|
305
|
+
// never see strike state; engine accounting is not a metric to game).
|
|
306
|
+
metadataHeaders(attributions: string[] | undefined, client: string | undefined, strikes: number | undefined, workerId: string, primaryWorkerId: string | undefined, workspaceId: string | undefined, loop: number | undefined, turn: number | undefined, callKind: ProviderCallKind | undefined): Record<string, string> {
|
|
307
|
+
if (!this.#firstPartyMetadata) return {};
|
|
308
|
+
const h: Record<string, string> = {};
|
|
309
|
+
if (attributions !== undefined && attributions.length > 0) h["Plurnk-Attribution"] = JSON.stringify(attributions);
|
|
310
|
+
if (client !== undefined && client.length > 0) h["Plurnk-Client"] = client;
|
|
311
|
+
if (strikes !== undefined && Number.isInteger(strikes) && strikes >= 0) h["Plurnk-Strikes"] = String(strikes);
|
|
312
|
+
// Worker identity: the opaque workerId
|
|
313
|
+
// the consumer already supplies, forwarded so the endpoint can key
|
|
314
|
+
// per-worker affinity/telemetry — same gate as every first-party signal.
|
|
315
|
+
h["Plurnk-Worker-Id"] = workerId;
|
|
316
|
+
// Root worker of the lineage ({§worker-primary}): the no-parent ancestor of this turn's
|
|
317
|
+
// worker tree. The consumer classifies primary-vs-spawned by equality
|
|
318
|
+
// (primaryWorkerId == workerId ⇒ the primary/root worker). The provider
|
|
319
|
+
// EMITS what the consumer supplies and never invents a primary; the
|
|
320
|
+
// consumer's contract is to stamp it EVERY turn (including the primary's
|
|
321
|
+
// own, where it equals workerId). Absence is the consumer's violation for
|
|
322
|
+
// the endpoint to surface, not a provider default.
|
|
323
|
+
if (primaryWorkerId !== undefined && primaryWorkerId.length > 0) h["Plurnk-Worker-Primary"] = primaryWorkerId;
|
|
324
|
+
// Turn coordinate ({§lifecycle-terms}): workspace/loop/turn, the
|
|
325
|
+
// daemon-side sequence the endpoint can never scrape from the wire.
|
|
326
|
+
// Coordinates are 1-based — 0 is not a real value, so no strikes-style
|
|
327
|
+
// zero exception; absent/empty/0 emits no header.
|
|
328
|
+
if (workspaceId !== undefined && workspaceId.length > 0) h["Plurnk-Workspace-Id"] = workspaceId;
|
|
329
|
+
if (loop !== undefined && Number.isInteger(loop) && loop >= 1) h["Plurnk-Loop"] = String(loop);
|
|
330
|
+
if (turn !== undefined && Number.isInteger(turn) && turn >= 1) h["Plurnk-Turn"] = String(turn);
|
|
331
|
+
if (callKind !== undefined) h["Plurnk-Call-Kind"] = callKind;
|
|
332
|
+
return h;
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
requestProviderOptions(
|
|
337
|
+
workerId: string,
|
|
338
|
+
nativeReasoningBudget: number | null,
|
|
339
|
+
): AiSdkProviderOptions | undefined {
|
|
340
|
+
const responseOptions = this.#reasoning.mode === "off"
|
|
341
|
+
? undefined
|
|
342
|
+
: this.#reasoningResponseProviderOptions;
|
|
343
|
+
const adaptiveOptions = this.#reasoning.mode === "adaptive"
|
|
344
|
+
&& nativeReasoningBudget === null
|
|
345
|
+
? this.#adaptiveReasoningProviderOptions
|
|
346
|
+
: undefined;
|
|
347
|
+
const nativeReasoning = nativeReasoningBudget !== null
|
|
348
|
+
? this.#additiveReasoningProvider === "anthropic"
|
|
349
|
+
? { anthropic: { thinking: { type: "enabled", budgetTokens: nativeReasoningBudget } } }
|
|
350
|
+
: this.#additiveReasoningProvider === "bedrock"
|
|
351
|
+
? { bedrock: { reasoningConfig: { type: "enabled", budgetTokens: nativeReasoningBudget } } }
|
|
352
|
+
: undefined
|
|
353
|
+
: undefined;
|
|
354
|
+
const options: AiSdkProviderOptions = {};
|
|
355
|
+
for (const part of [responseOptions, adaptiveOptions, nativeReasoning]) {
|
|
356
|
+
for (const [provider, values] of Object.entries(part ?? {})) {
|
|
357
|
+
options[provider] = mergeJsonObjects(options[provider] ?? {}, values);
|
|
358
|
+
}
|
|
359
|
+
}
|
|
360
|
+
if (this.#cacheAffinity?.target === "provider-option") {
|
|
361
|
+
const { provider, name } = this.#cacheAffinity;
|
|
362
|
+
options[provider] = { ...options[provider], [name]: workerId };
|
|
363
|
+
}
|
|
364
|
+
return Object.keys(options).length === 0 ? undefined : options;
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
|
|
368
|
+
nativeReasoningBudget(
|
|
369
|
+
outputBudget: number | null,
|
|
370
|
+
configuredReasoningBudget: number | null,
|
|
371
|
+
): number | null {
|
|
372
|
+
if (this.#additiveReasoningProvider === undefined || this.#reasoning.mode === "off") return null;
|
|
373
|
+
if (configuredReasoningBudget !== null) return configuredReasoningBudget;
|
|
374
|
+
if (this.#adaptiveReasoningProviderOptions !== undefined) return null;
|
|
375
|
+
if (outputBudget === null) {
|
|
376
|
+
throw new TypeError(
|
|
377
|
+
`${this.#source}: manual provider reasoning requires a resolved total output budget`,
|
|
378
|
+
);
|
|
379
|
+
}
|
|
380
|
+
if (outputBudget <= MANUAL_REASONING_MINIMUM) {
|
|
381
|
+
throw new TypeError(
|
|
382
|
+
`${this.#source}: total output budget must exceed the provider's ${MANUAL_REASONING_MINIMUM}-token minimum reasoning allowance`,
|
|
383
|
+
);
|
|
384
|
+
}
|
|
385
|
+
const fraction = MANUAL_REASONING_FRACTIONS[this.#reasoning.mode];
|
|
386
|
+
return Math.min(
|
|
387
|
+
outputBudget - 1,
|
|
388
|
+
Math.max(MANUAL_REASONING_MINIMUM, Math.round(outputBudget * fraction)),
|
|
389
|
+
);
|
|
390
|
+
}
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
nativeMaxOutputTokens(
|
|
394
|
+
outputBudget: number | null,
|
|
395
|
+
nativeReasoningBudget: number | null,
|
|
396
|
+
): number | undefined {
|
|
397
|
+
if (outputBudget === null) return undefined;
|
|
398
|
+
return nativeReasoningBudget !== null
|
|
399
|
+
? outputBudget - nativeReasoningBudget
|
|
400
|
+
: outputBudget;
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
// PLURNK_PROVIDERS_GBNF_DEBUG ({§gbnf-response-observation}): validate the supplied GBNF locally and fail
|
|
405
|
+
// hard if it's malformed, BEFORE any wire call — and the grammar is NOT
|
|
406
|
+
// transported, so the request runs unconstrained. A debug aid to catch invalid
|
|
407
|
+
// grammars (e.g. while editing the plurnk grammar) without a model round-trip;
|
|
408
|
+
// off in production. `validateGbnf(grammar, "")` parses the grammar + resolves
|
|
409
|
+
// its root, throwing iff the grammar itself is invalid (the empty input's
|
|
410
|
+
// verdict is irrelevant — we only care that parsing succeeded).
|
|
411
|
+
assertGrammarValid(grammar: string): void {
|
|
412
|
+
try {
|
|
413
|
+
validateGbnf(grammar, "");
|
|
414
|
+
} catch (cause) {
|
|
415
|
+
throw new Error(`grammar validation (PLURNK_PROVIDERS_GBNF_DEBUG): invalid GBNF — ${(cause as Error).message}`, { cause });
|
|
416
|
+
}
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
}
|
package/src/Mock.ts
CHANGED
|
@@ -5,6 +5,8 @@
|
|
|
5
5
|
// Provider contract. Production providers don't expose the `ops` escape
|
|
6
6
|
// hatch — that's an intg-only convenience.
|
|
7
7
|
|
|
8
|
+
import { chatMessageText } from "./types.ts";
|
|
9
|
+
import type { InputModality } from "./types.ts";
|
|
8
10
|
import type { ChatMessage, FinishReason, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderAssistant, ProviderCost, ProviderEncryptedReasoningItem, ProviderRequestAccounting, ProviderRequestCapacity, ProviderResponse, ProviderUsage } from "./types.ts";
|
|
9
11
|
import { resolveGenerationEnvelopeFromEnv } from "./env.ts";
|
|
10
12
|
import { REASONING_POLICIES } from "@plurnk/plurnk-contracts";
|
|
@@ -48,6 +50,9 @@ type MockGenerateArgs = Omit<Parameters<Provider["generate"]>[0], "workerId"> &
|
|
|
48
50
|
|
|
49
51
|
export default class Mock implements Provider {
|
|
50
52
|
#contextWindow: number | null;
|
|
53
|
+
#inputModalities: ReadonlySet<InputModality>;
|
|
54
|
+
// Every request as received, newest last — the witness for what reached the provider.
|
|
55
|
+
readonly received: ChatMessage[][] = [];
|
|
51
56
|
#outputBudget: number | null;
|
|
52
57
|
#reasoningBudget: number | null;
|
|
53
58
|
#queue: MockResponse[];
|
|
@@ -57,8 +62,9 @@ export default class Mock implements Provider {
|
|
|
57
62
|
// also the universal test fixture. No output budget means its context
|
|
58
63
|
// window alone cannot determine an input capacity. Mock has no alias
|
|
59
64
|
// identity, so it reads the bare knobs.
|
|
60
|
-
constructor({ contextWindow, responses }: { contextWindow: number | null; responses: MockResponse[] }) {
|
|
65
|
+
constructor({ contextWindow, responses, inputModalities = [] }: { contextWindow: number | null; responses: MockResponse[]; inputModalities?: Iterable<InputModality> }) {
|
|
61
66
|
this.#contextWindow = contextWindow;
|
|
67
|
+
this.#inputModalities = new Set(inputModalities);
|
|
62
68
|
const envelope = resolveGenerationEnvelopeFromEnv(process.env, contextWindow);
|
|
63
69
|
this.#outputBudget = envelope.outputBudget;
|
|
64
70
|
this.#reasoningBudget = envelope.reasoningBudget;
|
|
@@ -66,6 +72,7 @@ export default class Mock implements Provider {
|
|
|
66
72
|
}
|
|
67
73
|
|
|
68
74
|
get contextWindow(): number | null { return this.#contextWindow; }
|
|
75
|
+
get inputModalities(): ReadonlySet<InputModality> { return this.#inputModalities; }
|
|
69
76
|
get maxInputTokens(): number | null { return null; }
|
|
70
77
|
get maxOutputTokens(): number | null { return null; }
|
|
71
78
|
get outputBudget(): number | null { return this.#outputBudget; }
|
|
@@ -86,7 +93,7 @@ export default class Mock implements Provider {
|
|
|
86
93
|
async countPromptTokens(messages: readonly ChatMessage[]): Promise<PromptTokenMeasurement> {
|
|
87
94
|
return {
|
|
88
95
|
kind: "exact",
|
|
89
|
-
tokens: messages.reduce((sum,
|
|
96
|
+
tokens: messages.reduce((sum, message) => sum + Math.ceil(chatMessageText(message).length / 2), 0),
|
|
90
97
|
source: "mock:chars2",
|
|
91
98
|
};
|
|
92
99
|
}
|
|
@@ -120,6 +127,7 @@ export default class Mock implements Provider {
|
|
|
120
127
|
// "wire call" and must not exhaust a queued response
|
|
121
128
|
// ({§provider-failure-normalization}).
|
|
122
129
|
signal?.throwIfAborted();
|
|
130
|
+
this.received.push(messages.map((message) => ({ ...message })));
|
|
123
131
|
const capacity = await this.assessRequestCapacity(messages, maxOutputTokens);
|
|
124
132
|
if (capacity.decision === "reject") {
|
|
125
133
|
throw new ProviderError("mock", "capacity_exceeded", "Mock request exceeds its exact input capacity.", {
|
package/src/Pool.test.ts
CHANGED
|
@@ -49,6 +49,7 @@ const backend = (opts: FakeOpts = {}) => {
|
|
|
49
49
|
const served: string[] = [];
|
|
50
50
|
const b: Provider = {
|
|
51
51
|
model: opts.model ?? "gemma",
|
|
52
|
+
inputModalities: new Set(),
|
|
52
53
|
contextWindow: opts.window === undefined ? 48000 : opts.window,
|
|
53
54
|
maxInputTokens: opts.maxInputTokens ?? null,
|
|
54
55
|
maxOutputTokens: opts.maxOutputTokens ?? null,
|
package/src/Pool.ts
CHANGED
|
@@ -7,7 +7,7 @@ import Meta, {
|
|
|
7
7
|
type PluginAttributionContext,
|
|
8
8
|
} from "@plurnk/plurnk-meta";
|
|
9
9
|
import { effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget, requestCapacityDecision } from "./capacity.ts";
|
|
10
|
-
import type { ReasoningPolicy } from "./types.ts";
|
|
10
|
+
import type { InputModality, ReasoningPolicy } from "./types.ts";
|
|
11
11
|
|
|
12
12
|
// A backend-AVAILABILITY failure: the sub-provider already exhausted its OWN
|
|
13
13
|
// transient retries before throwing one of these, so re-hitting the same
|
|
@@ -80,6 +80,11 @@ export default class Pool implements Provider {
|
|
|
80
80
|
|
|
81
81
|
get model(): string { return this.#backends[0].model; }
|
|
82
82
|
get contextWindow(): number | null { return this.#floor.contextWindow; }
|
|
83
|
+
// {§provider-input-modalities} — a pool accepts a modality only when every backend does.
|
|
84
|
+
get inputModalities(): ReadonlySet<InputModality> {
|
|
85
|
+
const [first, ...rest] = this.#backends.map((backend) => backend.inputModalities);
|
|
86
|
+
return new Set([...(first ?? [])].filter((modality) => rest.every((set) => set.has(modality))));
|
|
87
|
+
}
|
|
83
88
|
#minimumKnown(project: (provider: Provider) => number | null): number | null {
|
|
84
89
|
const values = this.#backends.map(project);
|
|
85
90
|
return values.some((value) => value === null)
|
package/src/aiSdkTransport.ts
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
|
+
import { chatMessageText } from "./types.ts";
|
|
1
2
|
import { createOpenAICompatible, type ProviderErrorStructure } from "@ai-sdk/openai-compatible";
|
|
2
|
-
import { APICallError, generateText, streamText, type CallWarning, type JSONValue, type LanguageModel, type LanguageModelUsage } from "ai";
|
|
3
|
+
import { APICallError, generateText, streamText, type CallWarning, type JSONValue, type LanguageModel, type LanguageModelUsage, type ModelMessage } from "ai";
|
|
3
4
|
import { z } from "zod";
|
|
4
5
|
import type { ChatMessage, ProviderAttemptFinishReason, ProviderChargeEvidence, ProviderReasoningObserver, ProviderUsage, TokenLogprob } from "./types.ts";
|
|
5
6
|
import { normalizeUsage, type RawUsage } from "./usage.ts";
|
|
@@ -190,7 +191,7 @@ export type AiSdkTransportResponse = {
|
|
|
190
191
|
warnings: readonly CallWarning[];
|
|
191
192
|
};
|
|
192
193
|
|
|
193
|
-
|
|
194
|
+
type AiSdkModelRequest = Omit<AiSdkTransportRequest, "url" | "model" | "body" | "fetch"> & {
|
|
194
195
|
languageModel: LanguageModel;
|
|
195
196
|
providerOptions?: Record<string, Record<string, JSONValue | undefined>>;
|
|
196
197
|
systemProviderOptions?: Record<string, Record<string, JSONValue | undefined>>;
|
|
@@ -369,12 +370,30 @@ const executeModelOnce = async (
|
|
|
369
370
|
}
|
|
370
371
|
const instructions = request.messages.slice(0, instructionCount).map(({ content }, index) => ({
|
|
371
372
|
role: "system" as const,
|
|
372
|
-
content,
|
|
373
|
+
content: chatMessageText({ content }),
|
|
373
374
|
...(systemProviderOptions !== undefined && index === instructionCount - 1
|
|
374
375
|
? { providerOptions: systemProviderOptions }
|
|
375
376
|
: {}),
|
|
376
377
|
}));
|
|
377
|
-
|
|
378
|
+
// {§provider-input-modalities} — conversational messages in the SDK's own shape: a user message
|
|
379
|
+
// may be parts (text beside native images and files); every other role is text.
|
|
380
|
+
const messages: ModelMessage[] = request.messages.slice(instructionCount).map((message): ModelMessage => {
|
|
381
|
+
if (message.role === "user") {
|
|
382
|
+
return typeof message.content === "string"
|
|
383
|
+
? { role: "user", content: message.content }
|
|
384
|
+
: {
|
|
385
|
+
role: "user",
|
|
386
|
+
content: message.content.map((part) => part.type === "text"
|
|
387
|
+
? { type: "text" as const, text: part.text }
|
|
388
|
+
: part.type === "image"
|
|
389
|
+
? { type: "image" as const, image: part.image, mediaType: part.mediaType }
|
|
390
|
+
: { type: "file" as const, data: part.data, mediaType: part.mediaType }),
|
|
391
|
+
};
|
|
392
|
+
}
|
|
393
|
+
return message.role === "assistant"
|
|
394
|
+
? { role: "assistant", content: chatMessageText(message) }
|
|
395
|
+
: { role: "system", content: chatMessageText(message) };
|
|
396
|
+
});
|
|
378
397
|
const common = {
|
|
379
398
|
model,
|
|
380
399
|
...(instructions.length === 0 ? {} : { instructions }),
|
package/src/boundaries.test.ts
CHANGED
|
@@ -46,6 +46,7 @@ test("the OpenAI-compatible entrypoint excludes Node-owned provider machinery",
|
|
|
46
46
|
assertRuntimeNeutralGraph("openai.ts", new Set([
|
|
47
47
|
"accounting.ts",
|
|
48
48
|
"AiSdkProvider.ts",
|
|
49
|
+
"AiSdkRequestBody.ts",
|
|
49
50
|
"aiSdkTransport.ts",
|
|
50
51
|
"capacity.ts",
|
|
51
52
|
"cost.ts",
|
|
@@ -55,6 +56,7 @@ test("the OpenAI-compatible entrypoint excludes Node-owned provider machinery",
|
|
|
55
56
|
"openai.ts",
|
|
56
57
|
"promptTokens.ts",
|
|
57
58
|
"providerError.ts",
|
|
59
|
+
"reasoning-effort.ts",
|
|
58
60
|
"types.ts",
|
|
59
61
|
"usage.ts",
|
|
60
62
|
"warnings.ts",
|
package/src/capacity.ts
CHANGED
|
@@ -86,7 +86,7 @@ export const effectiveInputCapacity = ({
|
|
|
86
86
|
// unclaimed is guaranteed free and becomes response runway. Only an exact
|
|
87
87
|
// prompt measurement may claim slack — an estimate proves nothing about the
|
|
88
88
|
// true remainder — and the model's own maxOutputTokens still caps the grant.
|
|
89
|
-
|
|
89
|
+
const WIRE_FLEX_MARGIN = 256;
|
|
90
90
|
|
|
91
91
|
export const flexedResponseMax = ({
|
|
92
92
|
contextWindow,
|
package/src/catalogProvider.ts
CHANGED
|
@@ -28,7 +28,8 @@ import AiSdkProvider, {
|
|
|
28
28
|
} from "./AiSdkProvider.ts";
|
|
29
29
|
import { configuredProviderInfo, createSdkModel } from "./sdkModels.ts";
|
|
30
30
|
import { providerSource } from "./notices.ts";
|
|
31
|
-
import type { Provider, ProviderCostNormalizer } from "./types.ts";
|
|
31
|
+
import type { InputModality, Provider, ProviderCostNormalizer } from "./types.ts";
|
|
32
|
+
import { INPUT_MODALITIES } from "./types.ts";
|
|
32
33
|
import { REASONING_POLICIES, type ReasoningPolicy } from "@plurnk/plurnk-contracts";
|
|
33
34
|
import { estimateProviderCost } from "./cost.ts";
|
|
34
35
|
import { emitWarningOnce } from "./warnings.ts";
|
|
@@ -53,6 +54,11 @@ const reasoningStyleFromEnv = (
|
|
|
53
54
|
return value as ReasoningStyle;
|
|
54
55
|
};
|
|
55
56
|
|
|
57
|
+
// {§provider-input-modalities} — the catalog's input modalities, kept to the vocabulary the wire
|
|
58
|
+
// can carry; an unknown model declares none.
|
|
59
|
+
export const inputModalitiesOf = (input: readonly string[] | undefined): ReadonlySet<InputModality> =>
|
|
60
|
+
new Set((input ?? []).filter((modality): modality is InputModality => (INPUT_MODALITIES as readonly string[]).includes(modality)));
|
|
61
|
+
|
|
56
62
|
// {§provider-reasoning-policy} — the operator's affirmative declaration of efforts a provider's
|
|
57
63
|
// reasoning routes accept beyond Models.dev; the daemon adds none on its own (#439).
|
|
58
64
|
const DECLARABLE_EFFORTS: ReadonlySet<ModelReasoningEffort> = new Set([
|
|
@@ -337,6 +343,8 @@ export const providerFromSdkModel = ({
|
|
|
337
343
|
...(url === undefined ? {} : { url }),
|
|
338
344
|
...(headers === undefined ? {} : { headers: { ...headers } }),
|
|
339
345
|
contextWindow,
|
|
346
|
+
// {§provider-input-modalities} — the catalog's input modalities decide which native parts ride.
|
|
347
|
+
inputModalities: inputModalitiesOf(info?.modalities.input),
|
|
340
348
|
maxInputTokens,
|
|
341
349
|
maxOutputTokens,
|
|
342
350
|
outputBudget: envelope.outputBudget,
|
package/src/index.ts
CHANGED
|
@@ -96,3 +96,6 @@ export type {
|
|
|
96
96
|
export { default as Mock } from "./Mock.ts";
|
|
97
97
|
export type { MockAssistant, MockResponse, MockReturnedAssistant } from "./Mock.ts";
|
|
98
98
|
export { mockDefaultUsage } from "./Mock.ts";
|
|
99
|
+
export type { ChatContentPart, InputModality } from "./types.ts";
|
|
100
|
+
export { INPUT_MODALITIES } from "./types.ts";
|
|
101
|
+
export { chatMessageText } from "./types.ts";
|