@plurnk/plurnk-providers 1.14.2 → 1.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +1 -1
- package/SPEC.md +16 -2
- package/dist/AiSdkProvider.d.ts +3 -0
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +50 -339
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/AiSdkRequestBody.d.ts +40 -0
- package/dist/AiSdkRequestBody.d.ts.map +1 -0
- package/dist/AiSdkRequestBody.js +361 -0
- package/dist/AiSdkRequestBody.js.map +1 -0
- package/dist/Mock.d.ts +5 -1
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +9 -2
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +2 -1
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +5 -0
- package/dist/Pool.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +1 -1
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +21 -2
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/capacity.d.ts +0 -1
- package/dist/capacity.d.ts.map +1 -1
- package/dist/capacity.js +1 -1
- package/dist/capacity.js.map +1 -1
- package/dist/catalogProvider.d.ts +2 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +6 -0
- package/dist/catalogProvider.js.map +1 -1
- package/dist/index.d.ts +3 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -0
- package/dist/index.js.map +1 -1
- package/dist/promptTokens.d.ts.map +1 -1
- package/dist/promptTokens.js +2 -1
- package/dist/promptTokens.js.map +1 -1
- package/dist/reasoning-effort.d.ts +4 -0
- package/dist/reasoning-effort.d.ts.map +1 -0
- package/dist/reasoning-effort.js +14 -0
- package/dist/reasoning-effort.js.map +1 -0
- package/dist/types.d.ts +17 -1
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +5 -0
- package/dist/types.js.map +1 -1
- package/package.json +6 -6
- package/src/AiSdkProvider.test.ts +22 -3
- package/src/AiSdkProvider.ts +51 -375
- package/src/AiSdkRequestBody.ts +419 -0
- package/src/Mock.ts +10 -2
- package/src/Pool.test.ts +1 -0
- package/src/Pool.ts +6 -1
- package/src/aiSdkTransport.ts +23 -4
- package/src/boundaries.test.ts +2 -0
- package/src/capacity.ts +1 -1
- package/src/catalogProvider.ts +9 -1
- package/src/index.ts +3 -0
- package/src/inputModalities.test.ts +53 -0
- package/src/promptTokens.ts +2 -1
- package/src/reasoning-effort.ts +15 -0
- package/src/types.ts +18 -1
package/src/AiSdkProvider.ts
CHANGED
|
@@ -6,38 +6,18 @@
|
|
|
6
6
|
// ordinary vendor protocol. The compatible URL path remains only for PLURNK
|
|
7
7
|
// extensions and local endpoint probes the SDK cannot represent.
|
|
8
8
|
|
|
9
|
-
import type {
|
|
10
|
-
ChatMessage,
|
|
11
|
-
GrammarEvidence,
|
|
12
|
-
PromptTokenMeasurement,
|
|
13
|
-
Provider,
|
|
14
|
-
ProviderAttempt,
|
|
15
|
-
ProviderCostNormalizer,
|
|
16
|
-
ProviderCallKind,
|
|
17
|
-
ProviderGenerateArgs,
|
|
18
|
-
ProviderRequestAccounting,
|
|
19
|
-
ProviderRequestCapacity,
|
|
20
|
-
ProviderRequestSettlement,
|
|
21
|
-
ProviderResponse,
|
|
22
|
-
ProviderUsage,
|
|
23
|
-
ReasoningPolicy,
|
|
24
|
-
} from "./types.ts";
|
|
9
|
+
import type { ChatMessage, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderAttempt, ProviderCostNormalizer, ProviderGenerateArgs, ProviderRequestAccounting, ProviderRequestCapacity, ProviderRequestSettlement, ProviderResponse, ProviderUsage, ReasoningPolicy } from "./types.ts";
|
|
25
10
|
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
26
11
|
import { REASONING_POLICIES } from "@plurnk/plurnk-contracts";
|
|
27
12
|
import type { CallWarning, JSONValue } from "ai";
|
|
28
13
|
import { MAX_PROVIDER_TIMEOUT_MS, type Reasoning, type ReasoningResponseStyle } from "./env.ts";
|
|
29
14
|
import { UnsupportedReasoningPolicyError } from "./types.ts";
|
|
30
|
-
import {
|
|
31
|
-
|
|
32
|
-
executeOpenAICompatible,
|
|
33
|
-
transportFailureOutputObserved,
|
|
34
|
-
transportFailureEvidence,
|
|
35
|
-
} from "./aiSdkTransport.ts";
|
|
15
|
+
import type { InputModality } from "./types.ts";
|
|
16
|
+
import { executeAiSdkModel, executeOpenAICompatible, transportFailureOutputObserved, transportFailureEvidence } from "./aiSdkTransport.ts";
|
|
36
17
|
import type { LanguageModel } from "ai";
|
|
37
18
|
import { prepareRetries } from "ai/internal";
|
|
38
19
|
import { toProviderError, ProviderError, ProviderTimeoutError } from "./errors.ts";
|
|
39
20
|
import type { ProviderNotice } from "./notices.ts";
|
|
40
|
-
import { validateGbnf } from "@plurnk/gbnf";
|
|
41
21
|
import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.ts";
|
|
42
22
|
import { emitWarningOnce } from "./warnings.ts";
|
|
43
23
|
import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
|
|
@@ -45,9 +25,33 @@ import { resolveProviderCost } from "./cost.ts";
|
|
|
45
25
|
import { validateProviderRequestAccounting } from "./accounting.ts";
|
|
46
26
|
import { validateProviderUsage } from "./usage.ts";
|
|
47
27
|
import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.ts";
|
|
28
|
+
import { nativeFixedEffort } from "./reasoning-effort.ts";
|
|
29
|
+
import AiSdkRequestBody from "./AiSdkRequestBody.ts";
|
|
48
30
|
|
|
49
31
|
export type ProviderFetch = typeof globalThis.fetch;
|
|
50
32
|
|
|
33
|
+
// {§provider-connectivity} — an AbortSignal is advisory: a wedged transport that never observes it can
|
|
34
|
+
// hang the await past the deadline (#505). This backstops the deadline the same way an embedding request
|
|
35
|
+
// bounds a wedged adapter (#463): once the signal fires, a well-behaved transport unwinds and settles its
|
|
36
|
+
// own attempt (and accounting) within a short grace, and that path wins the race untouched; only a
|
|
37
|
+
// transport still wedged after the grace is force-rejected with the signal's own reason, so the existing
|
|
38
|
+
// operation/cancellation classification is unchanged. The grace is unwind slack, not a second deadline.
|
|
39
|
+
const OPERATION_DEADLINE_UNWIND_GRACE_MS = 1_000;
|
|
40
|
+
|
|
41
|
+
const raceAgainstDeadline = async <T>(work: PromiseLike<T>, signal: AbortSignal): Promise<T> => {
|
|
42
|
+
let timer: ReturnType<typeof setTimeout> | undefined;
|
|
43
|
+
const backstop = new Promise<never>((_resolve, reject) => {
|
|
44
|
+
const arm = (): void => { timer = setTimeout(() => reject(signal.reason), OPERATION_DEADLINE_UNWIND_GRACE_MS); };
|
|
45
|
+
if (signal.aborted) arm();
|
|
46
|
+
else signal.addEventListener("abort", arm, { once: true });
|
|
47
|
+
});
|
|
48
|
+
try {
|
|
49
|
+
return await Promise.race([work, backstop]);
|
|
50
|
+
} finally {
|
|
51
|
+
if (timer !== undefined) clearTimeout(timer);
|
|
52
|
+
}
|
|
53
|
+
};
|
|
54
|
+
|
|
51
55
|
// Backend wire spellings for the resolved reasoning intent. The switch beside each
|
|
52
56
|
// mapping retains any backend-specific omission/explicit-disable constraint.
|
|
53
57
|
export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "effort_required" | "thinking_effort" | "template" | "anthropic";
|
|
@@ -65,24 +69,7 @@ export type CacheAffinity =
|
|
|
65
69
|
|
|
66
70
|
export type AiSdkProviderOptions = Record<string, Record<string, JSONValue | undefined>>;
|
|
67
71
|
|
|
68
|
-
const isJsonObject = (value: JSONValue | undefined): value is Record<string, JSONValue> =>
|
|
69
|
-
typeof value === "object" && value !== null && !Array.isArray(value);
|
|
70
72
|
|
|
71
|
-
const mergeJsonObjects = (
|
|
72
|
-
left: Record<string, JSONValue | undefined>,
|
|
73
|
-
right: Record<string, JSONValue | undefined>,
|
|
74
|
-
): Record<string, JSONValue | undefined> => Object.fromEntries(
|
|
75
|
-
[...new Set([...Object.keys(left), ...Object.keys(right)])].map((key) => {
|
|
76
|
-
const leftValue = left[key];
|
|
77
|
-
const rightValue = right[key];
|
|
78
|
-
return [
|
|
79
|
-
key,
|
|
80
|
-
isJsonObject(leftValue) && isJsonObject(rightValue)
|
|
81
|
-
? mergeJsonObjects(leftValue, rightValue)
|
|
82
|
-
: rightValue ?? leftValue,
|
|
83
|
-
];
|
|
84
|
-
}),
|
|
85
|
-
);
|
|
86
73
|
|
|
87
74
|
export type AiSdkProviderConfig = {
|
|
88
75
|
model: string;
|
|
@@ -96,6 +83,7 @@ export type AiSdkProviderConfig = {
|
|
|
96
83
|
headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
|
|
97
84
|
fetch?: ProviderFetch; // per-instance request executor; default globalThis.fetch
|
|
98
85
|
contextWindow?: number | null; // default null; caller resolves-or-fails, narrows to required with the interface
|
|
86
|
+
inputModalities?: ReadonlySet<InputModality>; // {§provider-input-modalities} — native parts the route's model accepts; default none
|
|
99
87
|
maxInputTokens?: number | null;
|
|
100
88
|
maxOutputTokens?: number | null;
|
|
101
89
|
outputBudget?: number | null;
|
|
@@ -310,32 +298,6 @@ const projectTemplateReasoning = (content: string): TaggedReasoningProjection =>
|
|
|
310
298
|
return { content, reasoning: "", projected: false, contentStart: 0 };
|
|
311
299
|
};
|
|
312
300
|
|
|
313
|
-
const fixedEffort = (mode: ReasoningPolicy): "low" | "medium" | "high" | "xhigh" | "max" => {
|
|
314
|
-
if (mode === "low" || mode === "medium" || mode === "high" || mode === "xhigh" || mode === "max") return mode;
|
|
315
|
-
throw new TypeError(`reasoning policy '${mode}' is not a fixed effort`);
|
|
316
|
-
};
|
|
317
|
-
|
|
318
|
-
// The native SDK effort surface tops at xhigh; admission never grants a native
|
|
319
|
-
// route "max", so reaching it here is a contract violation, not a fallback site.
|
|
320
|
-
const nativeFixedEffort = (mode: ReasoningPolicy): "low" | "medium" | "high" | "xhigh" => {
|
|
321
|
-
const effort = fixedEffort(mode);
|
|
322
|
-
if (effort === "max") throw new TypeError(`reasoning policy 'max' has no native SDK effort surface`);
|
|
323
|
-
return effort;
|
|
324
|
-
};
|
|
325
|
-
|
|
326
|
-
// Anthropic's older manual-reasoning protocol needs an absolute allowance while
|
|
327
|
-
// PLURNK's durable contract names an effort. These fractions match the native
|
|
328
|
-
// SDK's policy projection, but apply to PLURNK's total envelope rather than the
|
|
329
|
-
// model's physical maximum. The minimum is imposed by the provider protocol.
|
|
330
|
-
const MANUAL_REASONING_FRACTIONS = Object.freeze({
|
|
331
|
-
adaptive: 0.6,
|
|
332
|
-
low: 0.1,
|
|
333
|
-
medium: 0.3,
|
|
334
|
-
high: 0.6,
|
|
335
|
-
xhigh: 0.75,
|
|
336
|
-
max: 0.85,
|
|
337
|
-
} satisfies Record<Exclude<ReasoningPolicy, "off">, number>);
|
|
338
|
-
const MANUAL_REASONING_MINIMUM = 1024;
|
|
339
301
|
|
|
340
302
|
const providerWarningMessage = (warning: CallWarning): string => {
|
|
341
303
|
switch (warning.type) {
|
|
@@ -347,27 +309,6 @@ const providerWarningMessage = (warning: CallWarning): string => {
|
|
|
347
309
|
}
|
|
348
310
|
};
|
|
349
311
|
|
|
350
|
-
// Body keys the provider owns — a caller's `sampling` passthrough may not set
|
|
351
|
-
// these. Two families:
|
|
352
|
-
// transport/managed — grammar transport, the stream/JSON choice, slot pinning,
|
|
353
|
-
// data capture ({§provider-evidence}: backend-specific fields never cross the contract);
|
|
354
|
-
// contract invariants — `n` (atomic single completion: choices[0] is the
|
|
355
|
-
// response; n>1 = paid, dropped output), the tool-calling family (tools-in-
|
|
356
|
-
// body doctrine, §2: native tool_calls return null content = a broken turn),
|
|
357
|
-
// modalities/audio (text-only contract), prediction (decode semantics, not
|
|
358
|
-
// sampling), and the token caps (the envelope is the managed maxOutputTokens —
|
|
359
|
-
// sampling must not bypass the consumer's cap).
|
|
360
|
-
// Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
|
|
361
|
-
// platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
|
|
362
|
-
// metadata, store, verbosity) pass through; the managed floors spread UNDER
|
|
363
|
-
// sampling stay deliberately caller-overridable.
|
|
364
|
-
const RESERVED_BODY_KEYS: ReadonlySet<string> = new Set([
|
|
365
|
-
"model", "messages", "stream", "stream_options", "grammar", "response_format", "id_slot", "logprobs", "top_logprobs",
|
|
366
|
-
"reasoning_format", "reasoning_effort", "thinking", "think", "include_reasoning", "chat_template_kwargs", "thinking_budget_tokens", // lexicon-allow: backend wire fields
|
|
367
|
-
"n", "tools", "tool_choice", "functions", "function_call", "parallel_tool_calls",
|
|
368
|
-
"modalities", "audio", "prediction", "max_tokens", "max_completion_tokens",
|
|
369
|
-
"prompt_cache_key",
|
|
370
|
-
]);
|
|
371
312
|
|
|
372
313
|
export default class AiSdkProvider implements Provider {
|
|
373
314
|
#model: string;
|
|
@@ -383,6 +324,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
383
324
|
#apiKeyRejectedMessage: string | undefined;
|
|
384
325
|
#eosText: string | undefined;
|
|
385
326
|
#contextWindow: number | null;
|
|
327
|
+
#inputModalities: ReadonlySet<InputModality>;
|
|
386
328
|
#maxInputTokens: number | null;
|
|
387
329
|
#maxOutputTokens: number | null;
|
|
388
330
|
#outputBudget: number | null;
|
|
@@ -433,6 +375,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
433
375
|
// tokenizeUrl (llama-server), so `provider.tokenize === undefined` remains
|
|
434
376
|
// the honest capability signal for every other backend.
|
|
435
377
|
tokenize?: (text: string) => Promise<number[]>;
|
|
378
|
+
readonly #requestBody: AiSdkRequestBody;
|
|
436
379
|
constructor(config: AiSdkProviderConfig) {
|
|
437
380
|
this.#model = config.model;
|
|
438
381
|
this.#url = config.url;
|
|
@@ -458,6 +401,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
458
401
|
this.#headers = config.headers ?? {};
|
|
459
402
|
this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
|
|
460
403
|
this.#contextWindow = config.contextWindow ?? null;
|
|
404
|
+
this.#inputModalities = config.inputModalities ?? new Set();
|
|
461
405
|
this.#maxInputTokens = config.maxInputTokens ?? null;
|
|
462
406
|
this.#maxOutputTokens = config.maxOutputTokens ?? null;
|
|
463
407
|
this.#outputBudget = config.outputBudget ?? null;
|
|
@@ -596,9 +540,11 @@ export default class AiSdkProvider implements Provider {
|
|
|
596
540
|
return tokens;
|
|
597
541
|
};
|
|
598
542
|
}
|
|
543
|
+
this.#requestBody = new AiSdkRequestBody({ reasoningBudget: this.#reasoningBudget, additiveReasoningProvider: this.#additiveReasoningProvider, reasoning: this.#reasoning, reasoningToggle: this.#reasoningToggle, compatibleAdaptiveReasoning: this.#compatibleAdaptiveReasoning, compatibleOffReasoning: this.#compatibleOffReasoning, adaptiveReasoningProviderOptions: this.#adaptiveReasoningProviderOptions, repeatPenalty: this.#repeatPenalty, frequencyPenalty: this.#frequencyPenalty, dryMultiplier: this.#dryMultiplier, dryBase: this.#dryBase, dryAllowedLength: this.#dryAllowedLength, repeatLastN: this.#repeatLastN, reasoningStyle: this.#reasoningStyle, source: this.#source, grammarStyle: this.#grammarStyle, cacheAffinity: this.#cacheAffinity, reasoningResponseProviderOptions: this.#reasoningResponseProviderOptions, firstPartyMetadata: this.#firstPartyMetadata, supportsSlotPinning: this.#supportsSlotPinning, slotCount: this.#slotCount });
|
|
599
544
|
}
|
|
600
545
|
|
|
601
546
|
get contextWindow(): number | null { return this.#contextWindow; }
|
|
547
|
+
get inputModalities(): ReadonlySet<InputModality> { return this.#inputModalities; }
|
|
602
548
|
get maxInputTokens(): number | null { return this.#maxInputTokens; }
|
|
603
549
|
get maxOutputTokens(): number | null { return this.#maxOutputTokens; }
|
|
604
550
|
get outputBudget(): number | null { return this.#outputBudget; }
|
|
@@ -648,7 +594,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
648
594
|
body: JSON.stringify({
|
|
649
595
|
model: this.#model,
|
|
650
596
|
messages,
|
|
651
|
-
...this.#reasoningBody(),
|
|
597
|
+
...this.#requestBody.reasoningBody(),
|
|
652
598
|
}),
|
|
653
599
|
...(requestSignal === undefined ? {} : { signal: requestSignal }),
|
|
654
600
|
});
|
|
@@ -703,204 +649,6 @@ export default class AiSdkProvider implements Provider {
|
|
|
703
649
|
measurement: await this.countPromptTokens(messages, signal),
|
|
704
650
|
});
|
|
705
651
|
}
|
|
706
|
-
// Reasoning activation and allowance are independent of grammar transport;
|
|
707
|
-
// only the response representation becomes lossless when evidence is needed.
|
|
708
|
-
// The llama-server template mapping is owned by {§llama-reasoning-request}.
|
|
709
|
-
#reasoningBody(
|
|
710
|
-
preserveGrammarSentence = false,
|
|
711
|
-
reasoningBudget = this.#reasoningBudget,
|
|
712
|
-
): Record<string, unknown> {
|
|
713
|
-
const { mode } = this.#reasoning;
|
|
714
|
-
const budget = reasoningBudget;
|
|
715
|
-
const on = mode !== "off";
|
|
716
|
-
switch (this.#reasoningStyle) {
|
|
717
|
-
case "template": {
|
|
718
|
-
const allowance = mode === "off"
|
|
719
|
-
? 0
|
|
720
|
-
: budget;
|
|
721
|
-
// A fixed effort rides into the template as its own variable; adaptive
|
|
722
|
-
// and off send none and leave the template's default in force.
|
|
723
|
-
const templateEffort = mode === "off" || mode === "adaptive" ? {} : { reasoning_effort: fixedEffort(mode) };
|
|
724
|
-
return {
|
|
725
|
-
chat_template_kwargs: { enable_thinking: on, ...templateEffort },
|
|
726
|
-
reasoning_format: preserveGrammarSentence ? "none" : "auto",
|
|
727
|
-
...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
|
|
728
|
-
};
|
|
729
|
-
}
|
|
730
|
-
case "think": return on ? { think: true } : {};
|
|
731
|
-
case "include_reasoning": return on ? { include_reasoning: true } : {};
|
|
732
|
-
case "effort": return mode === "off"
|
|
733
|
-
? this.#compatibleOffReasoning === undefined
|
|
734
|
-
? {}
|
|
735
|
-
: { reasoning_effort: this.#compatibleOffReasoning }
|
|
736
|
-
: mode === "adaptive"
|
|
737
|
-
? this.#compatibleAdaptiveReasoning === "provider-default"
|
|
738
|
-
? {}
|
|
739
|
-
: { reasoning_effort: this.#compatibleAdaptiveReasoning }
|
|
740
|
-
: { reasoning_effort: fixedEffort(mode) };
|
|
741
|
-
// Graded reasoning is mandatory when the route advertises an effort
|
|
742
|
-
// value. Cataloged routes supply the exact strongest legal value;
|
|
743
|
-
// construction rejects an unsupported off or fixed policy.
|
|
744
|
-
case "effort_required": {
|
|
745
|
-
if (mode === "off") {
|
|
746
|
-
if (this.#compatibleOffReasoning === undefined) {
|
|
747
|
-
throw new TypeError(`${this.#source}: required reasoning effort has no off projection`);
|
|
748
|
-
}
|
|
749
|
-
return { reasoning_effort: this.#compatibleOffReasoning };
|
|
750
|
-
}
|
|
751
|
-
if (mode === "adaptive") {
|
|
752
|
-
return this.#compatibleAdaptiveReasoning === "provider-default"
|
|
753
|
-
? {}
|
|
754
|
-
: { reasoning_effort: this.#compatibleAdaptiveReasoning };
|
|
755
|
-
}
|
|
756
|
-
return { reasoning_effort: fixedEffort(mode) };
|
|
757
|
-
}
|
|
758
|
-
// Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
|
|
759
|
-
// reason-by-default model (DeepSeek V4: default 'high') reasoning.
|
|
760
|
-
// ADAPTIVE omits the field UNLESS the catalog declares a toggle control:
|
|
761
|
-
// toggle routes (nemotron-lightning) default reasoning OFF, so adaptive
|
|
762
|
-
// sends the documented Fireworks Boolean enable (#457). The literal
|
|
763
|
-
// "adaptive" is MiniMax-M3-only — Fireworks 400s it for every other
|
|
764
|
-
// model (wire-verified; the 1.0.2 adaptive default refused to boot on
|
|
765
|
-
// it). V4 gotcha: integer efforts 400.
|
|
766
|
-
case "effort_explicit": return mode === "off"
|
|
767
|
-
? { reasoning_effort: "none" }
|
|
768
|
-
: mode === "adaptive"
|
|
769
|
-
? this.#reasoningToggle ? { reasoning_effort: true } : {}
|
|
770
|
-
: { reasoning_effort: fixedEffort(mode) };
|
|
771
|
-
// {§deepseek-reasoning-request}
|
|
772
|
-
case "thinking_effort": return mode === "off"
|
|
773
|
-
? { thinking: { type: "disabled" } }
|
|
774
|
-
: mode === "adaptive" ? { thinking: { type: "enabled" } } : {
|
|
775
|
-
thinking: { type: "enabled" },
|
|
776
|
-
reasoning_effort: fixedEffort(mode),
|
|
777
|
-
};
|
|
778
|
-
// Anthropic-compatible native dynamic or manual budget mode.
|
|
779
|
-
case "anthropic": return mode === "off"
|
|
780
|
-
? { thinking: { type: "disabled" } }
|
|
781
|
-
: mode === "adaptive" ? { thinking: { type: "adaptive" } } : {
|
|
782
|
-
thinking: {
|
|
783
|
-
type: "enabled",
|
|
784
|
-
budget_tokens: budget!,
|
|
785
|
-
},
|
|
786
|
-
};
|
|
787
|
-
case "none": return {};
|
|
788
|
-
}
|
|
789
|
-
}
|
|
790
|
-
|
|
791
|
-
// Per-worker slot affinity: the consumer passes which worker this is; the
|
|
792
|
-
// provider owns WHICH slot serves it. Sticky per workerId, round-robin across
|
|
793
|
-
// new runs (distinct runs → distinct slots while slots last), LRU-bounded
|
|
794
|
-
// bookkeeping so a long-lived daemon never grows the map unboundedly —
|
|
795
|
-
// an evicted-and-returning run simply re-pins, worst case one cold prefill.
|
|
796
|
-
#runSlots = new Map<string, number>();
|
|
797
|
-
#nextSlot = 0;
|
|
798
|
-
|
|
799
|
-
#slotBody(workerId: string): Record<string, unknown> {
|
|
800
|
-
if (!this.#supportsSlotPinning || this.#slotCount === null || this.#slotCount < 1) return {};
|
|
801
|
-
let slot = this.#runSlots.get(workerId);
|
|
802
|
-
if (slot === undefined) {
|
|
803
|
-
slot = this.#nextSlot++ % this.#slotCount;
|
|
804
|
-
if (this.#runSlots.size >= this.#slotCount * 8) {
|
|
805
|
-
this.#runSlots.delete(this.#runSlots.keys().next().value as string);
|
|
806
|
-
}
|
|
807
|
-
} else {
|
|
808
|
-
this.#runSlots.delete(workerId); // re-insert to refresh LRU recency
|
|
809
|
-
}
|
|
810
|
-
this.#runSlots.set(workerId, slot);
|
|
811
|
-
return { id_slot: slot };
|
|
812
|
-
}
|
|
813
|
-
|
|
814
|
-
// Optional local llama-server GBNF transport ({§gbnf-response-observation}). Unsupported
|
|
815
|
-
// backends receive no grammar-related field.
|
|
816
|
-
#grammarBody(grammar: string | undefined): Record<string, unknown> {
|
|
817
|
-
if (grammar === undefined) return {};
|
|
818
|
-
switch (this.#grammarStyle) {
|
|
819
|
-
// Grammar-constrained decoding can loop under the mask; a configured
|
|
820
|
-
// per-alias repeat_penalty is the measured remedy ({§provider-sampling-passthrough}).
|
|
821
|
-
case "llamacpp": return { grammar, ...(this.#repeatPenalty !== null ? { repeat_penalty: this.#repeatPenalty } : {}) };
|
|
822
|
-
case "none": return {};
|
|
823
|
-
}
|
|
824
|
-
}
|
|
825
|
-
|
|
826
|
-
// Anti-degeneration default on every request, keyed to the backend's wire
|
|
827
|
-
// convention - NOT grammar-bound. GBNF is a local constraint, so a cloud
|
|
828
|
-
// alias runs the sampler bare: firefast (deepseek/fireworks) ran 4/86 bench turns
|
|
829
|
-
// straight to the token cap on pure looped repetition (run52). Ships next to
|
|
830
|
-
// temperature so caller `sampling` can tune it; the grammar path re-asserts it as a
|
|
831
|
-
// managed FLOOR in #grammarBody. llama.cpp takes the repeat_penalty
|
|
832
|
-
// MULTIPLIER; the plain cloud path ("none") can't, so it gets
|
|
833
|
-
// frequency_penalty - OpenAI-standard, accepted by every OpenAI-compat backend (verified
|
|
834
|
-
// live: together/deepinfra/fireworks; it is OpenAI's own param, so real OpenAI takes it too).
|
|
835
|
-
#repetitionPenaltyBody(): Record<string, unknown> {
|
|
836
|
-
switch (this.#grammarStyle) {
|
|
837
|
-
// repeat_penalty + optional DRY (repeated-sequence penalty) + a wider
|
|
838
|
-
// repeat_last_n window — the loop-breaking tools a llama.cpp backend serves.
|
|
839
|
-
// Each rides only when its operator knob is set; absent = the box's default.
|
|
840
|
-
case "llamacpp": return {
|
|
841
|
-
...(this.#repeatPenalty !== null ? { repeat_penalty: this.#repeatPenalty } : {}),
|
|
842
|
-
...(this.#repeatLastN !== undefined ? { repeat_last_n: this.#repeatLastN } : {}),
|
|
843
|
-
...(this.#dryMultiplier !== undefined && this.#dryMultiplier > 0 ? {
|
|
844
|
-
dry_multiplier: this.#dryMultiplier,
|
|
845
|
-
...(this.#dryBase !== undefined ? { dry_base: this.#dryBase } : {}),
|
|
846
|
-
...(this.#dryAllowedLength !== undefined ? { dry_allowed_length: this.#dryAllowedLength } : {}),
|
|
847
|
-
} : {}),
|
|
848
|
-
};
|
|
849
|
-
case "none": return this.#frequencyPenalty > 0 ? { frequency_penalty: this.#frequencyPenalty } : {};
|
|
850
|
-
}
|
|
851
|
-
}
|
|
852
|
-
|
|
853
|
-
// First-party telemetry headers ({§provider-request-authority} {§provider-call-kind}): forwarded only when the spec
|
|
854
|
-
// opted in (the plurnk endpoint). The gate is here, not at the call site, so
|
|
855
|
-
// attributions/client/strikes can never reach a third-party backend even if
|
|
856
|
-
// the consumer passes them to the wrong provider. Empty values emit no header
|
|
857
|
-
// — EXCEPT strikes, where 0 is a real value (clean streak) distinct from
|
|
858
|
-
// absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
|
|
859
|
-
// ride HTTP headers only — the packet never carries them (the model must
|
|
860
|
-
// never see strike state; engine accounting is not a metric to game).
|
|
861
|
-
#metadataHeaders(attributions: string[] | undefined, client: string | undefined, strikes: number | undefined, workerId: string, primaryWorkerId: string | undefined, workspaceId: string | undefined, loop: number | undefined, turn: number | undefined, callKind: ProviderCallKind | undefined): Record<string, string> {
|
|
862
|
-
if (!this.#firstPartyMetadata) return {};
|
|
863
|
-
const h: Record<string, string> = {};
|
|
864
|
-
if (attributions !== undefined && attributions.length > 0) h["Plurnk-Attribution"] = JSON.stringify(attributions);
|
|
865
|
-
if (client !== undefined && client.length > 0) h["Plurnk-Client"] = client;
|
|
866
|
-
if (strikes !== undefined && Number.isInteger(strikes) && strikes >= 0) h["Plurnk-Strikes"] = String(strikes);
|
|
867
|
-
// Worker identity: the opaque workerId
|
|
868
|
-
// the consumer already supplies, forwarded so the endpoint can key
|
|
869
|
-
// per-worker affinity/telemetry — same gate as every first-party signal.
|
|
870
|
-
h["Plurnk-Worker-Id"] = workerId;
|
|
871
|
-
// Root worker of the lineage ({§worker-primary}): the no-parent ancestor of this turn's
|
|
872
|
-
// worker tree. The consumer classifies primary-vs-spawned by equality
|
|
873
|
-
// (primaryWorkerId == workerId ⇒ the primary/root worker). The provider
|
|
874
|
-
// EMITS what the consumer supplies and never invents a primary; the
|
|
875
|
-
// consumer's contract is to stamp it EVERY turn (including the primary's
|
|
876
|
-
// own, where it equals workerId). Absence is the consumer's violation for
|
|
877
|
-
// the endpoint to surface, not a provider default.
|
|
878
|
-
if (primaryWorkerId !== undefined && primaryWorkerId.length > 0) h["Plurnk-Worker-Primary"] = primaryWorkerId;
|
|
879
|
-
// Turn coordinate ({§lifecycle-terms}): workspace/loop/turn, the
|
|
880
|
-
// daemon-side sequence the endpoint can never scrape from the wire.
|
|
881
|
-
// Coordinates are 1-based — 0 is not a real value, so no strikes-style
|
|
882
|
-
// zero exception; absent/empty/0 emits no header.
|
|
883
|
-
if (workspaceId !== undefined && workspaceId.length > 0) h["Plurnk-Workspace-Id"] = workspaceId;
|
|
884
|
-
if (loop !== undefined && Number.isInteger(loop) && loop >= 1) h["Plurnk-Loop"] = String(loop);
|
|
885
|
-
if (turn !== undefined && Number.isInteger(turn) && turn >= 1) h["Plurnk-Turn"] = String(turn);
|
|
886
|
-
if (callKind !== undefined) h["Plurnk-Call-Kind"] = callKind;
|
|
887
|
-
return h;
|
|
888
|
-
}
|
|
889
|
-
|
|
890
|
-
// PLURNK_PROVIDERS_GBNF_DEBUG ({§gbnf-response-observation}): validate the supplied GBNF locally and fail
|
|
891
|
-
// hard if it's malformed, BEFORE any wire call — and the grammar is NOT
|
|
892
|
-
// transported, so the request runs unconstrained. A debug aid to catch invalid
|
|
893
|
-
// grammars (e.g. while editing the plurnk grammar) without a model round-trip;
|
|
894
|
-
// off in production. `validateGbnf(grammar, "")` parses the grammar + resolves
|
|
895
|
-
// its root, throwing iff the grammar itself is invalid (the empty input's
|
|
896
|
-
// verdict is irrelevant — we only care that parsing succeeded).
|
|
897
|
-
#assertGrammarValid(grammar: string): void {
|
|
898
|
-
try {
|
|
899
|
-
validateGbnf(grammar, "");
|
|
900
|
-
} catch (cause) {
|
|
901
|
-
throw new Error(`grammar validation (PLURNK_PROVIDERS_GBNF_DEBUG): invalid GBNF — ${(cause as Error).message}`, { cause });
|
|
902
|
-
}
|
|
903
|
-
}
|
|
904
652
|
|
|
905
653
|
// Per-turn metadata bag: pass the backend's non-standard top-level fields
|
|
906
654
|
// through verbatim. Providers do not reinterpret vendor currency or account
|
|
@@ -910,84 +658,6 @@ export default class AiSdkProvider implements Provider {
|
|
|
910
658
|
return Object.keys(meta).length > 0 ? meta : undefined;
|
|
911
659
|
}
|
|
912
660
|
|
|
913
|
-
// Caller-supplied OpenAI-compat sampling params (temperature, top_p, top_k,
|
|
914
|
-
// penalties, stop, seed, …) merged UNDER the managed body: model, messages,
|
|
915
|
-
// reasoning, grammar (+ its repeat-penalty floor), max_tokens and slot always
|
|
916
|
-
// win, and reserved transport/protocol keys are stripped so the passthrough
|
|
917
|
-
// can't smuggle a grammar, a stream toggle, or a backend slot
|
|
918
|
-
// ({§provider-request-authority}).
|
|
919
|
-
#samplingBody(sampling: Record<string, unknown> | undefined): Record<string, unknown> {
|
|
920
|
-
if (sampling === undefined) return {};
|
|
921
|
-
const out: Record<string, unknown> = {};
|
|
922
|
-
for (const [k, v] of Object.entries(sampling)) if (!RESERVED_BODY_KEYS.has(k)) out[k] = v;
|
|
923
|
-
return out;
|
|
924
|
-
}
|
|
925
|
-
|
|
926
|
-
#requestProviderOptions(
|
|
927
|
-
workerId: string,
|
|
928
|
-
nativeReasoningBudget: number | null,
|
|
929
|
-
): AiSdkProviderOptions | undefined {
|
|
930
|
-
const responseOptions = this.#reasoning.mode === "off"
|
|
931
|
-
? undefined
|
|
932
|
-
: this.#reasoningResponseProviderOptions;
|
|
933
|
-
const adaptiveOptions = this.#reasoning.mode === "adaptive"
|
|
934
|
-
&& nativeReasoningBudget === null
|
|
935
|
-
? this.#adaptiveReasoningProviderOptions
|
|
936
|
-
: undefined;
|
|
937
|
-
const nativeReasoning = nativeReasoningBudget !== null
|
|
938
|
-
? this.#additiveReasoningProvider === "anthropic"
|
|
939
|
-
? { anthropic: { thinking: { type: "enabled", budgetTokens: nativeReasoningBudget } } }
|
|
940
|
-
: this.#additiveReasoningProvider === "bedrock"
|
|
941
|
-
? { bedrock: { reasoningConfig: { type: "enabled", budgetTokens: nativeReasoningBudget } } }
|
|
942
|
-
: undefined
|
|
943
|
-
: undefined;
|
|
944
|
-
const options: AiSdkProviderOptions = {};
|
|
945
|
-
for (const part of [responseOptions, adaptiveOptions, nativeReasoning]) {
|
|
946
|
-
for (const [provider, values] of Object.entries(part ?? {})) {
|
|
947
|
-
options[provider] = mergeJsonObjects(options[provider] ?? {}, values);
|
|
948
|
-
}
|
|
949
|
-
}
|
|
950
|
-
if (this.#cacheAffinity?.target === "provider-option") {
|
|
951
|
-
const { provider, name } = this.#cacheAffinity;
|
|
952
|
-
options[provider] = { ...options[provider], [name]: workerId };
|
|
953
|
-
}
|
|
954
|
-
return Object.keys(options).length === 0 ? undefined : options;
|
|
955
|
-
}
|
|
956
|
-
|
|
957
|
-
#nativeMaxOutputTokens(
|
|
958
|
-
outputBudget: number | null,
|
|
959
|
-
nativeReasoningBudget: number | null,
|
|
960
|
-
): number | undefined {
|
|
961
|
-
if (outputBudget === null) return undefined;
|
|
962
|
-
return nativeReasoningBudget !== null
|
|
963
|
-
? outputBudget - nativeReasoningBudget
|
|
964
|
-
: outputBudget;
|
|
965
|
-
}
|
|
966
|
-
|
|
967
|
-
#nativeReasoningBudget(
|
|
968
|
-
outputBudget: number | null,
|
|
969
|
-
configuredReasoningBudget: number | null,
|
|
970
|
-
): number | null {
|
|
971
|
-
if (this.#additiveReasoningProvider === undefined || this.#reasoning.mode === "off") return null;
|
|
972
|
-
if (configuredReasoningBudget !== null) return configuredReasoningBudget;
|
|
973
|
-
if (this.#adaptiveReasoningProviderOptions !== undefined) return null;
|
|
974
|
-
if (outputBudget === null) {
|
|
975
|
-
throw new TypeError(
|
|
976
|
-
`${this.#source}: manual provider reasoning requires a resolved total output budget`,
|
|
977
|
-
);
|
|
978
|
-
}
|
|
979
|
-
if (outputBudget <= MANUAL_REASONING_MINIMUM) {
|
|
980
|
-
throw new TypeError(
|
|
981
|
-
`${this.#source}: total output budget must exceed the provider's ${MANUAL_REASONING_MINIMUM}-token minimum reasoning allowance`,
|
|
982
|
-
);
|
|
983
|
-
}
|
|
984
|
-
const fraction = MANUAL_REASONING_FRACTIONS[this.#reasoning.mode];
|
|
985
|
-
return Math.min(
|
|
986
|
-
outputBudget - 1,
|
|
987
|
-
Math.max(MANUAL_REASONING_MINIMUM, Math.round(outputBudget * fraction)),
|
|
988
|
-
);
|
|
989
|
-
}
|
|
990
|
-
|
|
991
661
|
#accounting(
|
|
992
662
|
outcome: ProviderRequestAccounting["outcome"],
|
|
993
663
|
usage: ProviderUsage | undefined,
|
|
@@ -1019,7 +689,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
1019
689
|
// Grammar handling ({§gbnf-response-observation}). Debug validates the
|
|
1020
690
|
// supplied grammar before the call but withholds it from the backend.
|
|
1021
691
|
const wantGrammar = grammar !== undefined && this.#grammarStyle !== "none";
|
|
1022
|
-
if (wantGrammar && this.#gbnfDebug) this.#assertGrammarValid(grammar!);
|
|
692
|
+
if (wantGrammar && this.#gbnfDebug) this.#requestBody.assertGrammarValid(grammar!);
|
|
1023
693
|
const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
|
|
1024
694
|
const preserveGrammarSentence = wantGrammar
|
|
1025
695
|
&& this.#reasoningStyle === "template";
|
|
@@ -1039,7 +709,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
1039
709
|
// {§provider-flexed-allowance} (#482): the wire grants the flexed
|
|
1040
710
|
// allowance — the floor, or the exactly-measured slack above it.
|
|
1041
711
|
const effectiveMaxOutputTokens = capacity.responseMax ?? capacity.outputBudget ?? undefined;
|
|
1042
|
-
const nativeReasoningBudget = this.#nativeReasoningBudget(
|
|
712
|
+
const nativeReasoningBudget = this.#requestBody.nativeReasoningBudget(
|
|
1043
713
|
capacity.outputBudget,
|
|
1044
714
|
capacity.reasoningBudget,
|
|
1045
715
|
);
|
|
@@ -1051,25 +721,25 @@ export default class AiSdkProvider implements Provider {
|
|
|
1051
721
|
const body: Record<string, unknown> = {
|
|
1052
722
|
// Floors are suppressed on router-owned-tuning providers (plurnk) —
|
|
1053
723
|
// the router's per-model tuning must not be overridden by client floors.
|
|
1054
|
-
...(this.#tuningFloors ? { ...(this.#temperature !== null ? { temperature: this.#temperature } : {}), ...this.#repetitionPenaltyBody() } : {}),
|
|
1055
|
-
...this.#samplingBody(sampling),
|
|
724
|
+
...(this.#tuningFloors ? { ...(this.#temperature !== null ? { temperature: this.#temperature } : {}), ...this.#requestBody.repetitionPenaltyBody() } : {}),
|
|
725
|
+
...this.#requestBody.samplingBody(sampling),
|
|
1056
726
|
...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
|
|
1057
727
|
model: this.#model,
|
|
1058
728
|
messages,
|
|
1059
|
-
...this.#reasoningBody(preserveGrammarSentence, capacity.reasoningBudget),
|
|
1060
|
-
...this.#grammarBody(sendGrammar),
|
|
729
|
+
...this.#requestBody.reasoningBody(preserveGrammarSentence, capacity.reasoningBudget),
|
|
730
|
+
...this.#requestBody.grammarBody(sendGrammar),
|
|
1061
731
|
...(effectiveMaxOutputTokens !== undefined ? { max_tokens: effectiveMaxOutputTokens } : {}),
|
|
1062
732
|
// Request per-token logprobs only when enabled (managed field —
|
|
1063
733
|
// reserved from caller sampling; the env flag is the single control).
|
|
1064
734
|
...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
|
|
1065
|
-
...this.#slotBody(workerId),
|
|
735
|
+
...this.#requestBody.slotBody(workerId),
|
|
1066
736
|
...(this.#cacheAffinity?.target === "body"
|
|
1067
737
|
? { [this.#cacheAffinity.name]: workerId }
|
|
1068
738
|
: {}),
|
|
1069
739
|
};
|
|
1070
740
|
|
|
1071
741
|
// Per-request headers = static auth/routing + any first-party telemetry.
|
|
1072
|
-
const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind);
|
|
742
|
+
const metaHeaders = this.#requestBody.metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind);
|
|
1073
743
|
const headers = new Headers(this.#headers);
|
|
1074
744
|
if (this.#cacheAffinity?.target === "header") {
|
|
1075
745
|
headers.set(this.#cacheAffinity.name, workerId);
|
|
@@ -1176,7 +846,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
1176
846
|
: await executeAiSdkModel({
|
|
1177
847
|
languageModel: this.#languageModel,
|
|
1178
848
|
headers: requestHeaders,
|
|
1179
|
-
providerOptions: this.#requestProviderOptions(workerId, nativeReasoningBudget),
|
|
849
|
+
providerOptions: this.#requestBody.requestProviderOptions(workerId, nativeReasoningBudget),
|
|
1180
850
|
systemProviderOptions: this.#systemCacheProviderOptions,
|
|
1181
851
|
messages,
|
|
1182
852
|
signal: operationSignal,
|
|
@@ -1201,7 +871,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
1201
871
|
? sampling.stop
|
|
1202
872
|
: undefined,
|
|
1203
873
|
seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
|
|
1204
|
-
maxOutputTokens: this.#nativeMaxOutputTokens(capacity.outputBudget, nativeReasoningBudget),
|
|
874
|
+
maxOutputTokens: this.#requestBody.nativeMaxOutputTokens(capacity.outputBudget, nativeReasoningBudget),
|
|
1205
875
|
reasoning: this.#reasoning.mode === "off"
|
|
1206
876
|
? "none"
|
|
1207
877
|
: this.#reasoning.mode === "adaptive"
|
|
@@ -1234,7 +904,13 @@ export default class AiSdkProvider implements Provider {
|
|
|
1234
904
|
maxRetries: this.#retryAttempts,
|
|
1235
905
|
abortSignal: operationSignal,
|
|
1236
906
|
});
|
|
1237
|
-
|
|
907
|
+
// {§provider-connectivity} — the operation deadline and caller cancellation are backstopped by
|
|
908
|
+
// racing, not only by the advisory operationSignal, so a transport that ignores the abort still
|
|
909
|
+
// surfaces the deadline (→ the operationTimeout/caller branches below) rather than hanging the
|
|
910
|
+
// loop (#505). A well-behaved transport settles first and wins the race with its own evidence.
|
|
911
|
+
raw = operationSignal === undefined
|
|
912
|
+
? await retry(executeRequest)
|
|
913
|
+
: await raceAgainstDeadline(retry(executeRequest), operationSignal);
|
|
1238
914
|
} catch (err) {
|
|
1239
915
|
if (err instanceof ProviderRequestObserverError
|
|
1240
916
|
|| err instanceof ProviderRequestAccountingError
|